feat: support rich content in long-form tweets (X Articles)
Add support for extracting rich content from X's long-form tweets, including embedded code snippets, markdown blocks, quoted tweets, and other structured content that was previously lost. Changes: - Add fieldToggles with withArticleRichContentState to TweetDetail API request - Implement Draft.js content_state parser (renderContentState) that converts blocks and entities to readable markdown format - Add content_state type definition to GraphqlTweetResult Supported content: - Block types: paragraphs, headers, ordered/unordered lists, blockquotes - Entity types: MARKDOWN (code blocks), DIVIDER, TWEET, LINK, IMAGE No impact on regular tweets - rich content only adds payload when present. Includes 20 unit tests for the parser and an opt-in live smoke test. Co-authored-by: Christian Catalan <[email protected]>
This commit is contained in:
committed by
Peter Steinberger
co-authored by
Christian Catalan
parent
e0b960db1a
commit
316cdf77be
@@ -177,9 +177,15 @@ export function withTweetDetails<TBase extends AbstractConstructor<TwitterClient
|
||||
rweb_video_timestamps_enabled: true,
|
||||
};
|
||||
|
||||
const fieldToggles = {
|
||||
...buildArticleFieldToggles(),
|
||||
withArticleRichContentState: true,
|
||||
};
|
||||
|
||||
const params = new URLSearchParams({
|
||||
variables: JSON.stringify(variables),
|
||||
features: JSON.stringify(features),
|
||||
fieldToggles: JSON.stringify(fieldToggles),
|
||||
});
|
||||
|
||||
try {
|
||||
|
||||
@@ -130,6 +130,25 @@ export type GraphqlTweetResult = {
|
||||
};
|
||||
}>;
|
||||
}>;
|
||||
/** Draft.js content state for rich article content */
|
||||
content_state?: {
|
||||
blocks: Array<{
|
||||
key: string;
|
||||
type: string;
|
||||
text: string;
|
||||
data?: Record<string, unknown>;
|
||||
entityRanges: Array<{ key: number; offset: number; length: number }>;
|
||||
inlineStyleRanges: Array<{ offset: number; length: number; style: string }>;
|
||||
}>;
|
||||
entityMap: Array<{
|
||||
key: string;
|
||||
value: {
|
||||
type: string;
|
||||
mutability: string;
|
||||
data: Record<string, unknown>;
|
||||
};
|
||||
}>;
|
||||
};
|
||||
};
|
||||
};
|
||||
plain_text?: string;
|
||||
|
||||
@@ -66,6 +66,242 @@ export function uniqueOrdered(values: string[]): string[] {
|
||||
return result;
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Draft.js Content State Types and Parser for Long-form Tweets (X Articles)
|
||||
// ============================================================================
|
||||
|
||||
/** Inline style range for text formatting (Bold, Italic, etc.) */
|
||||
interface InlineStyleRange {
|
||||
offset: number;
|
||||
length: number;
|
||||
style: string;
|
||||
}
|
||||
|
||||
/** Entity range linking a portion of text to an entity in entityMap */
|
||||
interface EntityRange {
|
||||
key: number;
|
||||
offset: number;
|
||||
length: number;
|
||||
}
|
||||
|
||||
/** A content block in Draft.js format */
|
||||
interface ContentBlock {
|
||||
key: string;
|
||||
type: string;
|
||||
text: string;
|
||||
data?: {
|
||||
mentions?: Array<{ fromIndex: number; toIndex: number; text: string }>;
|
||||
};
|
||||
entityRanges: EntityRange[];
|
||||
inlineStyleRanges: InlineStyleRange[];
|
||||
}
|
||||
|
||||
/** Entity data for different entity types */
|
||||
interface EntityValue {
|
||||
type: string;
|
||||
mutability: string;
|
||||
data: {
|
||||
markdown?: string;
|
||||
url?: string;
|
||||
tweetId?: string;
|
||||
};
|
||||
}
|
||||
|
||||
/** Entity map entry */
|
||||
interface EntityMapEntry {
|
||||
key: string;
|
||||
value: EntityValue;
|
||||
}
|
||||
|
||||
/** Draft.js content state structure */
|
||||
interface ContentState {
|
||||
blocks: ContentBlock[];
|
||||
entityMap: EntityMapEntry[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Renders a Draft.js content_state into readable markdown/text format.
|
||||
* Handles blocks (paragraphs, headers, lists) and entities (code blocks, links, tweets, dividers).
|
||||
*/
|
||||
export function renderContentState(contentState: ContentState | undefined): string | undefined {
|
||||
if (!contentState?.blocks || contentState.blocks.length === 0) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
// Build entity lookup map from array format
|
||||
const entityMap = new Map<number, EntityValue>();
|
||||
for (const entry of contentState.entityMap ?? []) {
|
||||
const key = Number.parseInt(entry.key, 10);
|
||||
if (!Number.isNaN(key)) {
|
||||
entityMap.set(key, entry.value);
|
||||
}
|
||||
}
|
||||
|
||||
const outputLines: string[] = [];
|
||||
let orderedListCounter = 0;
|
||||
let previousBlockType: string | undefined;
|
||||
|
||||
for (const block of contentState.blocks) {
|
||||
// Reset ordered list counter when leaving ordered list context
|
||||
if (block.type !== 'ordered-list-item' && previousBlockType === 'ordered-list-item') {
|
||||
orderedListCounter = 0;
|
||||
}
|
||||
|
||||
switch (block.type) {
|
||||
case 'unstyled': {
|
||||
// Plain paragraph - just output text with any inline formatting
|
||||
const text = renderBlockText(block, entityMap);
|
||||
if (text) {
|
||||
outputLines.push(text);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 'header-one': {
|
||||
const text = renderBlockText(block, entityMap);
|
||||
if (text) {
|
||||
outputLines.push(`# ${text}`);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 'header-two': {
|
||||
const text = renderBlockText(block, entityMap);
|
||||
if (text) {
|
||||
outputLines.push(`## ${text}`);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 'header-three': {
|
||||
const text = renderBlockText(block, entityMap);
|
||||
if (text) {
|
||||
outputLines.push(`### ${text}`);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 'unordered-list-item': {
|
||||
const text = renderBlockText(block, entityMap);
|
||||
if (text) {
|
||||
outputLines.push(`- ${text}`);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 'ordered-list-item': {
|
||||
orderedListCounter++;
|
||||
const text = renderBlockText(block, entityMap);
|
||||
if (text) {
|
||||
outputLines.push(`${orderedListCounter}. ${text}`);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 'blockquote': {
|
||||
const text = renderBlockText(block, entityMap);
|
||||
if (text) {
|
||||
outputLines.push(`> ${text}`);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case 'atomic': {
|
||||
// Atomic blocks are placeholders for embedded entities
|
||||
const entityContent = renderAtomicBlock(block, entityMap);
|
||||
if (entityContent) {
|
||||
outputLines.push(entityContent);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
default: {
|
||||
// Fallback: just output the text
|
||||
const text = renderBlockText(block, entityMap);
|
||||
if (text) {
|
||||
outputLines.push(text);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
previousBlockType = block.type;
|
||||
}
|
||||
|
||||
const result = outputLines.join('\n\n');
|
||||
return result.trim() || undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Renders text content of a block, applying inline link entities.
|
||||
*/
|
||||
function renderBlockText(block: ContentBlock, entityMap: Map<number, EntityValue>): string {
|
||||
let text = block.text;
|
||||
|
||||
// Handle LINK entities by appending URL in markdown format
|
||||
// Process in reverse order to not mess up offsets
|
||||
const linkRanges = block.entityRanges
|
||||
.filter((range) => {
|
||||
const entity = entityMap.get(range.key);
|
||||
return entity?.type === 'LINK' && entity.data.url;
|
||||
})
|
||||
.sort((a, b) => b.offset - a.offset);
|
||||
|
||||
for (const range of linkRanges) {
|
||||
const entity = entityMap.get(range.key);
|
||||
if (entity?.data.url) {
|
||||
const linkText = text.slice(range.offset, range.offset + range.length);
|
||||
const markdownLink = `[${linkText}](${entity.data.url})`;
|
||||
text = text.slice(0, range.offset) + markdownLink + text.slice(range.offset + range.length);
|
||||
}
|
||||
}
|
||||
|
||||
return text.trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* Renders an atomic block by looking up its entity and returning appropriate content.
|
||||
*/
|
||||
function renderAtomicBlock(block: ContentBlock, entityMap: Map<number, EntityValue>): string | undefined {
|
||||
if (block.entityRanges.length === 0) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const entityKey = block.entityRanges[0].key;
|
||||
const entity = entityMap.get(entityKey);
|
||||
|
||||
if (!entity) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
switch (entity.type) {
|
||||
case 'MARKDOWN':
|
||||
// Code blocks and other markdown content - output as-is
|
||||
return entity.data.markdown?.trim();
|
||||
|
||||
case 'DIVIDER':
|
||||
return '---';
|
||||
|
||||
case 'TWEET':
|
||||
if (entity.data.tweetId) {
|
||||
return `[Embedded Tweet: https://x.com/i/status/${entity.data.tweetId}]`;
|
||||
}
|
||||
return undefined;
|
||||
|
||||
case 'LINK':
|
||||
if (entity.data.url) {
|
||||
return `[Link: ${entity.data.url}]`;
|
||||
}
|
||||
return undefined;
|
||||
|
||||
case 'IMAGE':
|
||||
// Images in atomic blocks - could extract URL if available
|
||||
return '[Image]';
|
||||
|
||||
default:
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
export function extractArticleText(result: GraphqlTweetResult | undefined): string | undefined {
|
||||
const article = result?.article;
|
||||
if (!article) {
|
||||
@@ -87,7 +323,22 @@ export function extractArticleText(result: GraphqlTweetResult | undefined): stri
|
||||
),
|
||||
);
|
||||
}
|
||||
|
||||
const title = firstText(articleResult.title, article.title);
|
||||
|
||||
// Try to render from rich content_state first (Draft.js format with blocks + entityMap)
|
||||
// This preserves code blocks, embedded tweets, markdown, etc.
|
||||
const contentState = article.article_results?.result?.content_state;
|
||||
const richBody = renderContentState(contentState);
|
||||
if (richBody) {
|
||||
// Rich content found - prepend title if not already included
|
||||
if (title && !richBody.startsWith(title)) {
|
||||
return `${title}\n\n${richBody}`;
|
||||
}
|
||||
return richBody;
|
||||
}
|
||||
|
||||
// Fallback to plain text extraction for articles without rich content_state
|
||||
let body = firstText(
|
||||
articleResult.plain_text,
|
||||
article.plain_text,
|
||||
|
||||
Reference in New Issue
Block a user