From 316cdf77be1f0c25159238b05f20cd2fb8a7cd71 Mon Sep 17 00:00:00 2001 From: cc-vps Date: Wed, 7 Jan 2026 10:50:56 -0800 Subject: [PATCH] feat: support rich content in long-form tweets (X Articles) Add support for extracting rich content from X's long-form tweets, including embedded code snippets, markdown blocks, quoted tweets, and other structured content that was previously lost. Changes: - Add fieldToggles with withArticleRichContentState to TweetDetail API request - Implement Draft.js content_state parser (renderContentState) that converts blocks and entities to readable markdown format - Add content_state type definition to GraphqlTweetResult Supported content: - Block types: paragraphs, headers, ordered/unordered lists, blockquotes - Entity types: MARKDOWN (code blocks), DIVIDER, TWEET, LINK, IMAGE No impact on regular tweets - rich content only adds payload when present. Includes 20 unit tests for the parser and an opt-in live smoke test. Co-authored-by: Christian Catalan --- src/lib/twitter-client-tweet-detail.ts | 6 + src/lib/twitter-client-types.ts | 19 ++ src/lib/twitter-client-utils.ts | 251 +++++++++++++++++++ tests/live/live.test.ts | 17 ++ tests/render-content-state.test.ts | 322 +++++++++++++++++++++++++ 5 files changed, 615 insertions(+) create mode 100644 tests/render-content-state.test.ts diff --git a/src/lib/twitter-client-tweet-detail.ts b/src/lib/twitter-client-tweet-detail.ts index a17881e..c9cb016 100644 --- a/src/lib/twitter-client-tweet-detail.ts +++ b/src/lib/twitter-client-tweet-detail.ts @@ -177,9 +177,15 @@ export function withTweetDetails; }>; + /** Draft.js content state for rich article content */ + content_state?: { + blocks: Array<{ + key: string; + type: string; + text: string; + data?: Record; + entityRanges: Array<{ key: number; offset: number; length: number }>; + inlineStyleRanges: Array<{ offset: number; length: number; style: string }>; + }>; + entityMap: Array<{ + key: string; + value: { + type: string; + mutability: string; + data: Record; + }; + }>; + }; }; }; plain_text?: string; diff --git a/src/lib/twitter-client-utils.ts b/src/lib/twitter-client-utils.ts index daf049a..35b4280 100644 --- a/src/lib/twitter-client-utils.ts +++ b/src/lib/twitter-client-utils.ts @@ -66,6 +66,242 @@ export function uniqueOrdered(values: string[]): string[] { return result; } +// ============================================================================ +// Draft.js Content State Types and Parser for Long-form Tweets (X Articles) +// ============================================================================ + +/** Inline style range for text formatting (Bold, Italic, etc.) */ +interface InlineStyleRange { + offset: number; + length: number; + style: string; +} + +/** Entity range linking a portion of text to an entity in entityMap */ +interface EntityRange { + key: number; + offset: number; + length: number; +} + +/** A content block in Draft.js format */ +interface ContentBlock { + key: string; + type: string; + text: string; + data?: { + mentions?: Array<{ fromIndex: number; toIndex: number; text: string }>; + }; + entityRanges: EntityRange[]; + inlineStyleRanges: InlineStyleRange[]; +} + +/** Entity data for different entity types */ +interface EntityValue { + type: string; + mutability: string; + data: { + markdown?: string; + url?: string; + tweetId?: string; + }; +} + +/** Entity map entry */ +interface EntityMapEntry { + key: string; + value: EntityValue; +} + +/** Draft.js content state structure */ +interface ContentState { + blocks: ContentBlock[]; + entityMap: EntityMapEntry[]; +} + +/** + * Renders a Draft.js content_state into readable markdown/text format. + * Handles blocks (paragraphs, headers, lists) and entities (code blocks, links, tweets, dividers). + */ +export function renderContentState(contentState: ContentState | undefined): string | undefined { + if (!contentState?.blocks || contentState.blocks.length === 0) { + return undefined; + } + + // Build entity lookup map from array format + const entityMap = new Map(); + for (const entry of contentState.entityMap ?? []) { + const key = Number.parseInt(entry.key, 10); + if (!Number.isNaN(key)) { + entityMap.set(key, entry.value); + } + } + + const outputLines: string[] = []; + let orderedListCounter = 0; + let previousBlockType: string | undefined; + + for (const block of contentState.blocks) { + // Reset ordered list counter when leaving ordered list context + if (block.type !== 'ordered-list-item' && previousBlockType === 'ordered-list-item') { + orderedListCounter = 0; + } + + switch (block.type) { + case 'unstyled': { + // Plain paragraph - just output text with any inline formatting + const text = renderBlockText(block, entityMap); + if (text) { + outputLines.push(text); + } + break; + } + + case 'header-one': { + const text = renderBlockText(block, entityMap); + if (text) { + outputLines.push(`# ${text}`); + } + break; + } + + case 'header-two': { + const text = renderBlockText(block, entityMap); + if (text) { + outputLines.push(`## ${text}`); + } + break; + } + + case 'header-three': { + const text = renderBlockText(block, entityMap); + if (text) { + outputLines.push(`### ${text}`); + } + break; + } + + case 'unordered-list-item': { + const text = renderBlockText(block, entityMap); + if (text) { + outputLines.push(`- ${text}`); + } + break; + } + + case 'ordered-list-item': { + orderedListCounter++; + const text = renderBlockText(block, entityMap); + if (text) { + outputLines.push(`${orderedListCounter}. ${text}`); + } + break; + } + + case 'blockquote': { + const text = renderBlockText(block, entityMap); + if (text) { + outputLines.push(`> ${text}`); + } + break; + } + + case 'atomic': { + // Atomic blocks are placeholders for embedded entities + const entityContent = renderAtomicBlock(block, entityMap); + if (entityContent) { + outputLines.push(entityContent); + } + break; + } + + default: { + // Fallback: just output the text + const text = renderBlockText(block, entityMap); + if (text) { + outputLines.push(text); + } + } + } + + previousBlockType = block.type; + } + + const result = outputLines.join('\n\n'); + return result.trim() || undefined; +} + +/** + * Renders text content of a block, applying inline link entities. + */ +function renderBlockText(block: ContentBlock, entityMap: Map): string { + let text = block.text; + + // Handle LINK entities by appending URL in markdown format + // Process in reverse order to not mess up offsets + const linkRanges = block.entityRanges + .filter((range) => { + const entity = entityMap.get(range.key); + return entity?.type === 'LINK' && entity.data.url; + }) + .sort((a, b) => b.offset - a.offset); + + for (const range of linkRanges) { + const entity = entityMap.get(range.key); + if (entity?.data.url) { + const linkText = text.slice(range.offset, range.offset + range.length); + const markdownLink = `[${linkText}](${entity.data.url})`; + text = text.slice(0, range.offset) + markdownLink + text.slice(range.offset + range.length); + } + } + + return text.trim(); +} + +/** + * Renders an atomic block by looking up its entity and returning appropriate content. + */ +function renderAtomicBlock(block: ContentBlock, entityMap: Map): string | undefined { + if (block.entityRanges.length === 0) { + return undefined; + } + + const entityKey = block.entityRanges[0].key; + const entity = entityMap.get(entityKey); + + if (!entity) { + return undefined; + } + + switch (entity.type) { + case 'MARKDOWN': + // Code blocks and other markdown content - output as-is + return entity.data.markdown?.trim(); + + case 'DIVIDER': + return '---'; + + case 'TWEET': + if (entity.data.tweetId) { + return `[Embedded Tweet: https://x.com/i/status/${entity.data.tweetId}]`; + } + return undefined; + + case 'LINK': + if (entity.data.url) { + return `[Link: ${entity.data.url}]`; + } + return undefined; + + case 'IMAGE': + // Images in atomic blocks - could extract URL if available + return '[Image]'; + + default: + return undefined; + } +} + export function extractArticleText(result: GraphqlTweetResult | undefined): string | undefined { const article = result?.article; if (!article) { @@ -87,7 +323,22 @@ export function extractArticleText(result: GraphqlTweetResult | undefined): stri ), ); } + const title = firstText(articleResult.title, article.title); + + // Try to render from rich content_state first (Draft.js format with blocks + entityMap) + // This preserves code blocks, embedded tweets, markdown, etc. + const contentState = article.article_results?.result?.content_state; + const richBody = renderContentState(contentState); + if (richBody) { + // Rich content found - prepend title if not already included + if (title && !richBody.startsWith(title)) { + return `${title}\n\n${richBody}`; + } + return richBody; + } + + // Fallback to plain text extraction for articles without rich content_state let body = firstText( articleResult.plain_text, article.plain_text, diff --git a/tests/live/live.test.ts b/tests/live/live.test.ts index 5aa1a9c..2addb2d 100644 --- a/tests/live/live.test.ts +++ b/tests/live/live.test.ts @@ -333,4 +333,21 @@ d('live CLI (Twitter/X)', () => { expect(snapshot.cached).toBe(true); expect(snapshot.ids && Object.keys(snapshot.ids).length).toBeGreaterThan(0); }); + + it('long-form tweet (article) extracts rich content (opt-in)', async () => { + const longformTweetId = (process.env.BIRD_LIVE_LONGFORM_TWEET_ID ?? '').trim(); + if (!longformTweetId) { + // Skip unless explicitly provided - long-form tweets may be deleted/unavailable + return; + } + const read = await runBird([...baseArgs, '--cookie-timeout', cookieTimeoutArg, 'read', longformTweetId, '--json'], { + timeoutMs: 45_000, + }); + expect(read.exitCode).toBe(0); + const tweet = parseJson<{ id?: string; text?: string }>(read.stdout); + expect(tweet.id).toBe(longformTweetId); + // Long-form tweets (articles) typically have substantial content (>500 chars) + // This verifies the article content is being extracted, not just a stub + expect(tweet.text?.length).toBeGreaterThan(500); + }); }); diff --git a/tests/render-content-state.test.ts b/tests/render-content-state.test.ts new file mode 100644 index 0000000..c6fb83e --- /dev/null +++ b/tests/render-content-state.test.ts @@ -0,0 +1,322 @@ +import { describe, expect, it } from 'vitest'; +import { renderContentState } from '../src/lib/twitter-client-utils.js'; + +describe('renderContentState', () => { + it('returns undefined for undefined input', () => { + expect(renderContentState(undefined)).toBeUndefined(); + }); + + it('returns undefined for empty blocks', () => { + expect(renderContentState({ blocks: [], entityMap: [] })).toBeUndefined(); + }); + + it('renders unstyled blocks as plain paragraphs', () => { + const result = renderContentState({ + blocks: [ + { key: '1', type: 'unstyled', text: 'First paragraph', entityRanges: [], inlineStyleRanges: [] }, + { key: '2', type: 'unstyled', text: 'Second paragraph', entityRanges: [], inlineStyleRanges: [] }, + ], + entityMap: [], + }); + expect(result).toBe('First paragraph\n\nSecond paragraph'); + }); + + it('renders header-one as # heading', () => { + const result = renderContentState({ + blocks: [{ key: '1', type: 'header-one', text: 'Main Title', entityRanges: [], inlineStyleRanges: [] }], + entityMap: [], + }); + expect(result).toBe('# Main Title'); + }); + + it('renders header-two as ## heading', () => { + const result = renderContentState({ + blocks: [{ key: '1', type: 'header-two', text: 'Section Title', entityRanges: [], inlineStyleRanges: [] }], + entityMap: [], + }); + expect(result).toBe('## Section Title'); + }); + + it('renders header-three as ### heading', () => { + const result = renderContentState({ + blocks: [{ key: '1', type: 'header-three', text: 'Subsection', entityRanges: [], inlineStyleRanges: [] }], + entityMap: [], + }); + expect(result).toBe('### Subsection'); + }); + + it('renders unordered-list-item as bullet points', () => { + const result = renderContentState({ + blocks: [ + { key: '1', type: 'unordered-list-item', text: 'Item one', entityRanges: [], inlineStyleRanges: [] }, + { key: '2', type: 'unordered-list-item', text: 'Item two', entityRanges: [], inlineStyleRanges: [] }, + ], + entityMap: [], + }); + expect(result).toBe('- Item one\n\n- Item two'); + }); + + it('renders ordered-list-item with incrementing numbers', () => { + const result = renderContentState({ + blocks: [ + { key: '1', type: 'ordered-list-item', text: 'First step', entityRanges: [], inlineStyleRanges: [] }, + { key: '2', type: 'ordered-list-item', text: 'Second step', entityRanges: [], inlineStyleRanges: [] }, + { key: '3', type: 'ordered-list-item', text: 'Third step', entityRanges: [], inlineStyleRanges: [] }, + ], + entityMap: [], + }); + expect(result).toBe('1. First step\n\n2. Second step\n\n3. Third step'); + }); + + it('resets ordered list counter after non-list block', () => { + const result = renderContentState({ + blocks: [ + { key: '1', type: 'ordered-list-item', text: 'First', entityRanges: [], inlineStyleRanges: [] }, + { key: '2', type: 'ordered-list-item', text: 'Second', entityRanges: [], inlineStyleRanges: [] }, + { key: '3', type: 'unstyled', text: 'Paragraph break', entityRanges: [], inlineStyleRanges: [] }, + { key: '4', type: 'ordered-list-item', text: 'New first', entityRanges: [], inlineStyleRanges: [] }, + ], + entityMap: [], + }); + expect(result).toBe('1. First\n\n2. Second\n\nParagraph break\n\n1. New first'); + }); + + it('renders blockquote with > prefix', () => { + const result = renderContentState({ + blocks: [{ key: '1', type: 'blockquote', text: 'A wise quote', entityRanges: [], inlineStyleRanges: [] }], + entityMap: [], + }); + expect(result).toBe('> A wise quote'); + }); + + it('renders MARKDOWN entity as code block', () => { + const result = renderContentState({ + blocks: [ + { + key: '1', + type: 'atomic', + text: ' ', + entityRanges: [{ key: 0, offset: 0, length: 1 }], + inlineStyleRanges: [], + }, + ], + entityMap: [ + { + key: '0', + value: { + type: 'MARKDOWN', + mutability: 'Mutable', + data: { markdown: '```bash\necho "hello"\n```' }, + }, + }, + ], + }); + expect(result).toBe('```bash\necho "hello"\n```'); + }); + + it('renders DIVIDER entity as horizontal rule', () => { + const result = renderContentState({ + blocks: [ + { + key: '1', + type: 'atomic', + text: ' ', + entityRanges: [{ key: 0, offset: 0, length: 1 }], + inlineStyleRanges: [], + }, + ], + entityMap: [ + { + key: '0', + value: { + type: 'DIVIDER', + mutability: 'Immutable', + data: {}, + }, + }, + ], + }); + expect(result).toBe('---'); + }); + + it('renders TWEET entity with URL', () => { + const result = renderContentState({ + blocks: [ + { + key: '1', + type: 'atomic', + text: ' ', + entityRanges: [{ key: 0, offset: 0, length: 1 }], + inlineStyleRanges: [], + }, + ], + entityMap: [ + { + key: '0', + value: { + type: 'TWEET', + mutability: 'Immutable', + data: { tweetId: '1234567890' }, + }, + }, + ], + }); + expect(result).toBe('[Embedded Tweet: https://x.com/i/status/1234567890]'); + }); + + it('renders LINK entity in atomic block', () => { + const result = renderContentState({ + blocks: [ + { + key: '1', + type: 'atomic', + text: ' ', + entityRanges: [{ key: 0, offset: 0, length: 1 }], + inlineStyleRanges: [], + }, + ], + entityMap: [ + { + key: '0', + value: { + type: 'LINK', + mutability: 'Mutable', + data: { url: 'https://example.com' }, + }, + }, + ], + }); + expect(result).toBe('[Link: https://example.com]'); + }); + + it('renders inline LINK entity as markdown link', () => { + const result = renderContentState({ + blocks: [ + { + key: '1', + type: 'unstyled', + text: 'Check out this link for more info', + entityRanges: [{ key: 0, offset: 15, length: 4 }], + inlineStyleRanges: [], + }, + ], + entityMap: [ + { + key: '0', + value: { + type: 'LINK', + mutability: 'Mutable', + data: { url: 'https://example.com' }, + }, + }, + ], + }); + expect(result).toBe('Check out this [link](https://example.com) for more info'); + }); + + it('renders IMAGE entity as placeholder', () => { + const result = renderContentState({ + blocks: [ + { + key: '1', + type: 'atomic', + text: ' ', + entityRanges: [{ key: 0, offset: 0, length: 1 }], + inlineStyleRanges: [], + }, + ], + entityMap: [ + { + key: '0', + value: { + type: 'IMAGE', + mutability: 'Immutable', + data: {}, + }, + }, + ], + }); + expect(result).toBe('[Image]'); + }); + + it('handles complex document with mixed content', () => { + const result = renderContentState({ + blocks: [ + { key: '1', type: 'unstyled', text: 'Introduction paragraph.', entityRanges: [], inlineStyleRanges: [] }, + { key: '2', type: 'header-two', text: 'Getting Started', entityRanges: [], inlineStyleRanges: [] }, + { key: '3', type: 'ordered-list-item', text: 'Install dependencies', entityRanges: [], inlineStyleRanges: [] }, + { key: '4', type: 'ordered-list-item', text: 'Run the script', entityRanges: [], inlineStyleRanges: [] }, + { + key: '5', + type: 'atomic', + text: ' ', + entityRanges: [{ key: 0, offset: 0, length: 1 }], + inlineStyleRanges: [], + }, + { key: '6', type: 'unstyled', text: 'Conclusion.', entityRanges: [], inlineStyleRanges: [] }, + ], + entityMap: [ + { + key: '0', + value: { + type: 'MARKDOWN', + mutability: 'Mutable', + data: { markdown: '```bash\nnpm install\n```' }, + }, + }, + ], + }); + + expect(result).toBe( + 'Introduction paragraph.\n\n' + + '## Getting Started\n\n' + + '1. Install dependencies\n\n' + + '2. Run the script\n\n' + + '```bash\nnpm install\n```\n\n' + + 'Conclusion.', + ); + }); + + it('skips empty text blocks', () => { + const result = renderContentState({ + blocks: [ + { key: '1', type: 'unstyled', text: 'Content', entityRanges: [], inlineStyleRanges: [] }, + { key: '2', type: 'unstyled', text: ' ', entityRanges: [], inlineStyleRanges: [] }, + { key: '3', type: 'unstyled', text: '', entityRanges: [], inlineStyleRanges: [] }, + { key: '4', type: 'unstyled', text: 'More content', entityRanges: [], inlineStyleRanges: [] }, + ], + entityMap: [], + }); + expect(result).toBe('Content\n\nMore content'); + }); + + it('handles atomic block with missing entity gracefully', () => { + const result = renderContentState({ + blocks: [ + { key: '1', type: 'unstyled', text: 'Before', entityRanges: [], inlineStyleRanges: [] }, + { + key: '2', + type: 'atomic', + text: ' ', + entityRanges: [{ key: 99, offset: 0, length: 1 }], + inlineStyleRanges: [], + }, + { key: '3', type: 'unstyled', text: 'After', entityRanges: [], inlineStyleRanges: [] }, + ], + entityMap: [], + }); + expect(result).toBe('Before\n\nAfter'); + }); + + it('handles atomic block with no entityRanges gracefully', () => { + const result = renderContentState({ + blocks: [ + { key: '1', type: 'unstyled', text: 'Before', entityRanges: [], inlineStyleRanges: [] }, + { key: '2', type: 'atomic', text: ' ', entityRanges: [], inlineStyleRanges: [] }, + { key: '3', type: 'unstyled', text: 'After', entityRanges: [], inlineStyleRanges: [] }, + ], + entityMap: [], + }); + expect(result).toBe('Before\n\nAfter'); + }); +});