feat: support rich content in long-form tweets (X Articles)

Add support for extracting rich content from X's long-form tweets,
including embedded code snippets, markdown blocks, quoted tweets,
and other structured content that was previously lost.

Changes:
- Add fieldToggles with withArticleRichContentState to TweetDetail API request
- Implement Draft.js content_state parser (renderContentState) that converts
  blocks and entities to readable markdown format
- Add content_state type definition to GraphqlTweetResult

Supported content:
- Block types: paragraphs, headers, ordered/unordered lists, blockquotes
- Entity types: MARKDOWN (code blocks), DIVIDER, TWEET, LINK, IMAGE

No impact on regular tweets - rich content only adds payload when present.

Includes 20 unit tests for the parser and an opt-in live smoke test.

Co-authored-by: Christian Catalan <[email protected]>
This commit is contained in:
cc-vps
2026-01-12 05:12:33 +00:00
committed by Peter Steinberger
co-authored by Christian Catalan
parent e0b960db1a
commit 316cdf77be
5 changed files with 615 additions and 0 deletions
+6
View File
@@ -177,9 +177,15 @@ export function withTweetDetails<TBase extends AbstractConstructor<TwitterClient
rweb_video_timestamps_enabled: true, rweb_video_timestamps_enabled: true,
}; };
const fieldToggles = {
...buildArticleFieldToggles(),
withArticleRichContentState: true,
};
const params = new URLSearchParams({ const params = new URLSearchParams({
variables: JSON.stringify(variables), variables: JSON.stringify(variables),
features: JSON.stringify(features), features: JSON.stringify(features),
fieldToggles: JSON.stringify(fieldToggles),
}); });
try { try {
+19
View File
@@ -130,6 +130,25 @@ export type GraphqlTweetResult = {
}; };
}>; }>;
}>; }>;
/** Draft.js content state for rich article content */
content_state?: {
blocks: Array<{
key: string;
type: string;
text: string;
data?: Record<string, unknown>;
entityRanges: Array<{ key: number; offset: number; length: number }>;
inlineStyleRanges: Array<{ offset: number; length: number; style: string }>;
}>;
entityMap: Array<{
key: string;
value: {
type: string;
mutability: string;
data: Record<string, unknown>;
};
}>;
};
}; };
}; };
plain_text?: string; plain_text?: string;
+251
View File
@@ -66,6 +66,242 @@ export function uniqueOrdered(values: string[]): string[] {
return result; return result;
} }
// ============================================================================
// Draft.js Content State Types and Parser for Long-form Tweets (X Articles)
// ============================================================================
/** Inline style range for text formatting (Bold, Italic, etc.) */
interface InlineStyleRange {
offset: number;
length: number;
style: string;
}
/** Entity range linking a portion of text to an entity in entityMap */
interface EntityRange {
key: number;
offset: number;
length: number;
}
/** A content block in Draft.js format */
interface ContentBlock {
key: string;
type: string;
text: string;
data?: {
mentions?: Array<{ fromIndex: number; toIndex: number; text: string }>;
};
entityRanges: EntityRange[];
inlineStyleRanges: InlineStyleRange[];
}
/** Entity data for different entity types */
interface EntityValue {
type: string;
mutability: string;
data: {
markdown?: string;
url?: string;
tweetId?: string;
};
}
/** Entity map entry */
interface EntityMapEntry {
key: string;
value: EntityValue;
}
/** Draft.js content state structure */
interface ContentState {
blocks: ContentBlock[];
entityMap: EntityMapEntry[];
}
/**
* Renders a Draft.js content_state into readable markdown/text format.
* Handles blocks (paragraphs, headers, lists) and entities (code blocks, links, tweets, dividers).
*/
export function renderContentState(contentState: ContentState | undefined): string | undefined {
if (!contentState?.blocks || contentState.blocks.length === 0) {
return undefined;
}
// Build entity lookup map from array format
const entityMap = new Map<number, EntityValue>();
for (const entry of contentState.entityMap ?? []) {
const key = Number.parseInt(entry.key, 10);
if (!Number.isNaN(key)) {
entityMap.set(key, entry.value);
}
}
const outputLines: string[] = [];
let orderedListCounter = 0;
let previousBlockType: string | undefined;
for (const block of contentState.blocks) {
// Reset ordered list counter when leaving ordered list context
if (block.type !== 'ordered-list-item' && previousBlockType === 'ordered-list-item') {
orderedListCounter = 0;
}
switch (block.type) {
case 'unstyled': {
// Plain paragraph - just output text with any inline formatting
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(text);
}
break;
}
case 'header-one': {
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(`# ${text}`);
}
break;
}
case 'header-two': {
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(`## ${text}`);
}
break;
}
case 'header-three': {
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(`### ${text}`);
}
break;
}
case 'unordered-list-item': {
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(`- ${text}`);
}
break;
}
case 'ordered-list-item': {
orderedListCounter++;
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(`${orderedListCounter}. ${text}`);
}
break;
}
case 'blockquote': {
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(`> ${text}`);
}
break;
}
case 'atomic': {
// Atomic blocks are placeholders for embedded entities
const entityContent = renderAtomicBlock(block, entityMap);
if (entityContent) {
outputLines.push(entityContent);
}
break;
}
default: {
// Fallback: just output the text
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(text);
}
}
}
previousBlockType = block.type;
}
const result = outputLines.join('\n\n');
return result.trim() || undefined;
}
/**
* Renders text content of a block, applying inline link entities.
*/
function renderBlockText(block: ContentBlock, entityMap: Map<number, EntityValue>): string {
let text = block.text;
// Handle LINK entities by appending URL in markdown format
// Process in reverse order to not mess up offsets
const linkRanges = block.entityRanges
.filter((range) => {
const entity = entityMap.get(range.key);
return entity?.type === 'LINK' && entity.data.url;
})
.sort((a, b) => b.offset - a.offset);
for (const range of linkRanges) {
const entity = entityMap.get(range.key);
if (entity?.data.url) {
const linkText = text.slice(range.offset, range.offset + range.length);
const markdownLink = `[${linkText}](${entity.data.url})`;
text = text.slice(0, range.offset) + markdownLink + text.slice(range.offset + range.length);
}
}
return text.trim();
}
/**
* Renders an atomic block by looking up its entity and returning appropriate content.
*/
function renderAtomicBlock(block: ContentBlock, entityMap: Map<number, EntityValue>): string | undefined {
if (block.entityRanges.length === 0) {
return undefined;
}
const entityKey = block.entityRanges[0].key;
const entity = entityMap.get(entityKey);
if (!entity) {
return undefined;
}
switch (entity.type) {
case 'MARKDOWN':
// Code blocks and other markdown content - output as-is
return entity.data.markdown?.trim();
case 'DIVIDER':
return '---';
case 'TWEET':
if (entity.data.tweetId) {
return `[Embedded Tweet: https://x.com/i/status/${entity.data.tweetId}]`;
}
return undefined;
case 'LINK':
if (entity.data.url) {
return `[Link: ${entity.data.url}]`;
}
return undefined;
case 'IMAGE':
// Images in atomic blocks - could extract URL if available
return '[Image]';
default:
return undefined;
}
}
export function extractArticleText(result: GraphqlTweetResult | undefined): string | undefined { export function extractArticleText(result: GraphqlTweetResult | undefined): string | undefined {
const article = result?.article; const article = result?.article;
if (!article) { if (!article) {
@@ -87,7 +323,22 @@ export function extractArticleText(result: GraphqlTweetResult | undefined): stri
), ),
); );
} }
const title = firstText(articleResult.title, article.title); const title = firstText(articleResult.title, article.title);
// Try to render from rich content_state first (Draft.js format with blocks + entityMap)
// This preserves code blocks, embedded tweets, markdown, etc.
const contentState = article.article_results?.result?.content_state;
const richBody = renderContentState(contentState);
if (richBody) {
// Rich content found - prepend title if not already included
if (title && !richBody.startsWith(title)) {
return `${title}\n\n${richBody}`;
}
return richBody;
}
// Fallback to plain text extraction for articles without rich content_state
let body = firstText( let body = firstText(
articleResult.plain_text, articleResult.plain_text,
article.plain_text, article.plain_text,
+17
View File
@@ -333,4 +333,21 @@ d('live CLI (Twitter/X)', () => {
expect(snapshot.cached).toBe(true); expect(snapshot.cached).toBe(true);
expect(snapshot.ids && Object.keys(snapshot.ids).length).toBeGreaterThan(0); expect(snapshot.ids && Object.keys(snapshot.ids).length).toBeGreaterThan(0);
}); });
it('long-form tweet (article) extracts rich content (opt-in)', async () => {
const longformTweetId = (process.env.BIRD_LIVE_LONGFORM_TWEET_ID ?? '').trim();
if (!longformTweetId) {
// Skip unless explicitly provided - long-form tweets may be deleted/unavailable
return;
}
const read = await runBird([...baseArgs, '--cookie-timeout', cookieTimeoutArg, 'read', longformTweetId, '--json'], {
timeoutMs: 45_000,
});
expect(read.exitCode).toBe(0);
const tweet = parseJson<{ id?: string; text?: string }>(read.stdout);
expect(tweet.id).toBe(longformTweetId);
// Long-form tweets (articles) typically have substantial content (>500 chars)
// This verifies the article content is being extracted, not just a stub
expect(tweet.text?.length).toBeGreaterThan(500);
});
}); });
+322
View File
@@ -0,0 +1,322 @@
import { describe, expect, it } from 'vitest';
import { renderContentState } from '../src/lib/twitter-client-utils.js';
describe('renderContentState', () => {
it('returns undefined for undefined input', () => {
expect(renderContentState(undefined)).toBeUndefined();
});
it('returns undefined for empty blocks', () => {
expect(renderContentState({ blocks: [], entityMap: [] })).toBeUndefined();
});
it('renders unstyled blocks as plain paragraphs', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'unstyled', text: 'First paragraph', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'unstyled', text: 'Second paragraph', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('First paragraph\n\nSecond paragraph');
});
it('renders header-one as # heading', () => {
const result = renderContentState({
blocks: [{ key: '1', type: 'header-one', text: 'Main Title', entityRanges: [], inlineStyleRanges: [] }],
entityMap: [],
});
expect(result).toBe('# Main Title');
});
it('renders header-two as ## heading', () => {
const result = renderContentState({
blocks: [{ key: '1', type: 'header-two', text: 'Section Title', entityRanges: [], inlineStyleRanges: [] }],
entityMap: [],
});
expect(result).toBe('## Section Title');
});
it('renders header-three as ### heading', () => {
const result = renderContentState({
blocks: [{ key: '1', type: 'header-three', text: 'Subsection', entityRanges: [], inlineStyleRanges: [] }],
entityMap: [],
});
expect(result).toBe('### Subsection');
});
it('renders unordered-list-item as bullet points', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'unordered-list-item', text: 'Item one', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'unordered-list-item', text: 'Item two', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('- Item one\n\n- Item two');
});
it('renders ordered-list-item with incrementing numbers', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'ordered-list-item', text: 'First step', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'ordered-list-item', text: 'Second step', entityRanges: [], inlineStyleRanges: [] },
{ key: '3', type: 'ordered-list-item', text: 'Third step', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('1. First step\n\n2. Second step\n\n3. Third step');
});
it('resets ordered list counter after non-list block', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'ordered-list-item', text: 'First', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'ordered-list-item', text: 'Second', entityRanges: [], inlineStyleRanges: [] },
{ key: '3', type: 'unstyled', text: 'Paragraph break', entityRanges: [], inlineStyleRanges: [] },
{ key: '4', type: 'ordered-list-item', text: 'New first', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('1. First\n\n2. Second\n\nParagraph break\n\n1. New first');
});
it('renders blockquote with > prefix', () => {
const result = renderContentState({
blocks: [{ key: '1', type: 'blockquote', text: 'A wise quote', entityRanges: [], inlineStyleRanges: [] }],
entityMap: [],
});
expect(result).toBe('> A wise quote');
});
it('renders MARKDOWN entity as code block', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
],
entityMap: [
{
key: '0',
value: {
type: 'MARKDOWN',
mutability: 'Mutable',
data: { markdown: '```bash\necho "hello"\n```' },
},
},
],
});
expect(result).toBe('```bash\necho "hello"\n```');
});
it('renders DIVIDER entity as horizontal rule', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
],
entityMap: [
{
key: '0',
value: {
type: 'DIVIDER',
mutability: 'Immutable',
data: {},
},
},
],
});
expect(result).toBe('---');
});
it('renders TWEET entity with URL', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
],
entityMap: [
{
key: '0',
value: {
type: 'TWEET',
mutability: 'Immutable',
data: { tweetId: '1234567890' },
},
},
],
});
expect(result).toBe('[Embedded Tweet: https://x.com/i/status/1234567890]');
});
it('renders LINK entity in atomic block', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
],
entityMap: [
{
key: '0',
value: {
type: 'LINK',
mutability: 'Mutable',
data: { url: 'https://example.com' },
},
},
],
});
expect(result).toBe('[Link: https://example.com]');
});
it('renders inline LINK entity as markdown link', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'unstyled',
text: 'Check out this link for more info',
entityRanges: [{ key: 0, offset: 15, length: 4 }],
inlineStyleRanges: [],
},
],
entityMap: [
{
key: '0',
value: {
type: 'LINK',
mutability: 'Mutable',
data: { url: 'https://example.com' },
},
},
],
});
expect(result).toBe('Check out this [link](https://example.com) for more info');
});
it('renders IMAGE entity as placeholder', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
],
entityMap: [
{
key: '0',
value: {
type: 'IMAGE',
mutability: 'Immutable',
data: {},
},
},
],
});
expect(result).toBe('[Image]');
});
it('handles complex document with mixed content', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'unstyled', text: 'Introduction paragraph.', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'header-two', text: 'Getting Started', entityRanges: [], inlineStyleRanges: [] },
{ key: '3', type: 'ordered-list-item', text: 'Install dependencies', entityRanges: [], inlineStyleRanges: [] },
{ key: '4', type: 'ordered-list-item', text: 'Run the script', entityRanges: [], inlineStyleRanges: [] },
{
key: '5',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
{ key: '6', type: 'unstyled', text: 'Conclusion.', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [
{
key: '0',
value: {
type: 'MARKDOWN',
mutability: 'Mutable',
data: { markdown: '```bash\nnpm install\n```' },
},
},
],
});
expect(result).toBe(
'Introduction paragraph.\n\n' +
'## Getting Started\n\n' +
'1. Install dependencies\n\n' +
'2. Run the script\n\n' +
'```bash\nnpm install\n```\n\n' +
'Conclusion.',
);
});
it('skips empty text blocks', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'unstyled', text: 'Content', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'unstyled', text: ' ', entityRanges: [], inlineStyleRanges: [] },
{ key: '3', type: 'unstyled', text: '', entityRanges: [], inlineStyleRanges: [] },
{ key: '4', type: 'unstyled', text: 'More content', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('Content\n\nMore content');
});
it('handles atomic block with missing entity gracefully', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'unstyled', text: 'Before', entityRanges: [], inlineStyleRanges: [] },
{
key: '2',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 99, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
{ key: '3', type: 'unstyled', text: 'After', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('Before\n\nAfter');
});
it('handles atomic block with no entityRanges gracefully', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'unstyled', text: 'Before', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'atomic', text: ' ', entityRanges: [], inlineStyleRanges: [] },
{ key: '3', type: 'unstyled', text: 'After', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('Before\n\nAfter');
});
});