Merge pull request #36 from crcatala/feature/longform-tweet-rich-content

feat: support rich content in long-form tweets (X Articles)
This commit is contained in:
Peter Steinberger
2026-01-12 05:14:53 +00:00
committed by GitHub
8 changed files with 713 additions and 2 deletions
+1
View File
@@ -8,6 +8,7 @@
- Rich text output now shows article previews, quoted tweets, and media links (#32) — thanks @odysseus0. - Rich text output now shows article previews, quoted tweets, and media links (#32) — thanks @odysseus0.
- `user-tweets` command to fetch a user's profile timeline (#34) — thanks @crcatala. - `user-tweets` command to fetch a user's profile timeline (#34) — thanks @crcatala.
- `replies` and `thread` now support pagination (`--all`, `--max-pages`, `--cursor`, `--delay`) (#35) — thanks @crcatala. - `replies` and `thread` now support pagination (`--all`, `--max-pages`, `--cursor`, `--delay`) (#35) — thanks @crcatala.
- Long-form article tweets now render rich Draft.js content blocks/entities (#36) — thanks @crcatala.
### Changed ### Changed
- Library typing: `SearchResult` is now a discriminated union (so `error` only exists when `success: false`). - Library typing: `SearchResult` is now a discriminated union (so `error` only exists when `success: false`).
+6 -2
View File
@@ -4,6 +4,8 @@ import { buildSearchFeatures } from './twitter-client-features.js';
import type { SearchResult, TweetData } from './twitter-client-types.js'; import type { SearchResult, TweetData } from './twitter-client-types.js';
import { extractCursorFromInstructions, parseTweetsFromInstructions } from './twitter-client-utils.js'; import { extractCursorFromInstructions, parseTweetsFromInstructions } from './twitter-client-utils.js';
const RAW_QUERY_MISSING_REGEX = /must be defined/i;
/** Options for search methods */ /** Options for search methods */
export interface SearchFetchOptions { export interface SearchFetchOptions {
/** Include raw GraphQL response in `_raw` field */ /** Include raw GraphQL response in `_raw` field */
@@ -24,7 +26,7 @@ function isQueryIdMismatch(payload: string): boolean {
if (error?.extensions?.code === 'GRAPHQL_VALIDATION_FAILED') { if (error?.extensions?.code === 'GRAPHQL_VALIDATION_FAILED') {
return true; return true;
} }
if (error?.path?.includes('rawQuery') && /must be defined/i.test(error.message ?? '')) { if (error?.path?.includes('rawQuery') && RAW_QUERY_MISSING_REGEX.test(error.message ?? '')) {
return true; return true;
} }
return false; return false;
@@ -143,7 +145,9 @@ export function withSearch<TBase extends AbstractConstructor<TwitterClientBase>>
}; };
if (data.errors && data.errors.length > 0) { if (data.errors && data.errors.length > 0) {
const shouldRefreshQueryIds = data.errors.some((error) => error?.extensions?.code === 'GRAPHQL_VALIDATION_FAILED'); const shouldRefreshQueryIds = data.errors.some(
(error) => error?.extensions?.code === 'GRAPHQL_VALIDATION_FAILED',
);
return { return {
success: false as const, success: false as const,
error: data.errors.map((e) => e.message).join(', '), error: data.errors.map((e) => e.message).join(', '),
+6
View File
@@ -177,9 +177,15 @@ export function withTweetDetails<TBase extends AbstractConstructor<TwitterClient
rweb_video_timestamps_enabled: true, rweb_video_timestamps_enabled: true,
}; };
const fieldToggles = {
...buildArticleFieldToggles(),
withArticleRichContentState: true,
};
const params = new URLSearchParams({ const params = new URLSearchParams({
variables: JSON.stringify(variables), variables: JSON.stringify(variables),
features: JSON.stringify(features), features: JSON.stringify(features),
fieldToggles: JSON.stringify(fieldToggles),
}); });
try { try {
+28
View File
@@ -130,6 +130,34 @@ export type GraphqlTweetResult = {
}; };
}>; }>;
}>; }>;
/** Draft.js content state for rich article content */
content_state?: {
blocks: Array<{
key: string;
type: string;
text: string;
data?: Record<string, unknown>;
entityRanges?: Array<{ key: number; offset: number; length: number }>;
inlineStyleRanges?: Array<{ offset: number; length: number; style: string }>;
}>;
entityMap?:
| Array<{
key: string;
value: {
type: string;
mutability: string;
data: Record<string, unknown>;
};
}>
| Record<
string,
{
type: string;
mutability: string;
data: Record<string, unknown>;
}
>;
};
}; };
}; };
plain_text?: string; plain_text?: string;
+271
View File
@@ -66,6 +66,253 @@ export function uniqueOrdered(values: string[]): string[] {
return result; return result;
} }
// ============================================================================
// Draft.js Content State Types and Parser for Long-form Tweets (X Articles)
// ============================================================================
/** Inline style range for text formatting (Bold, Italic, etc.) */
interface InlineStyleRange {
offset: number;
length: number;
style: string;
}
/** Entity range linking a portion of text to an entity in entityMap */
interface EntityRange {
key: number;
offset: number;
length: number;
}
/** A content block in Draft.js format */
interface ContentBlock {
key: string;
type: string;
text: string;
data?: {
mentions?: Array<{ fromIndex: number; toIndex: number; text: string }>;
};
entityRanges?: EntityRange[];
inlineStyleRanges?: InlineStyleRange[];
}
/** Entity data for different entity types */
interface EntityValue {
type: string;
mutability: string;
data: {
markdown?: string;
url?: string;
tweetId?: string;
};
}
/** Entity map entry */
interface EntityMapEntry {
key: string;
value: EntityValue;
}
/** Draft.js content state structure */
interface ContentState {
blocks: ContentBlock[];
entityMap?: Array<EntityMapEntry> | Record<string, EntityValue>;
}
/**
* Renders a Draft.js content_state into readable markdown/text format.
* Handles blocks (paragraphs, headers, lists) and entities (code blocks, links, tweets, dividers).
*/
export function renderContentState(contentState: ContentState | undefined): string | undefined {
if (!contentState?.blocks || contentState.blocks.length === 0) {
return undefined;
}
// Build entity lookup map from array/object formats
const entityMap = new Map<number, EntityValue>();
const rawEntityMap = contentState.entityMap ?? [];
if (Array.isArray(rawEntityMap)) {
for (const entry of rawEntityMap) {
const key = Number.parseInt(entry.key, 10);
if (!Number.isNaN(key)) {
entityMap.set(key, entry.value);
}
}
} else {
for (const [key, value] of Object.entries(rawEntityMap)) {
const keyNumber = Number.parseInt(key, 10);
if (!Number.isNaN(keyNumber)) {
entityMap.set(keyNumber, value);
}
}
}
const outputLines: string[] = [];
let orderedListCounter = 0;
let previousBlockType: string | undefined;
for (const block of contentState.blocks) {
// Reset ordered list counter when leaving ordered list context
if (block.type !== 'ordered-list-item' && previousBlockType === 'ordered-list-item') {
orderedListCounter = 0;
}
switch (block.type) {
case 'unstyled': {
// Plain paragraph - just output text with any inline formatting
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(text);
}
break;
}
case 'header-one': {
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(`# ${text}`);
}
break;
}
case 'header-two': {
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(`## ${text}`);
}
break;
}
case 'header-three': {
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(`### ${text}`);
}
break;
}
case 'unordered-list-item': {
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(`- ${text}`);
}
break;
}
case 'ordered-list-item': {
orderedListCounter++;
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(`${orderedListCounter}. ${text}`);
}
break;
}
case 'blockquote': {
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(`> ${text}`);
}
break;
}
case 'atomic': {
// Atomic blocks are placeholders for embedded entities
const entityContent = renderAtomicBlock(block, entityMap);
if (entityContent) {
outputLines.push(entityContent);
}
break;
}
default: {
// Fallback: just output the text
const text = renderBlockText(block, entityMap);
if (text) {
outputLines.push(text);
}
}
}
previousBlockType = block.type;
}
const result = outputLines.join('\n\n');
return result.trim() || undefined;
}
/**
* Renders text content of a block, applying inline link entities.
*/
function renderBlockText(block: ContentBlock, entityMap: Map<number, EntityValue>): string {
let text = block.text;
// Handle LINK entities by appending URL in markdown format
// Process in reverse order to not mess up offsets
const linkRanges = (block.entityRanges ?? [])
.filter((range) => {
const entity = entityMap.get(range.key);
return entity?.type === 'LINK' && entity.data.url;
})
.sort((a, b) => b.offset - a.offset);
for (const range of linkRanges) {
const entity = entityMap.get(range.key);
if (entity?.data.url) {
const linkText = text.slice(range.offset, range.offset + range.length);
const markdownLink = `[${linkText}](${entity.data.url})`;
text = text.slice(0, range.offset) + markdownLink + text.slice(range.offset + range.length);
}
}
return text.trim();
}
/**
* Renders an atomic block by looking up its entity and returning appropriate content.
*/
function renderAtomicBlock(block: ContentBlock, entityMap: Map<number, EntityValue>): string | undefined {
const entityRanges = block.entityRanges ?? [];
if (entityRanges.length === 0) {
return undefined;
}
const entityKey = entityRanges[0].key;
const entity = entityMap.get(entityKey);
if (!entity) {
return undefined;
}
switch (entity.type) {
case 'MARKDOWN':
// Code blocks and other markdown content - output as-is
return entity.data.markdown?.trim();
case 'DIVIDER':
return '---';
case 'TWEET':
if (entity.data.tweetId) {
return `[Embedded Tweet: https://x.com/i/status/${entity.data.tweetId}]`;
}
return undefined;
case 'LINK':
if (entity.data.url) {
return `[Link: ${entity.data.url}]`;
}
return undefined;
case 'IMAGE':
// Images in atomic blocks - could extract URL if available
return '[Image]';
default:
return undefined;
}
}
export function extractArticleText(result: GraphqlTweetResult | undefined): string | undefined { export function extractArticleText(result: GraphqlTweetResult | undefined): string | undefined {
const article = result?.article; const article = result?.article;
if (!article) { if (!article) {
@@ -87,7 +334,31 @@ export function extractArticleText(result: GraphqlTweetResult | undefined): stri
), ),
); );
} }
const title = firstText(articleResult.title, article.title); const title = firstText(articleResult.title, article.title);
// Try to render from rich content_state first (Draft.js format with blocks + entityMap)
// This preserves code blocks, embedded tweets, markdown, etc.
const contentState = article.article_results?.result?.content_state;
const richBody = renderContentState(contentState);
if (richBody) {
// Rich content found - prepend title if not already included
if (title) {
const normalizedTitle = title.trim();
const trimmedBody = richBody.trimStart();
const headingMatches = [`# ${normalizedTitle}`, `## ${normalizedTitle}`, `### ${normalizedTitle}`];
const hasTitle =
trimmedBody === normalizedTitle ||
trimmedBody.startsWith(`${normalizedTitle}\n`) ||
headingMatches.some((heading) => trimmedBody.startsWith(heading));
if (!hasTitle) {
return `${title}\n\n${richBody}`;
}
}
return richBody;
}
// Fallback to plain text extraction for articles without rich content_state
let body = firstText( let body = firstText(
articleResult.plain_text, articleResult.plain_text,
article.plain_text, article.plain_text,
+33
View File
@@ -0,0 +1,33 @@
import { describe, expect, it } from 'vitest';
import type { GraphqlTweetResult } from '../src/lib/twitter-client-types.js';
import { extractArticleText } from '../src/lib/twitter-client-utils.js';
describe('extractArticleText', () => {
it('does not duplicate title when rich content starts with a heading', () => {
const result = {
rest_id: '1',
article: {
title: 'Hello World',
article_results: {
result: {
title: 'Hello World',
content_state: {
blocks: [
{
key: '1',
type: 'header-one',
text: 'Hello World',
entityRanges: [],
inlineStyleRanges: [],
},
],
entityMap: [],
},
},
},
},
} as GraphqlTweetResult;
expect(extractArticleText(result)).toBe('# Hello World');
});
});
+17
View File
@@ -333,4 +333,21 @@ d('live CLI (Twitter/X)', () => {
expect(snapshot.cached).toBe(true); expect(snapshot.cached).toBe(true);
expect(snapshot.ids && Object.keys(snapshot.ids).length).toBeGreaterThan(0); expect(snapshot.ids && Object.keys(snapshot.ids).length).toBeGreaterThan(0);
}); });
it('long-form tweet (article) extracts rich content (opt-in)', async () => {
const longformTweetId = (process.env.BIRD_LIVE_LONGFORM_TWEET_ID ?? '').trim();
if (!longformTweetId) {
// Skip unless explicitly provided - long-form tweets may be deleted/unavailable
return;
}
const read = await runBird([...baseArgs, '--cookie-timeout', cookieTimeoutArg, 'read', longformTweetId, '--json'], {
timeoutMs: 45_000,
});
expect(read.exitCode).toBe(0);
const tweet = parseJson<{ id?: string; text?: string }>(read.stdout);
expect(tweet.id).toBe(longformTweetId);
// Long-form tweets (articles) typically have substantial content (>500 chars)
// This verifies the article content is being extracted, not just a stub
expect(tweet.text?.length).toBeGreaterThan(500);
});
}); });
+351
View File
@@ -0,0 +1,351 @@
import { describe, expect, it } from 'vitest';
import { renderContentState } from '../src/lib/twitter-client-utils.js';
describe('renderContentState', () => {
it('returns undefined for undefined input', () => {
expect(renderContentState(undefined)).toBeUndefined();
});
it('returns undefined for empty blocks', () => {
expect(renderContentState({ blocks: [], entityMap: [] })).toBeUndefined();
});
it('renders unstyled blocks as plain paragraphs', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'unstyled', text: 'First paragraph', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'unstyled', text: 'Second paragraph', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('First paragraph\n\nSecond paragraph');
});
it('renders header-one as # heading', () => {
const result = renderContentState({
blocks: [{ key: '1', type: 'header-one', text: 'Main Title', entityRanges: [], inlineStyleRanges: [] }],
entityMap: [],
});
expect(result).toBe('# Main Title');
});
it('renders header-two as ## heading', () => {
const result = renderContentState({
blocks: [{ key: '1', type: 'header-two', text: 'Section Title', entityRanges: [], inlineStyleRanges: [] }],
entityMap: [],
});
expect(result).toBe('## Section Title');
});
it('renders header-three as ### heading', () => {
const result = renderContentState({
blocks: [{ key: '1', type: 'header-three', text: 'Subsection', entityRanges: [], inlineStyleRanges: [] }],
entityMap: [],
});
expect(result).toBe('### Subsection');
});
it('renders unordered-list-item as bullet points', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'unordered-list-item', text: 'Item one', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'unordered-list-item', text: 'Item two', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('- Item one\n\n- Item two');
});
it('renders ordered-list-item with incrementing numbers', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'ordered-list-item', text: 'First step', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'ordered-list-item', text: 'Second step', entityRanges: [], inlineStyleRanges: [] },
{ key: '3', type: 'ordered-list-item', text: 'Third step', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('1. First step\n\n2. Second step\n\n3. Third step');
});
it('resets ordered list counter after non-list block', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'ordered-list-item', text: 'First', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'ordered-list-item', text: 'Second', entityRanges: [], inlineStyleRanges: [] },
{ key: '3', type: 'unstyled', text: 'Paragraph break', entityRanges: [], inlineStyleRanges: [] },
{ key: '4', type: 'ordered-list-item', text: 'New first', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('1. First\n\n2. Second\n\nParagraph break\n\n1. New first');
});
it('renders blockquote with > prefix', () => {
const result = renderContentState({
blocks: [{ key: '1', type: 'blockquote', text: 'A wise quote', entityRanges: [], inlineStyleRanges: [] }],
entityMap: [],
});
expect(result).toBe('> A wise quote');
});
it('renders MARKDOWN entity as code block', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
],
entityMap: [
{
key: '0',
value: {
type: 'MARKDOWN',
mutability: 'Mutable',
data: { markdown: '```bash\necho "hello"\n```' },
},
},
],
});
expect(result).toBe('```bash\necho "hello"\n```');
});
it('handles entityMap object form', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
},
],
entityMap: {
0: {
type: 'MARKDOWN',
mutability: 'Mutable',
data: { markdown: '```js\nconsole.log("ok")\n```' },
},
},
});
expect(result).toBe('```js\nconsole.log("ok")\n```');
});
it('renders DIVIDER entity as horizontal rule', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
],
entityMap: [
{
key: '0',
value: {
type: 'DIVIDER',
mutability: 'Immutable',
data: {},
},
},
],
});
expect(result).toBe('---');
});
it('renders TWEET entity with URL', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
],
entityMap: [
{
key: '0',
value: {
type: 'TWEET',
mutability: 'Immutable',
data: { tweetId: '1234567890' },
},
},
],
});
expect(result).toBe('[Embedded Tweet: https://x.com/i/status/1234567890]');
});
it('renders LINK entity in atomic block', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
],
entityMap: [
{
key: '0',
value: {
type: 'LINK',
mutability: 'Mutable',
data: { url: 'https://example.com' },
},
},
],
});
expect(result).toBe('[Link: https://example.com]');
});
it('renders inline LINK entity as markdown link', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'unstyled',
text: 'Check out this link for more info',
entityRanges: [{ key: 0, offset: 15, length: 4 }],
inlineStyleRanges: [],
},
],
entityMap: [
{
key: '0',
value: {
type: 'LINK',
mutability: 'Mutable',
data: { url: 'https://example.com' },
},
},
],
});
expect(result).toBe('Check out this [link](https://example.com) for more info');
});
it('renders IMAGE entity as placeholder', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
],
entityMap: [
{
key: '0',
value: {
type: 'IMAGE',
mutability: 'Immutable',
data: {},
},
},
],
});
expect(result).toBe('[Image]');
});
it('handles complex document with mixed content', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'unstyled', text: 'Introduction paragraph.', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'header-two', text: 'Getting Started', entityRanges: [], inlineStyleRanges: [] },
{ key: '3', type: 'ordered-list-item', text: 'Install dependencies', entityRanges: [], inlineStyleRanges: [] },
{ key: '4', type: 'ordered-list-item', text: 'Run the script', entityRanges: [], inlineStyleRanges: [] },
{
key: '5',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
{ key: '6', type: 'unstyled', text: 'Conclusion.', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [
{
key: '0',
value: {
type: 'MARKDOWN',
mutability: 'Mutable',
data: { markdown: '```bash\nnpm install\n```' },
},
},
],
});
expect(result).toBe(
'Introduction paragraph.\n\n' +
'## Getting Started\n\n' +
'1. Install dependencies\n\n' +
'2. Run the script\n\n' +
'```bash\nnpm install\n```\n\n' +
'Conclusion.',
);
});
it('skips empty text blocks', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'unstyled', text: 'Content', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'unstyled', text: ' ', entityRanges: [], inlineStyleRanges: [] },
{ key: '3', type: 'unstyled', text: '', entityRanges: [], inlineStyleRanges: [] },
{ key: '4', type: 'unstyled', text: 'More content', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('Content\n\nMore content');
});
it('handles missing entityRanges on text blocks', () => {
const result = renderContentState({
blocks: [{ key: '1', type: 'unstyled', text: 'Content' }],
entityMap: [],
});
expect(result).toBe('Content');
});
it('handles atomic block with missing entity gracefully', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'unstyled', text: 'Before', entityRanges: [], inlineStyleRanges: [] },
{
key: '2',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 99, offset: 0, length: 1 }],
inlineStyleRanges: [],
},
{ key: '3', type: 'unstyled', text: 'After', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('Before\n\nAfter');
});
it('handles atomic block with no entityRanges gracefully', () => {
const result = renderContentState({
blocks: [
{ key: '1', type: 'unstyled', text: 'Before', entityRanges: [], inlineStyleRanges: [] },
{ key: '2', type: 'atomic', text: ' ', entityRanges: [], inlineStyleRanges: [] },
{ key: '3', type: 'unstyled', text: 'After', entityRanges: [], inlineStyleRanges: [] },
],
entityMap: [],
});
expect(result).toBe('Before\n\nAfter');
});
});