fix: harden article rich content rendering (#36) (thanks @crcatala)

This commit is contained in:
Peter Steinberger
2026-01-12 05:14:39 +00:00
parent 316cdf77be
commit 28695d2ef0
6 changed files with 121 additions and 25 deletions
+1
View File
@@ -8,6 +8,7 @@
- Rich text output now shows article previews, quoted tweets, and media links (#32) — thanks @odysseus0. - Rich text output now shows article previews, quoted tweets, and media links (#32) — thanks @odysseus0.
- `user-tweets` command to fetch a user's profile timeline (#34) — thanks @crcatala. - `user-tweets` command to fetch a user's profile timeline (#34) — thanks @crcatala.
- `replies` and `thread` now support pagination (`--all`, `--max-pages`, `--cursor`, `--delay`) (#35) — thanks @crcatala. - `replies` and `thread` now support pagination (`--all`, `--max-pages`, `--cursor`, `--delay`) (#35) — thanks @crcatala.
- Long-form article tweets now render rich Draft.js content blocks/entities (#36) — thanks @crcatala.
### Changed ### Changed
- Library typing: `SearchResult` is now a discriminated union (so `error` only exists when `success: false`). - Library typing: `SearchResult` is now a discriminated union (so `error` only exists when `success: false`).
+6 -2
View File
@@ -4,6 +4,8 @@ import { buildSearchFeatures } from './twitter-client-features.js';
import type { SearchResult, TweetData } from './twitter-client-types.js'; import type { SearchResult, TweetData } from './twitter-client-types.js';
import { extractCursorFromInstructions, parseTweetsFromInstructions } from './twitter-client-utils.js'; import { extractCursorFromInstructions, parseTweetsFromInstructions } from './twitter-client-utils.js';
const RAW_QUERY_MISSING_REGEX = /must be defined/i;
/** Options for search methods */ /** Options for search methods */
export interface SearchFetchOptions { export interface SearchFetchOptions {
/** Include raw GraphQL response in `_raw` field */ /** Include raw GraphQL response in `_raw` field */
@@ -24,7 +26,7 @@ function isQueryIdMismatch(payload: string): boolean {
if (error?.extensions?.code === 'GRAPHQL_VALIDATION_FAILED') { if (error?.extensions?.code === 'GRAPHQL_VALIDATION_FAILED') {
return true; return true;
} }
if (error?.path?.includes('rawQuery') && /must be defined/i.test(error.message ?? '')) { if (error?.path?.includes('rawQuery') && RAW_QUERY_MISSING_REGEX.test(error.message ?? '')) {
return true; return true;
} }
return false; return false;
@@ -143,7 +145,9 @@ export function withSearch<TBase extends AbstractConstructor<TwitterClientBase>>
}; };
if (data.errors && data.errors.length > 0) { if (data.errors && data.errors.length > 0) {
const shouldRefreshQueryIds = data.errors.some((error) => error?.extensions?.code === 'GRAPHQL_VALIDATION_FAILED'); const shouldRefreshQueryIds = data.errors.some(
(error) => error?.extensions?.code === 'GRAPHQL_VALIDATION_FAILED',
);
return { return {
success: false as const, success: false as const,
error: data.errors.map((e) => e.message).join(', '), error: data.errors.map((e) => e.message).join(', '),
+13 -4
View File
@@ -137,17 +137,26 @@ export type GraphqlTweetResult = {
type: string; type: string;
text: string; text: string;
data?: Record<string, unknown>; data?: Record<string, unknown>;
entityRanges: Array<{ key: number; offset: number; length: number }>; entityRanges?: Array<{ key: number; offset: number; length: number }>;
inlineStyleRanges: Array<{ offset: number; length: number; style: string }>; inlineStyleRanges?: Array<{ offset: number; length: number; style: string }>;
}>; }>;
entityMap: Array<{ entityMap?:
| Array<{
key: string; key: string;
value: { value: {
type: string; type: string;
mutability: string; mutability: string;
data: Record<string, unknown>; data: Record<string, unknown>;
}; };
}>; }>
| Record<
string,
{
type: string;
mutability: string;
data: Record<string, unknown>;
}
>;
}; };
}; };
}; };
+29 -9
View File
@@ -92,8 +92,8 @@ interface ContentBlock {
data?: { data?: {
mentions?: Array<{ fromIndex: number; toIndex: number; text: string }>; mentions?: Array<{ fromIndex: number; toIndex: number; text: string }>;
}; };
entityRanges: EntityRange[]; entityRanges?: EntityRange[];
inlineStyleRanges: InlineStyleRange[]; inlineStyleRanges?: InlineStyleRange[];
} }
/** Entity data for different entity types */ /** Entity data for different entity types */
@@ -116,7 +116,7 @@ interface EntityMapEntry {
/** Draft.js content state structure */ /** Draft.js content state structure */
interface ContentState { interface ContentState {
blocks: ContentBlock[]; blocks: ContentBlock[];
entityMap: EntityMapEntry[]; entityMap?: Array<EntityMapEntry> | Record<string, EntityValue>;
} }
/** /**
@@ -128,14 +128,24 @@ export function renderContentState(contentState: ContentState | undefined): stri
return undefined; return undefined;
} }
// Build entity lookup map from array format // Build entity lookup map from array/object formats
const entityMap = new Map<number, EntityValue>(); const entityMap = new Map<number, EntityValue>();
for (const entry of contentState.entityMap ?? []) { const rawEntityMap = contentState.entityMap ?? [];
if (Array.isArray(rawEntityMap)) {
for (const entry of rawEntityMap) {
const key = Number.parseInt(entry.key, 10); const key = Number.parseInt(entry.key, 10);
if (!Number.isNaN(key)) { if (!Number.isNaN(key)) {
entityMap.set(key, entry.value); entityMap.set(key, entry.value);
} }
} }
} else {
for (const [key, value] of Object.entries(rawEntityMap)) {
const keyNumber = Number.parseInt(key, 10);
if (!Number.isNaN(keyNumber)) {
entityMap.set(keyNumber, value);
}
}
}
const outputLines: string[] = []; const outputLines: string[] = [];
let orderedListCounter = 0; let orderedListCounter = 0;
@@ -239,7 +249,7 @@ function renderBlockText(block: ContentBlock, entityMap: Map<number, EntityValue
// Handle LINK entities by appending URL in markdown format // Handle LINK entities by appending URL in markdown format
// Process in reverse order to not mess up offsets // Process in reverse order to not mess up offsets
const linkRanges = block.entityRanges const linkRanges = (block.entityRanges ?? [])
.filter((range) => { .filter((range) => {
const entity = entityMap.get(range.key); const entity = entityMap.get(range.key);
return entity?.type === 'LINK' && entity.data.url; return entity?.type === 'LINK' && entity.data.url;
@@ -262,11 +272,12 @@ function renderBlockText(block: ContentBlock, entityMap: Map<number, EntityValue
* Renders an atomic block by looking up its entity and returning appropriate content. * Renders an atomic block by looking up its entity and returning appropriate content.
*/ */
function renderAtomicBlock(block: ContentBlock, entityMap: Map<number, EntityValue>): string | undefined { function renderAtomicBlock(block: ContentBlock, entityMap: Map<number, EntityValue>): string | undefined {
if (block.entityRanges.length === 0) { const entityRanges = block.entityRanges ?? [];
if (entityRanges.length === 0) {
return undefined; return undefined;
} }
const entityKey = block.entityRanges[0].key; const entityKey = entityRanges[0].key;
const entity = entityMap.get(entityKey); const entity = entityMap.get(entityKey);
if (!entity) { if (!entity) {
@@ -332,9 +343,18 @@ export function extractArticleText(result: GraphqlTweetResult | undefined): stri
const richBody = renderContentState(contentState); const richBody = renderContentState(contentState);
if (richBody) { if (richBody) {
// Rich content found - prepend title if not already included // Rich content found - prepend title if not already included
if (title && !richBody.startsWith(title)) { if (title) {
const normalizedTitle = title.trim();
const trimmedBody = richBody.trimStart();
const headingMatches = [`# ${normalizedTitle}`, `## ${normalizedTitle}`, `### ${normalizedTitle}`];
const hasTitle =
trimmedBody === normalizedTitle ||
trimmedBody.startsWith(`${normalizedTitle}\n`) ||
headingMatches.some((heading) => trimmedBody.startsWith(heading));
if (!hasTitle) {
return `${title}\n\n${richBody}`; return `${title}\n\n${richBody}`;
} }
}
return richBody; return richBody;
} }
+33
View File
@@ -0,0 +1,33 @@
import { describe, expect, it } from 'vitest';
import type { GraphqlTweetResult } from '../src/lib/twitter-client-types.js';
import { extractArticleText } from '../src/lib/twitter-client-utils.js';
describe('extractArticleText', () => {
it('does not duplicate title when rich content starts with a heading', () => {
const result = {
rest_id: '1',
article: {
title: 'Hello World',
article_results: {
result: {
title: 'Hello World',
content_state: {
blocks: [
{
key: '1',
type: 'header-one',
text: 'Hello World',
entityRanges: [],
inlineStyleRanges: [],
},
],
entityMap: [],
},
},
},
},
} as GraphqlTweetResult;
expect(extractArticleText(result)).toBe('# Hello World');
});
});
+29
View File
@@ -114,6 +114,27 @@ describe('renderContentState', () => {
expect(result).toBe('```bash\necho "hello"\n```'); expect(result).toBe('```bash\necho "hello"\n```');
}); });
it('handles entityMap object form', () => {
const result = renderContentState({
blocks: [
{
key: '1',
type: 'atomic',
text: ' ',
entityRanges: [{ key: 0, offset: 0, length: 1 }],
},
],
entityMap: {
0: {
type: 'MARKDOWN',
mutability: 'Mutable',
data: { markdown: '```js\nconsole.log("ok")\n```' },
},
},
});
expect(result).toBe('```js\nconsole.log("ok")\n```');
});
it('renders DIVIDER entity as horizontal rule', () => { it('renders DIVIDER entity as horizontal rule', () => {
const result = renderContentState({ const result = renderContentState({
blocks: [ blocks: [
@@ -290,6 +311,14 @@ describe('renderContentState', () => {
expect(result).toBe('Content\n\nMore content'); expect(result).toBe('Content\n\nMore content');
}); });
it('handles missing entityRanges on text blocks', () => {
const result = renderContentState({
blocks: [{ key: '1', type: 'unstyled', text: 'Content' }],
entityMap: [],
});
expect(result).toBe('Content');
});
it('handles atomic block with missing entity gracefully', () => { it('handles atomic block with missing entity gracefully', () => {
const result = renderContentState({ const result = renderContentState({
blocks: [ blocks: [