fix: paginate search results

This commit is contained in:
Peter Steinberger
2026-01-01 12:15:23 +01:00
parent cfd4045d3e
commit 3cca0ee989
3 changed files with 158 additions and 26 deletions
+1
View File
@@ -13,6 +13,7 @@
- Query ID updater now tracks BookmarkFolderTimeline and keeps bookmark query IDs seeded. - Query ID updater now tracks BookmarkFolderTimeline and keeps bookmark query IDs seeded.
- `following`/`followers` JSON user fields are now camelCase (`followersCount`, `followingCount`, `isBlueVerified`, `profileImageUrl`, `createdAt`). - `following`/`followers` JSON user fields are now camelCase (`followersCount`, `followingCount`, `isBlueVerified`, `profileImageUrl`, `createdAt`).
- Cookie extraction timeout is now configurable (default 30s on macOS) via `--cookie-timeout` / `BIRD_COOKIE_TIMEOUT_MS` (thanks @tylerseymour). - Cookie extraction timeout is now configurable (default 30s on macOS) via `--cookie-timeout` / `BIRD_COOKIE_TIMEOUT_MS` (thanks @tylerseymour).
- Search now paginates beyond 20 results when using `-n` (thanks @ryanh-ai).
## 0.4.1 — 2025-12-31 ## 0.4.1 — 2025-12-31
### Added ### Added
+77 -21
View File
@@ -924,6 +924,30 @@ export class TwitterClient {
return tweets; return tweets;
} }
private extractCursorFromInstructions(
instructions:
| Array<{
entries?: Array<{
content?: {
cursorType?: string;
value?: string;
};
}>;
}>
| undefined,
cursorType = 'Bottom',
): string | undefined {
for (const instruction of instructions ?? []) {
for (const entry of instruction.entries ?? []) {
const content = entry.content;
if (content?.cursorType === cursorType && typeof content.value === 'string' && content.value.length > 0) {
return content.value;
}
}
}
return undefined;
}
private buildArticleFeatures(): Record<string, boolean> { private buildArticleFeatures(): Record<string, boolean> {
return { return {
rweb_video_screen_enabled: true, rweb_video_screen_enabled: true,
@@ -1650,25 +1674,30 @@ export class TwitterClient {
* Search for tweets matching a query * Search for tweets matching a query
*/ */
async search(query: string, count = 20): Promise<SearchResult> { async search(query: string, count = 20): Promise<SearchResult> {
const variables = {
rawQuery: query,
count,
querySource: 'typed_query',
product: 'Latest',
};
const features = this.buildSearchFeatures(); const features = this.buildSearchFeatures();
const pageSize = 20;
const seen = new Set<string>();
const tweets: TweetData[] = [];
let cursor: string | undefined;
const params = new URLSearchParams({ const fetchPage = async (pageCount: number, pageCursor?: string) => {
variables: JSON.stringify(variables),
});
const tryOnce = async () => {
let lastError: string | undefined; let lastError: string | undefined;
let had404 = false; let had404 = false;
const queryIds = await this.getSearchTimelineQueryIds(); const queryIds = await this.getSearchTimelineQueryIds();
for (const queryId of queryIds) { for (const queryId of queryIds) {
const variables = {
rawQuery: query,
count: pageCount,
querySource: 'typed_query',
product: 'Latest',
...(pageCursor ? { cursor: pageCursor } : {}),
};
const params = new URLSearchParams({
variables: JSON.stringify(variables),
});
const url = `${TWITTER_API_BASE}/${queryId}/SearchTimeline?${params.toString()}`; const url = `${TWITTER_API_BASE}/${queryId}/SearchTimeline?${params.toString()}`;
try { try {
@@ -1737,9 +1766,10 @@ export class TwitterClient {
} }
const instructions = data.data?.search_by_raw_query?.search_timeline?.timeline?.instructions; const instructions = data.data?.search_by_raw_query?.search_timeline?.timeline?.instructions;
const tweets = this.parseTweetsFromInstructions(instructions); const pageTweets = this.parseTweetsFromInstructions(instructions);
const nextCursor = this.extractCursorFromInstructions(instructions);
return { success: true as const, tweets, had404 }; return { success: true as const, tweets: pageTweets, cursor: nextCursor, had404 };
} catch (error) { } catch (error) {
lastError = error instanceof Error ? error.message : String(error); lastError = error instanceof Error ? error.message : String(error);
} }
@@ -1748,21 +1778,47 @@ export class TwitterClient {
return { success: false as const, error: lastError ?? 'Unknown error fetching search results', had404 }; return { success: false as const, error: lastError ?? 'Unknown error fetching search results', had404 };
}; };
const firstAttempt = await tryOnce(); const fetchWithRefresh = async (pageCount: number, pageCursor?: string) => {
const firstAttempt = await fetchPage(pageCount, pageCursor);
if (firstAttempt.success) { if (firstAttempt.success) {
return { success: true, tweets: firstAttempt.tweets }; return firstAttempt;
} }
if (firstAttempt.had404) { if (firstAttempt.had404) {
await this.refreshQueryIds(); await this.refreshQueryIds();
const secondAttempt = await tryOnce(); const secondAttempt = await fetchPage(pageCount, pageCursor);
if (secondAttempt.success) { if (secondAttempt.success) {
return { success: true, tweets: secondAttempt.tweets }; return secondAttempt;
} }
return { success: false, error: secondAttempt.error }; return { success: false as const, error: secondAttempt.error };
}
return { success: false as const, error: firstAttempt.error };
};
while (tweets.length < count) {
const pageCount = Math.min(pageSize, count - tweets.length);
const page = await fetchWithRefresh(pageCount, cursor);
if (!page.success) {
return { success: false, error: page.error };
} }
return { success: false, error: firstAttempt.error }; for (const tweet of page.tweets) {
if (seen.has(tweet.id)) {
continue;
}
seen.add(tweet.id);
tweets.push(tweet);
if (tweets.length >= count) {
break;
}
}
if (!page.cursor || page.cursor === cursor || page.tweets.length === 0) {
break;
}
cursor = page.cursor;
}
return { success: true, tweets };
} }
/** /**
+75
View File
@@ -1167,6 +1167,81 @@ describe('TwitterClient', () => {
expect(result.error).toContain('Unknown error fetching search results'); expect(result.error).toContain('Unknown error fetching search results');
expect(mockFetch).not.toHaveBeenCalled(); expect(mockFetch).not.toHaveBeenCalled();
}); });
it('paginates search results using the bottom cursor', async () => {
const makeSearchEntry = (id: string, text: string) => ({
content: {
itemContent: {
tweet_results: {
result: {
rest_id: id,
legacy: {
full_text: text,
created_at: '2024-01-01T00:00:00Z',
reply_count: 0,
retweet_count: 0,
favorite_count: 0,
conversation_id_str: id,
},
core: {
user_results: {
result: { legacy: { screen_name: 'root', name: 'Root' } },
},
},
},
},
},
},
});
const makeResponse = (ids: string[], cursor?: string) => ({
data: {
search_by_raw_query: {
search_timeline: {
timeline: {
instructions: [
{
entries: [
...ids.map((id) => makeSearchEntry(id, `tweet-${id}`)),
...(cursor ? [{ content: { cursorType: 'Bottom', value: cursor } }] : []),
],
},
],
},
},
},
},
});
mockFetch
.mockResolvedValueOnce({
ok: true,
status: 200,
json: async () => makeResponse(['1', '2'], 'cursor-1'),
})
.mockResolvedValueOnce({
ok: true,
status: 200,
json: async () => makeResponse(['2', '3']),
});
const client = new TwitterClient({ cookies: validCookies });
const result = await client.search('needle', 3);
expect(result.success).toBe(true);
expect(result.tweets?.map((tweet) => tweet.id)).toEqual(['1', '2', '3']);
expect(mockFetch).toHaveBeenCalledTimes(2);
const firstVars = JSON.parse(
new URL(mockFetch.mock.calls[0][0] as string).searchParams.get('variables') as string,
) as { cursor?: string };
const secondVars = JSON.parse(
new URL(mockFetch.mock.calls[1][0] as string).searchParams.get('variables') as string,
) as { cursor?: string };
expect(firstVars.cursor).toBeUndefined();
expect(secondVars.cursor).toBe('cursor-1');
});
}); });
describe('bookmarks', () => { describe('bookmarks', () => {