Files
bird/src/lib/twitter-client-news.ts
T

431 lines
13 KiB
TypeScript

import type { AbstractConstructor, Mixin, TwitterClientBase } from './twitter-client-base.js';
import { TWITTER_API_BASE } from './twitter-client-constants.js';
import { buildExploreFeatures } from './twitter-client-features.js';
import type { SearchResult, TweetData } from './twitter-client-types.js';
const POST_COUNT_REGEX = /[\d.]+[KMB]?\s*posts?/i;
const POST_COUNT_MATCH_REGEX = /([\d.]+)([KMB]?)\s*posts?/i;
// Timeline IDs for different Explore tabs
const TIMELINE_IDS = {
forYou: 'VGltZWxpbmU6DAC2CwABAAAAB2Zvcl95b3UAAA==',
trending: 'VGltZWxpbmU6DAC2CwABAAAACHRyZW5kaW5nAAA=',
news: 'VGltZWxpbmU6DAC2CwABAAAABG5ld3MAAA==',
sports: 'VGltZWxpbmU6DAC2CwABAAAABnNwb3J0cwAA',
entertainment: 'VGltZWxpbmU6DAC2CwABAAAADWVudGVydGFpbm1lbnQAAA==',
} as const;
export type ExploreTab = keyof typeof TIMELINE_IDS;
/** Options for news fetch methods */
export interface NewsFetchOptions {
/** Include raw GraphQL response in `_raw` field */
includeRaw?: boolean;
/** Also fetch related tweets for each news item */
withTweets?: boolean;
/** Number of tweets to fetch per news item (default: 5) */
tweetsPerItem?: number;
/** Filter to show only AI-curated news items */
aiOnly?: boolean;
/** Fetch from specific tabs only (default: all tabs) */
tabs?: ExploreTab[];
}
export interface NewsItem {
id: string;
headline: string;
category?: string;
timeAgo?: string;
postCount?: number;
description?: string;
url?: string;
tweets?: TweetData[];
// biome-ignore lint/suspicious/noExplicitAny: Raw API response can have any structure
_raw?: any;
}
export type NewsResult =
| {
success: true;
items: NewsItem[];
}
| {
success: false;
error: string;
};
export interface TwitterClientNewsMethods {
getNews(count?: number, options?: NewsFetchOptions): Promise<NewsResult>;
}
export function withNews<TBase extends AbstractConstructor<TwitterClientBase>>(
Base: TBase,
): Mixin<TBase, TwitterClientNewsMethods> {
abstract class TwitterClientNews extends Base {
// biome-ignore lint/complexity/noUselessConstructor lint/suspicious/noExplicitAny: TS mixin constructor requirement.
constructor(...args: any[]) {
super(...args);
}
/**
* Fetch news and trending topics from Twitter's Explore page tabs
*/
async getNews(count = 10, options: NewsFetchOptions = {}): Promise<NewsResult> {
const {
includeRaw = false,
withTweets = false,
tweetsPerItem = 5,
aiOnly = false,
tabs = ['forYou', 'news', 'sports', 'entertainment'],
} = options;
const debug = process.env.BIRD_DEBUG === '1';
if (debug) {
console.error(`[getNews] Fetching from tabs: ${tabs.join(', ')}`);
}
const allItems: NewsItem[] = [];
const seenHeadlines = new Set<string>();
// Fetch from each tab
for (const tab of tabs) {
const timelineId = TIMELINE_IDS[tab];
if (!timelineId) {
continue;
}
try {
const tabItems = await this.fetchTimelineTab(tab, timelineId, count, aiOnly, includeRaw);
// Deduplicate across tabs
for (const item of tabItems) {
if (!seenHeadlines.has(item.headline)) {
seenHeadlines.add(item.headline);
allItems.push(item);
}
}
if (debug) {
console.error(`[getNews] Tab ${tab}: found ${tabItems.length} items, total unique: ${allItems.length}`);
}
// Stop early if we have enough
if (allItems.length >= count) {
break;
}
} catch (error) {
if (debug) {
console.error(`[getNews] Error fetching tab ${tab}:`, error);
}
// Continue with other tabs
}
}
if (allItems.length === 0) {
return { success: false, error: 'No news items found' };
}
// Limit to requested count
const items = allItems.slice(0, count);
if (withTweets) {
await this.enrichWithTweets(items, tweetsPerItem, includeRaw);
}
return { success: true, items };
}
/**
* Fetch a specific timeline tab using GenericTimelineById
*/
private async fetchTimelineTab(
tabName: string,
timelineId: string,
maxCount: number,
aiOnly: boolean,
includeRaw: boolean,
): Promise<NewsItem[]> {
const queryId = await this.getQueryId('GenericTimelineById');
const features = buildExploreFeatures();
const variables = {
timelineId: timelineId,
count: maxCount * 2, // Fetch more to account for filtering
includePromotedContent: false,
};
const params = new URLSearchParams({
variables: JSON.stringify(variables),
features: JSON.stringify(features),
});
const url = `${TWITTER_API_BASE}/${queryId}/GenericTimelineById?${params.toString()}`;
const response = await this.fetchWithTimeout(url, {
method: 'GET',
headers: this.getHeaders(),
});
if (!response.ok) {
const text = await response.text();
throw new Error(`HTTP ${response.status}: ${text.slice(0, 200)}`);
}
const data = (await response.json()) as {
// biome-ignore lint/suspicious/noExplicitAny: API response structure is complex
data?: any;
// biome-ignore lint/suspicious/noExplicitAny: API errors can have any structure
errors?: Array<{ message: string; code?: number; [key: string]: any }>;
};
// Debug: save response if BIRD_DEBUG_JSON is set
if (process.env.BIRD_DEBUG_JSON) {
const fs = await import('node:fs/promises');
const debugPath = process.env.BIRD_DEBUG_JSON.replace('.json', `-${tabName}.json`);
await fs.writeFile(debugPath, JSON.stringify(data, null, 2)).catch(() => {});
}
if (data.errors && data.errors.length > 0) {
throw new Error(data.errors.map((e) => e.message).join('; '));
}
// Parse timeline response
return this.parseTimelineTabItems(data, tabName, maxCount, aiOnly, includeRaw);
}
/**
* Parse items from a GenericTimelineById response
*/
private parseTimelineTabItems(
// biome-ignore lint/suspicious/noExplicitAny: API response structure is complex
data: any,
source: string,
maxCount: number,
aiOnly: boolean,
includeRaw: boolean,
): NewsItem[] {
const items: NewsItem[] = [];
const seenHeadlines = new Set<string>();
// Navigate to timeline instructions
const timeline = data?.data?.timeline?.timeline;
if (!timeline) {
return [];
}
const instructions = timeline.instructions || [];
for (const instruction of instructions) {
const entries = instruction.entries ?? (instruction.entry ? [instruction.entry] : []);
if (!entries || entries.length === 0) {
continue;
}
for (const entry of entries) {
if (items.length >= maxCount) {
break;
}
const content = entry.content;
if (!content) {
continue;
}
// Handle TimelineTimelineItem (single trend item)
if (content.itemContent) {
const newsItem = this.parseNewsItemFromContent(
content.itemContent,
entry.entryId,
source,
seenHeadlines,
aiOnly,
includeRaw,
);
if (newsItem) {
items.push(newsItem);
}
}
// Handle TimelineTimelineModule (multiple items)
const itemsArray = content?.items || [];
for (const data of itemsArray) {
if (items.length >= maxCount) {
break;
}
// Structure can be data.itemContent OR data.item.itemContent
const itemContent = data?.itemContent || data?.item?.itemContent;
if (!itemContent) {
continue;
}
const newsItem = this.parseNewsItemFromContent(
itemContent,
entry.entryId,
source,
seenHeadlines,
aiOnly,
includeRaw,
);
if (newsItem) {
items.push(newsItem);
}
}
}
}
return items;
}
private parseNewsItemFromContent(
// biome-ignore lint/suspicious/noExplicitAny: API response structure is complex
itemContent: any,
entryId: string,
source: string,
seenHeadlines: Set<string>,
aiOnly: boolean,
includeRaw: boolean,
): NewsItem | null {
const headline = itemContent.name || itemContent.title;
if (!headline) {
return null;
}
const trendMetadata = itemContent?.trend_metadata;
const trendUrl = itemContent.trend_url?.url || trendMetadata?.url?.url;
// Detect AI news by characteristics:
// 1. Full sentence headlines (contains spaces and is longer)
// 2. Has social_context with "News" category
// 3. Or explicitly marked as is_ai_trend
const socialContext = itemContent?.social_context?.text || '';
const hasNewsCategory = socialContext.includes('News') || socialContext.includes('hours ago');
const isFullSentence = headline.split(' ').length >= 5; // AI news are full sentences
const isExplicitlyAiTrend = itemContent.is_ai_trend === true;
const isAiNews = isExplicitlyAiTrend || (isFullSentence && hasNewsCategory);
// Filter AI trends if aiOnly is enabled
if (aiOnly && !isAiNews) {
return null;
}
if (seenHeadlines.has(headline)) {
return null;
}
seenHeadlines.add(headline);
let postCount: number | undefined;
let timeAgo: string | undefined;
let category = 'Trending';
// Parse social context for metadata
const socialCtx = itemContent?.social_context;
if (socialCtx?.text) {
const socialContextText = socialCtx.text;
const parts = socialContextText.split('·').map((s: string) => s.trim());
for (const part of parts) {
if (part.includes('ago')) {
timeAgo = part;
} else if (part.match(POST_COUNT_REGEX)) {
const match = part.match(POST_COUNT_MATCH_REGEX);
if (match) {
let num = Number.parseFloat(match[1]);
const suffix = match[2]?.toUpperCase();
if (suffix === 'K') {
num *= 1000;
} else if (suffix === 'M') {
num *= 1_000_000;
} else if (suffix === 'B') {
num *= 1_000_000_000;
}
postCount = Math.round(num);
}
} else {
category = part;
}
}
}
// Parse trend metadata
if (trendMetadata?.meta_description) {
const metaDesc = trendMetadata.meta_description;
const postMatch = metaDesc.match(POST_COUNT_MATCH_REGEX);
if (postMatch) {
let num = Number.parseFloat(postMatch[1]);
const suffix = postMatch[2]?.toUpperCase();
if (suffix === 'K') {
num *= 1000;
} else if (suffix === 'M') {
num *= 1_000_000;
} else if (suffix === 'B') {
num *= 1_000_000_000;
}
postCount = Math.round(num);
}
}
if (trendMetadata?.domain_context && (category === 'Trending' || category === 'News')) {
category = trendMetadata.domain_context;
}
const item: NewsItem = {
id: trendUrl ?? (entryId ? `${entryId}-${headline}` : `${source}-${headline}`),
headline,
category: isAiNews ? `AI · ${category}` : category,
timeAgo,
postCount,
description: itemContent.description,
url: trendUrl,
};
if (includeRaw) {
item._raw = itemContent;
}
return item;
}
private async enrichWithTweets(items: NewsItem[], tweetsPerItem: number, includeRaw: boolean): Promise<void> {
const debug = process.env.BIRD_DEBUG === '1';
for (const item of items) {
try {
const searchQuery = item.headline;
if (!searchQuery) {
continue;
}
// Use the search method if available (requires search mixin)
if ('search' in this && typeof (this as { search?: unknown }).search === 'function') {
const result = (await (
this as { search: (q: string, c: number, o: { includeRaw: boolean }) => Promise<SearchResult> }
).search(searchQuery, tweetsPerItem, { includeRaw })) as SearchResult;
if (result.success && result.tweets) {
item.tweets = result.tweets;
}
}
} catch {
if (debug) {
console.error('[getNews] Failed to enrich item with tweets:', item.headline);
}
}
}
}
}
return TwitterClientNews;
}