// The X/Twitter FALLBACK fetcher: drive an authenticated Chromium and read the // timeline from inside the page. // // Why it exists: gallery-dl hardcodes X's GraphQL query IDs, so it breaks every // time X rotates them (historically every 2–4 weeks) until upstream catches up. // A real browser is immune to that specific failure — X's own JavaScript // computes the current query IDs, and the page's own `fetch` carries the bearer // token, CSRF header and cookies. So this is a TRANSPORT swap, not a rewrite: // the payload is the same tweet shape gallery-dl parses, and the `Post` // normalizer (xNormalize.ts) is shared verbatim with the primary fetcher. // // Known costs, accepted deliberately: // - Playwright Chromium's JA3/TLS fingerprint matches no real Chrome release // (plus the HeadlessChrome user agent and CDP artifacts; navigator.webdriver // is off since release 16 slice XL, xBrowser.ts), so X CAN detect it. // - Ban risk on the logged-in account is higher than an offline cookie read. // - A browser process per fetch is slow and RAM-hungry. // - It will NOT run in the minimal Docker build container — post fetching // stays an editor-host concern, never a build-time one. // Which is exactly why it is the FALLBACK: registered after x-gallery-dl, so // URL detection prefers the cheap subprocess, and this is opted into per // channel via `postFetcher: "x-playwright"`. // // Strategy: response INTERCEPTION (scroll the profile, capture UserTweets JSON) // rather than re-issuing GraphQL by hand. It is the simpler of the two variants // in the design and needs no knowledge of X's endpoint names beyond a substring // match, so it survives more drift. // // NOT VERIFIED AGAINST LIVE X — it is tested against a local fixture page // serving a recorded payload, never x.com, so the suite stays deterministic and // no test run risks the account. import type { Post } from "../lib/posts"; import { normalizeXTweets, type XTweetRaw } from "./xNormalize"; import { launchXProfile } from "./xSessionBroker"; import { getPaths } from "../lib/paths"; import { importPlaywright, type BrowserContextLike, type BrowserLike, } from "./playwrightRuntime"; import { buildXBrowserLaunchOptions } from "./xBrowser"; import { isXComAuth, xCookiesFromBrowser } from "./xBrowserLogin"; import { registerSocialFetcher, type PostFetchInput, type PostFetchResult, type SocialFetcher, type SocialFetcherProbe, } from "./fetchers"; // The GraphQL operations that carry timeline tweets. Matched as substrings of // the response URL, so a version suffix or a renamed query id still hits. const TIMELINE_OPS = [ "UserTweets", "UserTweetsAndReplies", "UserMedia", "UserWithProfileTweetsQueryV2", ]; export function isTimelineResponseUrl(url: string): boolean { return TIMELINE_OPS.some((op) => url.includes(op)); } // Pull every tweet-shaped object out of a GraphQL timeline response. X nests // them several layers deep and has moved them repeatedly, so rather than // walking a fixed path this recurses and collects anything that looks like a // tweet result. Tolerant by construction — the shape is a moving target. export function collectTweetsFromGraphQL(payload: unknown): XTweetRaw[] { const out: XTweetRaw[] = []; const seen = new Set(); const visit = (node: unknown): void => { if (!node || typeof node !== "object") return; if (seen.has(node)) return; seen.add(node); if (Array.isArray(node)) { for (const item of node) visit(item); return; } const rec = node as Record; // A tweet result: `__typename: "Tweet"` with a legacy block, or a bare // legacy object carrying the tweet fields. const legacy = rec.legacy as Record | undefined; if (rec.__typename === "Tweet" && legacy && typeof legacy === "object") { out.push(mergeTweet(rec, legacy)); // Claim the legacy block so the bare-legacy branch below doesn't emit // the same tweet a second time when the walk descends into it. seen.add(legacy); } else if ( typeof rec.full_text === "string" && (typeof rec.id_str === "string" || typeof rec.conversation_id_str === "string") ) { out.push(rec); } for (const value of Object.values(rec)) visit(value); }; visit(payload); return out; } // Flatten a `{ rest_id, legacy, core }` tweet result into the flat record shape // the shared normalizer understands. function mergeTweet( rec: Record, legacy: Record, ): XTweetRaw { const merged: XTweetRaw = { ...legacy }; if (typeof rec.rest_id === "string") merged.rest_id = rec.rest_id; if (rec.note_tweet) merged.note_tweet = rec.note_tweet; // The author lives under core.user_results.result.legacy.screen_name. const core = rec.core as Record | undefined; const userResults = core?.user_results as Record | undefined; const userResult = userResults?.result as Record | undefined; const userLegacy = userResult?.legacy as Record | undefined; if (userLegacy) { merged.user = { name: userLegacy.screen_name, nick: userLegacy.name, }; } return merged; } export const xPlaywrightFetcher: SocialFetcher = { id: "x-playwright", label: "X / Twitter (Playwright fallback)", platform: "twitter", fields: { limit: true }, // Never claims a URL by detection: gallery-dl is the primary path and is // registered first, and this is opted into explicitly per channel. Returning // false keeps it from ever being auto-selected. detect(): boolean { return false; }, async probe(url: string): Promise { // The probe would cost a full browser launch against x.com for very little // information, and every such visit carries ban risk. The account URL alone // is enough to configure the channel. const handle = /(?:x|twitter)\.com\/([^/?#]+)/i.exec(url)?.[1]; if (!handle) { return { ok: false, error: "Could not read an X handle from that URL" }; } return { ok: true, name: handle, handle, url: `https://x.com/${handle}`, }; }, async fetch(input: PostFetchInput): Promise { const { accountUrl, channelSlug, since, seenIds, limit, signal, onLog } = input; const paths = getPaths(); // The login source (xCookieSource.ts). "browser": a fresh headless // context carrying the operator's browser cookies, read now — the browser // store this repo reads itself is Firefox's (xBrowserLogin.ts); any other // is refused by name rather than run logged out. Otherwise the broker's // persistent profile, as before. let context: BrowserContextLike; let browser: BrowserLike | undefined; if (input.cookieSource === "browser") { const read = await xCookiesFromBrowser(input.browserCookies); if (!read.ok) { throw new Error(`The X login source is the browser, and it cannot be read here: ${read.message}`); } const hasAuth = read.cookies.some(isXComAuth); onLog?.( `Launching a headless browser with ${read.cookies.length} X cookie(s) from ${read.browser} ` + `(fallback path)${hasAuth ? "" : " — WARNING: no auth_token for x.com, the browser is not logged in to X"}.`, ); const { chromium } = await importPlaywright(); browser = await chromium.launch( buildXBrowserLaunchOptions({ browser: { kind: "bundled" }, headless: true }), ); context = await browser.newContext({}); await context.addCookies( read.cookies.map(({ lastAccessedMs: _unused, ...c }) => c), ); } else { onLog?.("Launching the authenticated browser profile (fallback path)."); context = await launchXProfile(paths, { onLog }); } const captured: XTweetRaw[] = []; try { const page = context.pages()[0] ?? (await context.newPage()); // Response interception: X's own JS issues the timeline queries with the // CURRENT query ids, so nothing here needs to know them. page.on("response", (res: { url: () => string; json: () => Promise }) => { const url = res.url(); if (!isTimelineResponseUrl(url)) return; void res .json() .then((body) => { captured.push(...collectTweetsFromGraphQL(body)); }) .catch(() => { /* a non-JSON or already-consumed response — ignore */ }); }); await page.goto(accountUrl, { waitUntil: "domcontentloaded", timeout: 60_000, }); // Scroll until we stop learning anything new, hit the cap, or are // cancelled. Bounded so a large account can't spin forever. const MAX_SCROLLS = 60; let lastCount = -1; let idleRounds = 0; for (let i = 0; i < MAX_SCROLLS; i++) { if (signal.aborted) break; if (limit && captured.length >= limit) break; await page.evaluate("window.scrollBy(0, document.body.scrollHeight)"); await page.waitForTimeout(1200); if (captured.length === lastCount) { if (++idleRounds >= 3) break; // timeline exhausted } else { idleRounds = 0; lastCount = captured.length; } } onLog?.(`Captured ${captured.length} tweet record(s) from the timeline.`); } finally { await context.close().catch(() => {}); await browser?.close().catch(() => {}); } // From here on it is identical to the gallery-dl path — same normalizer, // same stop conditions. const all = normalizeXTweets(captured, channelSlug); const posts: Post[] = []; for (const post of all) { if (seenIds.has(post.id)) continue; if (since && post.createdAt <= since) continue; posts.push(post); } const capped = Boolean(limit && posts.length >= limit); if (capped) posts.length = limit!; return { posts, complete: !capped }; }, }; registerSocialFetcher(xPlaywrightFetcher);