// X/Twitter via a Nitter instance, driven through a real browser. // // Why this exists as a THIRD X path: the guest-token GraphQL route // (x-gallery-dl without cookies) is throttled hard and returned only a handful // of stale posts for some accounts, while an authenticated route needs an // account to keep alive. A Nitter instance does the talking to X for us, so // there is no account to juggle and no X quota to burn — the archive only ever // talks to the instance. // // Why a BROWSER rather than plain HTTP: nitter.poast.org answers non-browser // clients with a hard `503` (verified — curl and gallery-dl both get it, and // gallery-dl's own Nitter extractor therefore fails), but a real browser gets // the fully rendered page. The browser IS the mechanism that works, so this // fetcher owns one. // // Because of that, the HTTP status is NOT authoritative: poast serves 503 // alongside a perfectly good timeline. Presence of `.timeline-item` is the // real success signal. // // Public Nitter instances are famously short-lived, so the instance list is // configurable and tried in order. import { postPermalink, uploadDateFromCreatedAt, type Post, type PostAvailability, type PostRef, } from "../lib/posts"; import { importPlaywright } from "./playwrightRuntime"; import { registerSocialFetcher, type PostFetchInput, type PostFetchResult, type SocialFetcher, type SocialFetcherProbe, } from "./fetchers"; // Ordered by preference. Override with NITTER_INSTANCES (comma-separated) — // instances die often enough that hardcoding one is a liability. export function nitterInstances(): string[] { const configured = process.env.NITTER_INSTANCES?.trim(); if (configured) { return configured .split(",") .map((s) => s.trim().replace(/\/+$/, "")) .filter(Boolean); } return [ "https://nitter.poast.org", "https://nitter.net", "https://nitter.space", ]; } // Bound how far back one run walks, so a huge account can't spin forever. const MAX_PAGES = 40; // --------------------------------------------------------------------------- // Pure parsing helpers (no DOM, no network — unit-tested against real values) // --------------------------------------------------------------------------- const MONTHS: Record = { Jan: 1, Feb: 2, Mar: 3, Apr: 4, May: 5, Jun: 6, Jul: 7, Aug: 8, Sep: 9, Oct: 10, Nov: 11, Dec: 12, }; // Nitter renders its timestamps as `Aug 6, 2026 · 8:51 PM UTC` in the // `.tweet-date a[title]` attribute. Always UTC, so the conversion is exact // rather than locale-dependent. Returns an ISO-8601 string, or null when the // format drifts (in which case the item is skipped rather than mis-dated). export function parseNitterDate(title: string): string | null { const m = /^([A-Z][a-z]{2})\s+(\d{1,2}),\s+(\d{4})\s*·\s*(\d{1,2}):(\d{2})\s*(AM|PM)\s*UTC$/.exec( title.trim(), ); if (!m) return null; const [, mon, day, year, hourRaw, minute, meridiem] = m; const month = MONTHS[mon]; if (!month) return null; let hour = Number(hourRaw) % 12; if (meridiem === "PM") hour += 12; const iso = `${year}-${String(month).padStart(2, "0")}-${String(Number(day)).padStart(2, "0")}T${String(hour).padStart(2, "0")}:${minute}:00.000Z`; return Number.isFinite(Date.parse(iso)) ? iso : null; } // `/TheQuartering/status/2085468926012702808#m` -> the id, AS A STRING. // X ids exceed 2^53, so they must never round-trip through a number. export function nitterIdFromHref(href: string): string | null { const m = /\/status\/(\d+)/.exec(href); return m ? m[1] : null; } // `/TheQuartering/status/123#m` -> `TheQuartering` export function nitterHandleFromHref(href: string): string | null { const m = /^\/([^/]+)\/status\/\d+/.exec(href); return m ? m[1] : null; } // Nitter prints counts with thousands separators ("7,602") and an empty string // when a stat is zero/absent. export function parseNitterStat(raw: string | null | undefined): number | undefined { if (!raw) return undefined; const cleaned = raw.replace(/[,\s]/g, ""); if (!/^\d+$/.test(cleaned)) return undefined; return Number(cleaned); } // The `?cursor=…` of the "Load more" link, which is how Nitter paginates. export function cursorFromShowMore(href: string | null | undefined): string | null { if (!href) return null; const m = /[?&]cursor=([^&]+)/.exec(href); return m ? decodeURIComponent(m[1]) : null; } // One `.timeline-item`, already flattened out of the DOM by the page-side // scraper below. Kept as a plain shape so normalization is pure and testable. export type NitterRawItem = { href: string | null; dateTitle: string | null; text: string | null; fullname: string | null; username: string | null; // "@handle" isRetweet: boolean; isReply: boolean; replyingTo: string[]; // handles, without "@" quoteHref: string | null; stats: { icon: string; value: string }[]; mediaCount: number; }; export function normalizeNitterItem( item: NitterRawItem, channelSlug: string, ): Post | null { const href = item.href ?? ""; const id = nitterIdFromHref(href); if (!id) return null; const createdAt = item.dateTitle ? parseNitterDate(item.dateTitle) : null; if (!createdAt) return null; const author = (item.username ?? "").replace(/^@/, "").trim() || nitterHandleFromHref(href) || ""; const stat = (icon: string) => parseNitterStat(item.stats.find((s) => s.icon.includes(icon))?.value); const post: Post = { id, slug: `${channelSlug}/${id}`, channelSlug, author, createdAt, uploadDate: uploadDateFromCreatedAt(createdAt), text: (item.text ?? "").trim(), // Always link to x.com, never to the Nitter instance: the instance is a // transport detail and is likely to be dead by the time anyone clicks. url: postPermalink("twitter", author, id), platform: "twitter", isReply: item.isReply, isRepost: item.isRetweet, links: [], threadId: id, }; if (item.fullname) post.authorName = item.fullname; if (item.mediaCount > 0) post.mediaCount = item.mediaCount; const parent = item.replyingTo[0]; if (parent) { // Nitter names the handle being replied to but not the parent's id, so the // ref carries the author only — enough to render "replying to @x". const ref: PostRef = { platform: "twitter", id: "", author: parent }; if (ref.author) post.replyTo = ref; } const quotedId = item.quoteHref ? nitterIdFromHref(item.quoteHref) : null; if (quotedId) { const quotedAuthor = item.quoteHref ? nitterHandleFromHref(item.quoteHref) : null; post.quoted = { platform: "twitter", id: quotedId, ...(quotedAuthor ? { author: quotedAuthor, url: postPermalink("twitter", quotedAuthor, quotedId) } : {}), }; } const engagement: NonNullable = {}; const replies = stat("comment"); const reposts = stat("retweet"); const likes = stat("heart"); const quotes = stat("quote"); if (likes !== undefined) engagement.likes = likes; if (reposts !== undefined) engagement.reposts = reposts; if (replies !== undefined) engagement.replies = replies; if (quotes !== undefined) engagement.quotes = quotes; if (Object.keys(engagement).length > 0) post.engagement = engagement; return post; } // The page-side scraper, as a string so it can be handed to page.evaluate() // without pulling DOM types into this module. Returns NitterRawItem[] plus the // next cursor. const SCRAPE = `(() => { const items = [...document.querySelectorAll('.timeline-item')].map((el) => ({ href: el.querySelector('a.tweet-link')?.getAttribute('href') ?? null, dateTitle: el.querySelector('.tweet-date a')?.getAttribute('title') ?? null, text: el.querySelector('.tweet-content')?.textContent ?? null, fullname: el.querySelector('a.fullname')?.getAttribute('title') ?? null, username: el.querySelector('a.username')?.getAttribute('title') ?? null, isRetweet: !!el.querySelector('.retweet-header'), isReply: !!el.querySelector('.replying-to'), replyingTo: [...el.querySelectorAll('.replying-to a')].map((a) => (a.textContent || '').replace(/^@/, '').trim()).filter(Boolean), quoteHref: el.querySelector('.quote a.quote-link')?.getAttribute('href') ?? el.querySelector('.quote .tweet-link')?.getAttribute('href') ?? null, stats: [...el.querySelectorAll('.tweet-stat')].map((s) => ({ // The icon is the inner SPAN (span.icon-comment); the wrapping // div.icon-container also matches [class*="icon-"] and would shadow it, // which silently produced zero engagement stats. icon: (s.querySelector('span[class*="icon-"]')?.className) || '', value: (s.textContent || '').trim(), })), mediaCount: el.querySelectorAll('.attachments .attachment').length, })); const more = [...document.querySelectorAll('.show-more a')].pop(); return { items, next: more ? more.getAttribute('href') : null }; })()`; export const xNitterFetcher: SocialFetcher = { id: "x-nitter", label: "X / Twitter (Nitter, no account)", platform: "twitter", fields: { limit: true }, // Claims Nitter URLs only. An x.com channel opts in explicitly via // `postFetcher: "x-nitter"`, so URL detection keeps preferring gallery-dl. detect(url: string): boolean { try { return /(^|\.)nitter\./i.test(new URL(url).hostname); } catch { return false; } }, async probe(url: string): Promise { const handle = handleOf(url); if (!handle) { return { ok: false, error: "Could not read an X handle from that URL" }; } // No network: a probe would cost a full browser launch for a handle we can // already read off the URL. return { ok: true, name: handle, handle, url: `https://x.com/${handle}` }; }, async fetch(input: PostFetchInput): Promise { const { accountUrl, handle: rawHandle, channelSlug, since, seenIds, limit, signal, onLog } = input; const handle = rawHandle || handleOf(accountUrl); if (!handle) throw new Error("No X handle to fetch"); const { chromium } = await importPlaywright(); const browser = await chromium.launch({ headless: true }); const posts: Post[] = []; let complete = false; let lastError = ""; try { const ctx = await browser.newContext({ userAgent: "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", }); const page = await ctx.newPage(); for (const instance of nitterInstances()) { if (signal.aborted) break; posts.length = 0; let cursor: string | null = input.cursor ?? null; let pages = 0; let ok = false; try { for (; pages < MAX_PAGES; pages++) { if (signal.aborted) break; const url = `${instance}/${handle}${cursor ? `?cursor=${encodeURIComponent(cursor)}` : ""}`; await page.goto(url, { waitUntil: "domcontentloaded", timeout: 45_000 }); // The HTTP status is NOT the success signal — poast serves 503 // alongside a good timeline. Wait for real content instead. await page .waitForSelector(".timeline-item", { timeout: 20_000 }) .catch(() => {}); const { items, next } = (await page.evaluate(SCRAPE)) as { items: NitterRawItem[]; next: string | null; }; if (items.length === 0) { if (pages === 0) throw new Error("no timeline items rendered"); complete = true; break; } ok = true; let stop = false; for (const raw of items) { const post = normalizeNitterItem(raw, channelSlug); if (!post) continue; if (seenIds.has(post.id)) { stop = true; continue; } if (since && post.createdAt <= since) { stop = true; continue; } posts.push(post); } onLog?.( `Nitter page ${pages + 1} (${instanceHost(instance)}): ${items.length} items, ${posts.length} new so far`, ); if (stop) { complete = true; break; } if (limit && posts.length >= limit) { posts.length = limit; break; } cursor = cursorFromShowMore(next); if (!cursor) { complete = true; break; } } } catch (err) { lastError = (err as Error).message; onLog?.( `Nitter instance ${instanceHost(instance)} failed: ${lastError} — trying the next one.`, ); continue; } if (ok) { return { posts, complete, ...(cursor && !complete ? { cursor } : {}) }; } } } finally { await browser.close().catch(() => {}); } throw new Error( `No Nitter instance returned a timeline for @${handle}` + (lastError ? ` (last error: ${lastError})` : "") + ". Public instances are short-lived — set NITTER_INSTANCES to a working one.", ); }, }; function instanceHost(instance: string): string { try { return new URL(instance).hostname; } catch { return instance; } } function handleOf(url: string): string | null { const trimmed = (url ?? "").trim(); if (!trimmed) return null; if (!/^https?:\/\//i.test(trimmed)) return trimmed.replace(/^@/, "") || null; try { const segments = new URL(trimmed).pathname.split("/").filter(Boolean); return segments[0] ? decodeURIComponent(segments[0]).replace(/^@/, "") : null; } catch { return null; } } // A deleted tweet's Nitter status page renders an error rather than a tweet, // so presence of `.main-tweet .tweet-content` is the liveness signal. One page // load per post makes this much heavier than Bluesky's batch lookup, so the // controller caps how many it checks per run. xNitterFetcher.checkAvailability = async ({ ids, handle, signal, onLog }) => { const out = new Map(); if (ids.length === 0) return out; const { chromium } = await importPlaywright(); const browser = await chromium.launch({ headless: true }); try { const ctx = await browser.newContext({ userAgent: "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36", }); const page = await ctx.newPage(); const [instance] = nitterInstances(); for (const id of ids) { if (signal.aborted) break; try { await page.goto(`${instance}/${handle}/status/${id}`, { waitUntil: "domcontentloaded", timeout: 30_000, }); const verdict = (await page.evaluate(`(() => { const hasTweet = !!document.querySelector('.main-tweet .tweet-content, .conversation .tweet-content'); const body = (document.body.textContent || '').toLowerCase(); if (hasTweet) return 'available'; if (/tweet not found|not found|deleted|no longer exists/.test(body)) return 'deleted'; if (/protected|suspended|does not exist|user not found/.test(body)) return 'account_unavailable'; return 'error'; })()`)) as PostAvailability; out.set(id, verdict); } catch { // A failed load is not evidence of deletion. out.set(id, "error"); } } } finally { await browser.close().catch(() => {}); } onLog?.( `Checked ${out.size} post(s): ${[...out.values()].filter((v) => v === "deleted").length} deleted.`, ); return out; }; registerSocialFetcher(xNitterFetcher);