import { test } from "node:test"; import assert from "node:assert/strict"; import { normalizeXTweet, normalizeXTweets, xCreatedAt, xIdOf, xLinks, xText, } from "./xNormalize"; import { parsePost } from "../lib/posts"; import { buildGalleryDlArgs, parseGalleryDlOutput, quoteBigIntegers, timelineUrlFor, } from "./xGalleryDlFetcher"; import { isXCookie, toNetscapeCookieFile } from "./xSessionBroker"; import { collectTweetsFromGraphQL, isTimelineResponseUrl, xPlaywrightFetcher, } from "./xPlaywrightFetcher"; // A gallery-dl `--dump-json` metadata record for a text tweet, in the shape its // twitter extractor documents (flattened author/entities, "YYYY-MM-DD HH:MM:SS" // dates). No network in tests — X is NEVER contacted by the suite. const GALLERY_DL_RECORD = { tweet_id: "1799999999999999999", conversation_id: "1799999999999999999", date: "2026-05-04 17:32:11", content: "a plain text tweet about archives https://t.co/abc", lang: "en", author: { name: "someaccount", nick: "Some Account" }, user: { name: "someaccount", nick: "Some Account" }, favorite_count: 12, retweet_count: 3, reply_count: 1, quote_count: 0, entities: { urls: [ { url: "https://t.co/abc", expanded_url: "https://example.com/article" }, ], }, }; // The legacy/raw GraphQL shape the Playwright fallback sees. Same normalizer. const RAW_GRAPHQL_RECORD = { id_str: "1800000000000000001", conversation_id_str: "1799999999999999999", created_at: "Wed Oct 10 20:19:24 +0000 2018", full_text: "a reply in the same conversation", lang: "en", user: { name: "someaccount", nick: "Some Account" }, in_reply_to_status_id_str: "1799999999999999999", in_reply_to_screen_name: "someaccount", favorite_count: 4, }; test("normalizes a gallery-dl record into a valid Post", () => { const post = normalizeXTweet(GALLERY_DL_RECORD, "xchan"); assert.ok(post); assert.equal(post!.platform, "twitter"); assert.equal(post!.id, "1799999999999999999"); assert.equal(post!.slug, "xchan/1799999999999999999"); assert.equal(post!.author, "someaccount"); assert.equal(post!.authorName, "Some Account"); assert.equal(post!.uploadDate, "20260504"); assert.equal(post!.url, "https://x.com/someaccount/status/1799999999999999999"); assert.equal(post!.isReply, false); assert.equal(post!.isRepost, false); assert.deepEqual(post!.engagement, { likes: 12, reposts: 3, replies: 1, quotes: 0, }); // Survives the on-disk round trip unchanged. assert.deepEqual(parsePost(JSON.parse(JSON.stringify(post))), post); }); test("the SAME normalizer handles the raw GraphQL shape", () => { // This is what makes the Playwright fallback a transport swap rather than a // rewrite: both payloads land on identical Post fields. const post = normalizeXTweet(RAW_GRAPHQL_RECORD, "xchan"); assert.ok(post); assert.equal(post!.id, "1800000000000000001"); assert.equal(post!.isReply, true); assert.equal(post!.replyTo?.id, "1799999999999999999"); assert.equal(post!.threadId, "1799999999999999999"); assert.equal(post!.uploadDate, "20181010"); }); test("keeps 64-bit ids as strings (never through a JSON number)", () => { // 1799999999999999999 > 2^53: parsing it as a number would silently corrupt // the tweet id and break every permalink and dedupe. assert.equal(xIdOf({ tweet_id: "1799999999999999999" }), "1799999999999999999"); assert.ok(!Number.isSafeInteger(Number("1799999999999999999"))); }); test("parses both date dialects to ISO-8601 UTC", () => { assert.equal(xCreatedAt({ date: "2026-05-04 17:32:11" }), "2026-05-04T17:32:11.000Z"); assert.equal( xCreatedAt({ created_at: "Wed Oct 10 20:19:24 +0000 2018" }), "2018-10-10T20:19:24.000Z", ); assert.equal(xCreatedAt({ date: "2026-05-04T17:32:11.000Z" }), "2026-05-04T17:32:11.000Z"); assert.equal(xCreatedAt({}), undefined); }); test("expands outbound links and drops t.co self-links", () => { assert.deepEqual(xLinks(GALLERY_DL_RECORD), ["https://example.com/article"]); assert.deepEqual( xLinks({ entities: { urls: [{ url: "https://t.co/xyz" }] } }), [], ); }); test("prefers a note-tweet body over the truncated text", () => { const long = "x".repeat(400); const rec = { full_text: "x".repeat(140) + "…", note_tweet: { note_tweet_results: { result: { text: long } } }, }; assert.equal(xText(rec), long); }); test("marks a retweet and points repostOf at the original", () => { const post = normalizeXTweet( { ...GALLERY_DL_RECORD, retweeted_status: { id_str: "1700000000000000000", user: { name: "original" }, }, }, "xchan", ); assert.ok(post); assert.equal(post!.isRepost, true); assert.equal(post!.repostOf?.id, "1700000000000000000"); assert.equal(post!.repostOf?.author, "original"); }); // ─── gallery-dl's flat reference fields (regression) ─── // // gallery-dl 1.32.9 does not pass X's in_reply_to_* / retweeted_status through: // it emits `reply_id`, `retweet_id`, `quote_id` (0 when unset) and `reply_to`. // Reading only X's names marked every archived post as neither a reply nor a // repost, and credited each retweet to the account that was retweeted. test("a gallery-dl reply is a reply, with its parent", () => { const post = normalizeXTweet( { ...GALLERY_DL_RECORD, content: "@otheraccount agreed", reply_id: "1799999999999999990", reply_to: "otheraccount", retweet_id: 0, quote_id: 0, }, "xchan", )!; assert.equal(post.isReply, true); assert.equal(post.replyTo?.id, "1799999999999999990"); assert.equal(post.replyTo?.author, "otheraccount"); assert.equal(post.replyTo?.url, "https://x.com/otheraccount/status/1799999999999999990"); assert.equal(post.isRepost, false); }); test("gallery-dl's zeroed references mean none", () => { const post = normalizeXTweet( { ...GALLERY_DL_RECORD, reply_id: 0, retweet_id: 0, quote_id: 0, conversation_id: 0 }, "xchan", )!; assert.equal(post.isReply, false); assert.equal(post.isRepost, false); assert.equal(post.replyTo, undefined); assert.equal(post.repostOf, undefined); assert.equal(post.quoted, undefined); }); test("a gallery-dl retweet is the timeline owner's repost of the original", () => { // Shape from real output: `author` is the retweeted account, `user` the // timeline's owner, and gallery-dl prefixes the body with "RT @author:". const post = normalizeXTweet( { tweet_id: "2058535364466192493", retweet_id: "2058530000000000000", reply_id: 0, quote_id: 0, conversation_id: "2058530000000000000", date: "2026-05-24 13:07:31", content: "RT @originalacct: the original words", author: { name: "originalacct", nick: "Original Account" }, user: { name: "timelineowner", nick: "Timeline Owner" }, favorite_count: 0, retweet_count: 1246, }, "xchan", )!; assert.equal(post.isRepost, true); assert.equal(post.id, "2058535364466192493"); assert.equal(post.author, "timelineowner"); assert.equal(post.authorName, "Timeline Owner"); assert.equal(post.url, "https://x.com/timelineowner/status/2058535364466192493"); assert.equal(post.repostOf?.id, "2058530000000000000"); assert.equal(post.repostOf?.author, "originalacct"); assert.equal(post.text, "RT @originalacct: the original words"); assert.deepEqual(parsePost(JSON.parse(JSON.stringify(post))), post); }); test("gallery-dl's quote_id names the QUOTING tweet, so it is never read as the quoted one", () => { const post = normalizeXTweet( { ...GALLERY_DL_RECORD, quote_id: "1799999999999999995", quote_by: "quoter" }, "xchan", )!; assert.equal(post.quoted, undefined); }); test("X's quoted_status_id_str gives the quoted tweet", () => { const post = normalizeXTweet( { ...RAW_GRAPHQL_RECORD, quoted_status_id_str: "1700000000000000001" }, "xchan", )!; assert.equal(post.quoted?.id, "1700000000000000001"); assert.equal(post.quoted?.url, "https://x.com/i/status/1700000000000000001"); }); test("gallery-dl's unquoted 64-bit reply_id and retweet_id survive parsing", () => { const reply = '{"tweet_id": 2085320225776427457, "reply_id": 2085116299223425392, "retweet_id": 0, "quote_id": 0, "reply_to": "someone", "date": "2026-08-06 11:00:59", "content": "@someone yes", "author": {"name": "NASA"}, "user": {"name": "NASA"}}'; const [r] = parseGalleryDlOutput(reply); const rp = normalizeXTweet(r, "xchan")!; assert.equal(rp.isReply, true); assert.equal(rp.replyTo?.id, "2085116299223425392"); const retweet = '{"tweet_id": 2085320225776427458, "retweet_id": 2085116299223425393, "reply_id": 0, "quote_id": 0, "date": "2026-08-06 11:00:59", "content": "RT @someone: hi", "author": {"name": "someone"}, "user": {"name": "NASA"}}'; const [t] = parseGalleryDlOutput(retweet); const rt = normalizeXTweet(t, "xchan")!; assert.equal(rt.isRepost, true); assert.equal(rt.author, "NASA"); assert.equal(rt.repostOf?.id, "2085116299223425393"); }); test("rejects records with no id or no timestamp", () => { assert.equal(normalizeXTweet({}, "c"), null); assert.equal(normalizeXTweet({ tweet_id: "1" }, "c"), null, "no date -> rejected"); assert.equal( normalizeXTweet({ date: "2026-05-04 17:32:11" }, "c"), null, "no id -> rejected", ); }); test("batch normalize dedupes by id and skips unparseable records", () => { const posts = normalizeXTweets( [GALLERY_DL_RECORD, GALLERY_DL_RECORD, {}, RAW_GRAPHQL_RECORD], "xchan", ); assert.equal(posts.length, 2); }); // ─── gallery-dl invocation contract ─── test("gallery-dl argv enables text-tweets and downloads nothing", () => { const args = buildGalleryDlArgs({ accountUrl: "https://x.com/someaccount", cookies: "firefox", limit: 50, }); const joined = args.join(" "); // Text-only timeline extraction is the whole point — without this flag // gallery-dl skips every tweet that has no media. assert.match(joined, /extractor\.twitter\.text-tweets=true/); assert.ok(args.includes("--no-download"), "v1 archives no media"); assert.ok(args.includes("--dump-json")); assert.deepEqual( args.slice(args.indexOf("--cookies-from-browser"), args.indexOf("--cookies-from-browser") + 2), ["--cookies-from-browser", "firefox"], ); assert.match(joined, /--post-range 1-50/); // `--range` counts FILES: with downloads off it never stopped the walk. assert.ok(!args.includes("--range"), "a limit caps posts, not files"); // Normalized to the timeline sub-extractor — a bare profile URL yields no // tweets at all (see timelineUrlFor). assert.equal(args[args.length - 1], "https://x.com/someaccount/timeline"); }); test("gallery-dl argv omits cookies when none are resolved", () => { const args = buildGalleryDlArgs({ accountUrl: "https://x.com/a" }); assert.ok(!args.includes("--cookies-from-browser")); assert.ok(!args.join(" ").includes("range 1-")); }); test("parses JSON-lines, whole-array and tuple dump forms", () => { const lines = `{"tweet_id":"1","date":"2026-01-01 00:00:00"}\n{"tweet_id":"2","date":"2026-01-02 00:00:00"}`; assert.equal(parseGalleryDlOutput(lines).length, 2); const arr = JSON.stringify([{ tweet_id: "3" }, { tweet_id: "4" }]); assert.equal(parseGalleryDlOutput(arr).length, 2); // gallery-dl also emits [, , ] tuples. const tuples = JSON.stringify([[3, "https://x.com/a/status/5", { tweet_id: "5" }]]); const parsed = parseGalleryDlOutput(tuples); assert.ok(parsed.some((r) => r.tweet_id === "5")); // Progress noise interleaved with JSON must not break parsing. assert.equal( parseGalleryDlOutput(`downloading...\n{"tweet_id":"6"}\ndone`).length, 1, ); assert.deepEqual(parseGalleryDlOutput(""), []); }); // ─── X session broker (cookie jar format) ─── test("exports cookies in the Netscape format gallery-dl reads", () => { const jar = toNetscapeCookieFile([ { name: "auth_token", value: "secret", domain: ".x.com", path: "/", expires: 1893456000, httpOnly: true, secure: true, }, { name: "ct0", value: "csrf", domain: "x.com", path: "/", // A session cookie (-1) must serialize as 0, not as a negative number. expires: -1, httpOnly: false, secure: true, }, ]); const lines = jar .trim() .split("\n") .filter((l) => l.trim() !== "" && !l.startsWith("#")); assert.equal(lines.length, 2); assert.deepEqual(lines[0].split("\t"), [ ".x.com", "TRUE", "/", "TRUE", "1893456000", "auth_token", "secret", ]); assert.deepEqual(lines[1].split("\t"), [ "x.com", "FALSE", "/", "TRUE", "0", "ct0", "csrf", ]); assert.match(jar, /^# Netscape HTTP Cookie File/); }); test("only X's own cookies are exported", () => { const mk = (domain: string) => ({ name: "n", value: "v", domain, path: "/", expires: 0, httpOnly: false, secure: true, }); assert.equal(isXCookie(mk(".x.com")), true); assert.equal(isXCookie(mk("x.com")), true); assert.equal(isXCookie(mk("api.twitter.com")), true); // A profile can hold unrelated cookies; handing those to a subprocess would // leak them for no benefit. assert.equal(isXCookie(mk("google.com")), false); assert.equal(isXCookie(mk("notx.com")), false); }); test("gallery-dl prefers the broker's cookie file over browser extraction", () => { const args = buildGalleryDlArgs({ accountUrl: "https://x.com/a", cookies: "firefox", cookieFile: "/data/.x-session/cookies.txt", }); assert.ok(args.includes("--cookies")); assert.ok( !args.includes("--cookies-from-browser"), "the live-profile jar wins — it is what survives X's short cookie expiry", ); assert.equal(args[args.indexOf("--cookies") + 1], "/data/.x-session/cookies.txt"); }); // ─── X Playwright fallback (GraphQL payload extraction) ─── // // Driven entirely off a recorded-shape payload. The fallback is NEVER pointed // at x.com in tests: the suite must stay deterministic and no run may risk the // logged-in account. test("recognises the timeline GraphQL operations by substring", () => { assert.equal( isTimelineResponseUrl("https://x.com/i/api/graphql/AbC123/UserTweets?variables=%7B%7D"), true, ); assert.equal(isTimelineResponseUrl("https://x.com/i/api/graphql/XyZ/UserTweetsAndReplies"), true); // A query-id rotation must NOT break the match — that is the whole reason // this fetcher exists. assert.equal(isTimelineResponseUrl("https://x.com/i/api/graphql/TOTALLY-NEW-ID/UserTweets"), true); assert.equal(isTimelineResponseUrl("https://x.com/i/api/2/notifications/all.json"), false); }); test("extracts tweets from a nested GraphQL timeline payload", () => { // The real payload buries tweets under // data.user.result.timeline_v2.timeline.instructions[].entries[].content... const payload = { data: { user: { result: { timeline_v2: { timeline: { instructions: [ { type: "TimelineAddEntries", entries: [ { entryId: "tweet-1899000000000000001", content: { itemContent: { tweet_results: { result: { __typename: "Tweet", rest_id: "1899000000000000001", core: { user_results: { result: { legacy: { screen_name: "someaccount", name: "Some Account", }, }, }, }, legacy: { id_str: "1899000000000000001", conversation_id_str: "1899000000000000001", created_at: "Wed Oct 10 20:19:24 +0000 2018", full_text: "a tweet captured through the fallback", lang: "en", favorite_count: 7, }, }, }, }, }, }, { entryId: "cursor-bottom", content: { value: "DAABC" } }, ], }, ], }, }, }, }, }, }; const tweets = collectTweetsFromGraphQL(payload); assert.equal(tweets.length, 1); // The SHARED normalizer turns it into the same Post shape the gallery-dl // path produces — this is what makes the fallback a transport swap. const posts = normalizeXTweets(tweets, "xchan"); assert.equal(posts.length, 1); assert.equal(posts[0].id, "1899000000000000001"); assert.equal(posts[0].author, "someaccount"); assert.equal(posts[0].authorName, "Some Account"); assert.equal(posts[0].text, "a tweet captured through the fallback"); assert.equal(posts[0].platform, "twitter"); assert.equal(posts[0].uploadDate, "20181010"); }); test("GraphQL extraction survives cycles and ignores non-tweet nodes", () => { const cyclic: Record = { data: {} }; cyclic.self = cyclic; assert.deepEqual(collectTweetsFromGraphQL(cyclic), []); assert.deepEqual(collectTweetsFromGraphQL(null), []); assert.deepEqual(collectTweetsFromGraphQL({ data: { user: null } }), []); }); test("the fallback never claims a URL by detection", () => { // gallery-dl is the primary path; the fallback is opted into per channel. assert.equal(xPlaywrightFetcher.detect("https://x.com/someaccount"), false); assert.equal(xPlaywrightFetcher.platform, "twitter"); }); // ─── 64-bit id precision (regression) ─── // // Captured from a real `gallery-dl 1.32.9 --dump-json` run: it emits tweet_id // as an UNQUOTED JSON number. The earlier test here fed a *string* id, so it // passed while the real path silently corrupted every id. test("gallery-dl's unquoted 64-bit tweet_id survives parsing intact", () => { // Verbatim shape from real output (id > 2^53). const line = '{"tweet_id": 2085320225776427457, "date": "2026-08-06 11:00:59", "content": "hi", "author": {"name": "NASA"}}'; // A plain JSON.parse loses precision — this is the bug being guarded. assert.equal(String(JSON.parse(line).tweet_id), "2085320225776427500"); const [rec] = parseGalleryDlOutput(line); assert.equal(rec.tweet_id, "2085320225776427457", "exact id preserved"); const post = normalizeXTweet(rec, "xchan")!; assert.equal(post.id, "2085320225776427457"); assert.equal(post.slug, "xchan/2085320225776427457"); // A corrupted id would 404 and, worse, break the archive dedupe key. assert.match(post.url, /status\/2085320225776427457$/); }); test("quoteBigIntegers only touches long integer JSON values", () => { // Adjacent long ints in an array must BOTH be quoted (delimiter reuse). assert.equal( quoteBigIntegers('{"a":[2085320225776427457,2085116299223425392]}'), '{"a":["2085320225776427457","2085116299223425392"]}', ); // Short numbers, floats and digits inside strings are left alone. assert.equal(quoteBigIntegers('{"n":42,"f":1.5}'), '{"n":42,"f":1.5}'); assert.equal( quoteBigIntegers('{"s":"2085320225776427457"}'), '{"s":"2085320225776427457"}', ); // Still valid JSON afterwards. assert.deepEqual( JSON.parse(quoteBigIntegers('{"id":2085320225776427457,"n":7}')), { id: "2085320225776427457", n: 7 }, ); }); // ─── profile URL must target the timeline sub-extractor ─── test("a bare profile URL is normalized to /timeline", () => { // gallery-dl yields only a type-6 DISPATCH record for a bare profile URL and // no tweets at all — this is what made a real channel sync 0 posts. assert.equal(timelineUrlFor("https://x.com/TheQuartering"), "https://x.com/TheQuartering/timeline"); assert.equal(timelineUrlFor("https://x.com/TheQuartering/"), "https://x.com/TheQuartering/timeline"); assert.equal(timelineUrlFor("https://twitter.com/nasa"), "https://twitter.com/nasa/timeline"); }); test("an explicit sub-route is left alone", () => { for (const u of [ "https://x.com/nasa/timeline", "https://x.com/nasa/with_replies", "https://x.com/nasa/media", "https://x.com/NASA/status/2085320225776427457", ]) { assert.equal(timelineUrlFor(u), u.replace(/\/+$/, "")); } }); test("buildGalleryDlArgs emits the timeline URL, and cookies stay optional", () => { const guest = buildGalleryDlArgs({ accountUrl: "https://x.com/TheQuartering" }); assert.equal(guest[guest.length - 1], "https://x.com/TheQuartering/timeline"); // No credentials configured must NOT be a failure — guest reads work. assert.ok(!guest.includes("--cookies")); assert.ok(!guest.includes("--cookies-from-browser")); });