// gallery-dl's argv per X login source (release 16 slice XL). The rest of // gallery-dl's argv and parsing is covered in xNormalize.test.ts. // // Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xGalleryDlFetcher.test.ts import { test } from "node:test"; import assert from "node:assert/strict"; import { buildGalleryDlArgs, buildGalleryDlCaptureArgs, CAPTURE_PRINT_TAG, galleryDlCookieChoice, parseCapturePrints, xRequestPauseMs, OLDER_WINDOW_PAUSE_MAX_MS, OLDER_WINDOW_PAUSE_MIN_MS, olderWindowPauseMs, X_SLEEP_REQUEST, } from "./xGalleryDlFetcher"; const ACCOUNT = "https://x.com/someaccount"; const JAR = "/corpus/.x-session/cookies.txt"; function argvFor(choice: ReturnType): string[] { return buildGalleryDlArgs({ accountUrl: ACCOUNT, cookies: choice.cookies, cookieFile: choice.cookieFile, }); } function flagValue(argv: string[], flag: string): string | undefined { const i = argv.indexOf(flag); return i >= 0 ? argv[i + 1] : undefined; } test("browser source: --cookies-from-browser with the spec, never the jar", () => { const argv = argvFor( galleryDlCookieChoice({ source: "browser", browserCookies: "firefox", jarFile: JAR }), ); assert.equal(flagValue(argv, "--cookies-from-browser"), "firefox"); assert.equal(argv.includes("--cookies"), false); assert.equal(argv[argv.length - 1], "https://x.com/someaccount/timeline"); }); test("browser source: the spec is passed verbatim (profile, container, gallery-dl's /DOMAIN)", () => { for (const spec of ["firefox:abc.default-release", "firefox::Work", "chromium+gnomekeyring:Default", "firefox/.x.com"]) { const argv = argvFor(galleryDlCookieChoice({ source: "browser", browserCookies: spec })); assert.equal(flagValue(argv, "--cookies-from-browser"), spec); } }); test("browser source: passed whatever the cookieMode — the source governs X", () => { // No "always"-mode spec (cookieMode when-required or defer), still passed. const choice = galleryDlCookieChoice({ source: "browser", browserCookies: "firefox", alwaysCookies: undefined, }); assert.equal(choice.cookies, "firefox"); assert.match(choice.note, /the browser \(firefox\)/); }); test("browser source with no spec: a guest run that says why", () => { const choice = galleryDlCookieChoice({ source: "browser", browserCookies: " ", jarFile: JAR }); const argv = argvFor(choice); assert.equal(argv.includes("--cookies-from-browser"), false); assert.equal(argv.includes("--cookies"), false); assert.match(choice.note, /cookiesFromBrowser is empty/); }); test("profile source: the jar, and no --cookies-from-browser", () => { const argv = argvFor( galleryDlCookieChoice({ source: "profile", browserCookies: "firefox", jarFile: JAR, alwaysCookies: "firefox" }), ); assert.equal(flagValue(argv, "--cookies"), JAR); assert.equal(argv.includes("--cookies-from-browser"), false); }); test("profile source with no logged-in jar: the cookieMode \"always\" spec, else a guest", () => { const always = argvFor(galleryDlCookieChoice({ source: "profile", browserCookies: "firefox", alwaysCookies: "firefox" })); assert.equal(flagValue(always, "--cookies-from-browser"), "firefox"); const guest = galleryDlCookieChoice({ source: "profile", browserCookies: "firefox" }); const argv = argvFor(guest); assert.equal(argv.includes("--cookies-from-browser"), false); assert.equal(argv.includes("--cookies"), false); assert.match(guest.note, /guest/); }); test("no source resolved (a caller from before the choice): the jar, else the always-mode spec", () => { assert.equal(flagValue(argvFor(galleryDlCookieChoice({ jarFile: JAR, alwaysCookies: "firefox" })), "--cookies"), JAR); assert.equal( flagValue(argvFor(galleryDlCookieChoice({ alwaysCookies: "firefox" })), "--cookies-from-browser"), "firefox", ); }); // --------------------------------------------------------------------------- // Streaming, resume and early stop (gallery-dl output read as it runs) // --------------------------------------------------------------------------- import { GalleryDlStream, RESTART_FROM_TOP_CURSOR, STOP_AFTER_KNOWN, } from "./xGalleryDlFetcher"; function tweet(n: number, date = "2026-06-01 12:00:00") { // Ids above 2^53, as X's are, so a numeric parse would show. const id = String(1899000000000000000n + BigInt(n)); return { tweet_id: id, conversation_id: id, date, content: `fake tweet ${n}`, author: { name: "faketester", nick: "Fake" }, user: { name: "faketester", nick: "Fake" }, }; } // gallery-dl's jsonl: a [type, kwdict] (or [type, url, kwdict]) tuple per line, // with the id as a bare (unquoted) number. const jsonl = (n: number, date?: string) => JSON.stringify([2, tweet(n, date)]).replace(/"tweet_id":"(\d+)"/, '"tweet_id":$1'); function newStream(opts: Partial[0]> = {}) { return new GalleryDlStream({ channelSlug: "fake-x", seenIds: new Set(), stopAtKnown: true, ...opts, }); } test("streaming argv: jsonl always; --verbose and the cursor only when asked", () => { const probe = buildGalleryDlArgs({ accountUrl: ACCOUNT, limit: 1 }); assert.equal(probe.includes("output.jsonl=true"), true); assert.equal(probe.includes("--verbose"), false); assert.equal(probe.some((a) => a.startsWith("extractor.twitter.cursor=")), false); const run = buildGalleryDlArgs({ accountUrl: ACCOUNT, cursor: "2_123/DAAB", stream: true }); assert.equal(run.includes("--verbose"), true); assert.equal(flagValue(run, "-o") !== undefined, true); assert.equal(run.includes("extractor.twitter.cursor=2_123/DAAB"), true); assert.equal(run[run.length - 1], "https://x.com/someaccount/timeline"); }); test("stream: records parse from jsonl tuples, keep big ids exact, count each tweet once", () => { const s = newStream(); s.pushStdout(jsonl(1)); s.pushStdout(JSON.stringify([3, "https://pbs.twimg.com/x.jpg", tweet(1)])); // same tweet, media record s.pushStdout(jsonl(2)); const posts = s.finish(); assert.equal(s.distinct, 2); assert.equal(posts.length, 2); // The id survives exactly (a JSON.parse of the bare number would round it). assert.ok(posts.some((p) => p.id.includes("1899000000000000001")), posts.map((p) => p.id).join(",")); }); test("stream: a checkpoint carries the PREVIOUS cursor; the final cursor is the latest", () => { const s = newStream(); s.pushStdout(jsonl(1)); s.pushStderr("[twitter][debug] Cursor: 1/AAA"); // Only one cursor so far: nothing is provably covered yet. assert.equal(s.takeCheckpoint(), null); s.pushStdout(jsonl(2)); s.pushStderr("[twitter][debug] Cursor: 1/BBB"); const cp = s.takeCheckpoint(); assert.equal(cp?.cursor, "1/AAA"); assert.equal(cp?.posts.length, 2); // Nothing new since: no second checkpoint. assert.equal(s.takeCheckpoint(), null); assert.equal(s.cursor, "1/BBB"); }); test("stream: debug noise is dropped; info lines are logged and rate limits noticed", () => { const logged: string[] = []; const s = newStream({ onLog: (l) => logged.push(l) }); assert.equal( s.pushStderr("[urllib3.connectionpool][debug] https://x.com:443 \"GET /i/api HTTP/1.1\" 200"), undefined, ); // A wait on the rate limit, and a page's cursor, are each a point between // pages — where a drain may stop the run. assert.equal( s.pushStderr("[twitter][info] Waiting for 14 minutes until 15:11:22 (rate limit)"), "rate-limit", ); assert.equal(s.pushStderr("[twitter][debug] Cursor: 1/ABC"), "cursor"); assert.equal(s.pushStderr("[twitter][info] something else"), undefined); assert.equal(s.rateLimited, true); assert.deepEqual(logged, [ "gallery-dl: [twitter][info] Waiting for 14 minutes until 15:11:22 (rate limit)", "gallery-dl: [twitter][info] something else", ]); assert.match(s.stderrTail(), /rate limit/); }); test("stream: stops after STOP_AFTER_KNOWN archived-or-older posts IN A ROW, not before", () => { // Archive ids 0..2*STOP_AFTER_KNOWN+10 (as normalized post ids). const archived = newStream(); for (let n = 0; n <= 2 * STOP_AFTER_KNOWN + 10; n++) archived.pushStdout(jsonl(n)); const seen = new Set(archived.finish().map((p) => p.id)); const s = newStream({ seenIds: seen }); s.pushStdout(jsonl(9000)); // new for (let n = 0; n < STOP_AFTER_KNOWN - 1; n++) s.pushStdout(jsonl(n)); // 99 known assert.equal(s.stopReached, false); s.pushStdout(jsonl(9001)); // a new post resets the run for (let n = STOP_AFTER_KNOWN; n < 2 * STOP_AFTER_KNOWN - 1; n++) s.pushStdout(jsonl(n)); // 99 known assert.equal(s.stopReached, false); s.pushStdout(jsonl(2 * STOP_AFTER_KNOWN)); // the 100th in a row assert.equal(s.stopReached, true); assert.equal(s.finish().length, 2); }); test("stream: posts older than the watermark are not new; a --full run never stops early", () => { const s = newStream({ since: "2026-05-01T00:00:00.000Z", stopAtKnown: false }); s.pushStdout(jsonl(1, "2026-06-01 12:00:00")); for (let n = 2; n < STOP_AFTER_KNOWN + 10; n++) s.pushStdout(jsonl(n, "2026-04-01 12:00:00")); assert.equal(s.stopReached, false); assert.equal(s.older, STOP_AFTER_KNOWN + 8); assert.equal(s.finish().length, 1); }); test("stream: a gallery-dl that ignores output.jsonl (one pretty array at exit) still parses", () => { const s = newStream(); const pretty = JSON.stringify([[2, tweet(1)], [2, tweet(2)]], null, 2); for (const line of pretty.split("\n")) s.pushStdout(line); assert.equal(s.finish().length, 2); }); test("restart-from-top cursor is the timeline extractor's first state", () => { assert.equal(RESTART_FROM_TOP_CURSOR, "1/"); }); // --------------------------------------------------------------------------- // Search: the older-posts backfill's argv // --------------------------------------------------------------------------- import { buildGalleryDlSearchArgs, xSearchHandle, xSearchQuery, xSearchUrl, } from "./xGalleryDlFetcher"; // What gallery-dl's search extractor does with the URL's q: "+" to a space, // THEN unquote (extractor/twitter.py, TwitterSearchExtractor.tweets). function galleryDlQueryOf(url: string): string { const m = /\/search\/?\?(?:[^&#]+&)*q=([^&#]+)/.exec(url); assert.ok(m, url); return decodeURIComponent(m[1].replace(/\+/g, " ")); } test("search query: from/since/until, reposts included, max_id only when resuming", () => { const window = { since: "2020-12-11", until: "2021-03-11" }; assert.equal( xSearchQuery("example_user", window), "from:example_user since:2020-12-11 until:2021-03-11 include:nativeretweets", ); assert.equal( xSearchQuery("@example_user", { ...window, maxId: "1899000000000000001" }), "from:example_user since:2020-12-11 until:2021-03-11 include:nativeretweets max_id:1899000000000000001", ); assert.throws(() => xSearchQuery("example_user", { ...window, maxId: "12 OR x" }), /not a post id/); }); test("search handle: anything search would read as more than a name is refused", () => { assert.equal(xSearchHandle(" @Example_User1 "), "Example_User1"); for (const bad of ["example user", "example:user", 'ex"ample', "example.user", "", "a-b"]) { assert.throws(() => xSearchHandle(bad), /not an X handle/, bad); } }); test("search URL: the query survives gallery-dl's decoding exactly, spaces and colons included", () => { const query = xSearchQuery("example_user", { since: "2020-12-11", until: "2021-03-11", maxId: "1899000000000000001", }); const url = xSearchUrl(query); assert.match(url, /^https:\/\/x\.com\/search\?q=from%3Aexample_user%20since%3A2020-12-11%20/); assert.ok(url.endsWith("&f=live")); assert.equal(url.includes(" "), false); assert.equal(galleryDlQueryOf(url), query); // An encoded "+" would stay a "+", not become a space. assert.equal(galleryDlQueryOf(xSearchUrl("a+b c")), "a+b c"); }); test("search argv: the timeline's flags and login, latest-first max_id paging, no cursor", () => { const argv = buildGalleryDlSearchArgs({ handle: "example_user", position: { since: "2020-12-11", until: "2021-03-11" }, cookieFile: JAR, limit: 50, }); assert.equal(flagValue(argv, "--cookies"), JAR); assert.equal(flagValue(argv, "--post-range"), "1-50"); for (const opt of [ "output.jsonl=true", "extractor.twitter.text-tweets=true", "extractor.twitter.retweets=true", "extractor.twitter.replies=true", "extractor.twitter.search-results=latest", "extractor.twitter.search-pagination=max_id", "extractor.twitter.sleep-request=4.0-10.0", "extractor.twitter.ratelimit=wait", ]) { assert.ok(argv.includes(opt), opt); } assert.equal(argv.some((a) => a.startsWith("extractor.twitter.cursor=")), false); assert.equal(argv.includes("--no-download"), true); assert.equal( galleryDlQueryOf(argv[argv.length - 1]), "from:example_user since:2020-12-11 until:2021-03-11 include:nativeretweets", ); }); test("stream: tracks the oldest post id read (any post, archived or not) and the account's creation time", () => { const known = String(1899000000000000000n + 1n); const s = newStream({ seenIds: new Set([`${known}`]), stopAtKnown: false, accountHandle: "FakeTester" }); s.pushStdout(jsonl(5)); s.pushStdout(jsonl(1)); // archived already: still the oldest read s.pushStdout(jsonl(3)); assert.equal(s.oldestId, known); assert.equal(s.accountCreatedAt, undefined); s.pushStdout( JSON.stringify([2, { ...tweet(2), user: { name: "faketester", date: "2009-03-10 18:00:00" } }]) .replace(/"tweet_id":"(\d+)"/, '"tweet_id":$1'), ); assert.equal(s.accountCreatedAt, "2009-03-10T18:00:00.000Z"); assert.equal(s.takePending().length, 3); assert.equal(s.pendingCount(), 0); }); test("every X read paces itself: a random 4–10 s before each request, rate limits waited out", () => { for (const argv of [ buildGalleryDlArgs({ accountUrl: "https://x.com/example_user" }), buildGalleryDlSearchArgs({ handle: "example_user", position: { since: "2020-12-11", until: "2021-03-11" } }), ]) { assert.ok(argv.includes(`extractor.twitter.sleep-request=${X_SLEEP_REQUEST}`)); assert.ok(argv.includes("extractor.twitter.ratelimit=wait")); } const [lo, hi] = X_SLEEP_REQUEST.split("-").map(Number); assert.ok(lo >= 4 && hi > lo, "a range, not a fixed gap"); }); test("the pause between search windows is a fresh random draw in 45–120 s", () => { assert.equal(olderWindowPauseMs(() => 0), OLDER_WINDOW_PAUSE_MIN_MS); assert.equal(olderWindowPauseMs(() => 1), OLDER_WINDOW_PAUSE_MAX_MS); const mid = olderWindowPauseMs(() => 0.5); assert.ok(mid > OLDER_WINDOW_PAUSE_MIN_MS && mid < OLDER_WINDOW_PAUSE_MAX_MS); assert.ok(OLDER_WINDOW_PAUSE_MIN_MS >= 45_000); }); // The post capture's media download: the one gallery-dl run that DOWNLOADS. test("capture argv: downloads (no --no-download, no --dump-json), videos on, into exactly the post's dir", () => { const argv = buildGalleryDlCaptureArgs({ id: "1234567890", dir: "/corpus/ch/posts-media/1234567890" }); assert.ok(!argv.includes("--no-download")); assert.ok(!argv.includes("--dump-json")); assert.ok(argv.includes("extractor.twitter.videos=true")); assert.ok(!argv.includes("extractor.twitter.videos=false")); assert.equal(flagValue(argv, "-D"), "/corpus/ch/posts-media/1234567890"); assert.equal(flagValue(argv, "-f"), "{tweet_id}_{num}.{extension}"); assert.equal(argv[argv.length - 1], "https://x.com/i/status/1234567890"); // A tagged line per file downloaded, and per file already there. const prints = argv.filter((_, i) => argv[i - 1] === "--Print"); assert.deepEqual(prints, [ `after:${CAPTURE_PRINT_TAG}\t{_url}\t{_path}`, `skip:${CAPTURE_PRINT_TAG}\t{_url}\t{_path}`, ]); }); test("capture argv: paced and logged in exactly as the reads are", () => { const read = buildGalleryDlArgs({ accountUrl: ACCOUNT, cookieFile: JAR }); const capture = buildGalleryDlCaptureArgs({ id: "42", dir: "/d", cookieFile: JAR }); for (const flag of [ `extractor.twitter.sleep-request=${X_SLEEP_REQUEST}`, "extractor.twitter.ratelimit=wait", ]) { assert.ok(read.includes(flag) && capture.includes(flag), flag); } assert.equal(flagValue(capture, "--cookies"), JAR); const browser = buildGalleryDlCaptureArgs({ id: "42", dir: "/d", cookies: "firefox" }); assert.equal(flagValue(browser, "--cookies-from-browser"), "firefox"); assert.ok(!browser.includes("--cookies")); const guest = buildGalleryDlCaptureArgs({ id: "42", dir: "/d" }); assert.ok(!guest.includes("--cookies") && !guest.includes("--cookies-from-browser")); }); test("capture argv: an id that is not X's digits is refused, not turned into a URL", () => { for (const id of ["../etc", "12a", "", "1 2"]) { assert.throws(() => buildGalleryDlCaptureArgs({ id, dir: "/d" }), /is not a post id/); } }); test("capture prints: tagged lines name each file once, with its source URL; the rest is ignored", () => { const out = [ "/corpus/ch/posts-media/42/42_1.jpg", `${CAPTURE_PRINT_TAG}\thttps://pbs.twimg.com/media/abc?format=jpg&name=orig\t/corpus/ch/posts-media/42/42_1.jpg`, `${CAPTURE_PRINT_TAG}\thttps://video.twimg.com/v/clip.mp4\t/corpus/ch/posts-media/42/42_2.mp4`, `${CAPTURE_PRINT_TAG}\thttps://pbs.twimg.com/media/abc?format=jpg&name=orig\t/corpus/ch/posts-media/42/42_1.jpg`, `${CAPTURE_PRINT_TAG}\tNone\t/corpus/ch/posts-media/42/42_3.png`, "# /corpus/ch/posts-media/42/42_1.jpg", ].join("\n"); assert.deepEqual(parseCapturePrints(out), [ { name: "42_1.jpg", url: "https://pbs.twimg.com/media/abc?format=jpg&name=orig" }, { name: "42_2.mp4", url: "https://video.twimg.com/v/clip.mp4" }, { name: "42_3.png" }, ]); }); test("the capture's gap before each contact is the reads' 4–10 s, drawn fresh", () => { assert.equal(xRequestPauseMs(() => 0), 4_000); assert.equal(xRequestPauseMs(() => 1), 10_000); assert.equal(xRequestPauseMs(() => 0.5), 7_000); });