import { test } from "node:test"; import assert from "node:assert/strict"; import { mkdtempSync, writeFileSync, chmodSync, mkdirSync } from "node:fs"; import os from "node:os"; import path from "node:path"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test ytdlp/sweepIncomplete.test.ts // // A 429 part-way through a full enumeration is "incomplete", not a listing and // not a failure (release 5 slice R). getPaths() memoizes, so the env is set // before anything imports it. const ROOT = mkdtempSync(path.join(os.tmpdir(), "sweep-incomplete-")); process.env.TRANSCRIPTS_DIR = ROOT; process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); writeFileSync(process.env.SETTINGS_FILE, "{}\n"); mkdirSync(path.join(ROOT, "channels", "c"), { recursive: true }); const { fetchFlatPlaylistUrls, EnumerationIncompleteError, lastListingPage, fullSweepDue, } = await import("./runYtdlp"); const { getPaths } = await import("../lib/paths"); const { recordDownloadBackoff } = await import("../jobs/downloadBackoff"); // A stand-in yt-dlp: prints `urls` on stdout, `stderr` on stderr, exits `code`. function fakeBin(name: string, urls: string[], stderr: string, code: number) { const bin = path.join(ROOT, name); writeFileSync( bin, `#!/bin/sh\ncat <<'OUT'\n${urls.join("\n")}\nOUT\ncat >&2 <<'ERR'\n${stderr}\nERR\nexit ${code}\n`, ); chmodSync(bin, 0o755); return bin; } function run(bin: string) { return fetchFlatPlaylistUrls({ channelConfig: { handling: "youtube", url: "https://rumble.com/c/x" } as never, paths: { ...getPaths(), ytdlpBin: bin }, channelSlug: "c", onLog: () => {}, signal: new AbortController().signal, }); } const PAGES = [1, 2, 3] .map((n) => `[RumbleChannel] x: Downloading page ${n}`) .join("\n"); const URLS = Array.from({ length: 15 }, (_, i) => `https://rumble.com/v${i}-t.html`); test("lastListingPage reads the extractor's last page line", () => { assert.equal(lastListingPage(PAGES), 3); assert.equal(lastListingPage("nothing here"), null); }); test("a 429 mid-listing throws EnumerationIncompleteError with page and count", async () => { const bin = fakeBin( "ytdlp-429", URLS, `${PAGES}\nERROR: x: Unable to download webpage: HTTP Error 429: Too Many Requests`, 1, ); await assert.rejects(run(bin), (err: unknown) => { assert.ok(err instanceof EnumerationIncompleteError); assert.equal(err.platform, "rumble"); assert.equal(err.pagesReached, 3); assert.equal(err.count, 15); assert.match(err.message, /^yt-dlp exited with code 1/); return true; }); }); test("any other non-zero exit keeps throwing the plain error", async () => { const bin = fakeBin("ytdlp-other", URLS, "ERROR: Unsupported URL", 1); await assert.rejects(run(bin), (err: unknown) => { assert.ok(err instanceof Error); assert.ok(!(err instanceof EnumerationIncompleteError)); assert.equal(err.message, "yt-dlp exited with code 1"); return true; }); }); test("fullSweepDue is false while the channel's platform cools down", async () => { const paths = getPaths(); const due = { channelConfig: { handling: "youtube", url: "https://rumble.com/c/x", fullSweepIntervalMinutes: 60, lastFullSweepAt: "2020-01-01T00:00:00.000Z", } as never, paths, }; const youtube = { channelConfig: { handling: "youtube", url: "https://www.youtube.com/@x", fullSweepIntervalMinutes: 60, lastFullSweepAt: "2020-01-01T00:00:00.000Z", } as never, paths, }; assert.equal(await fullSweepDue(due), true); await recordDownloadBackoff("rumble", paths); assert.equal(await fullSweepDue(due), false); // Another platform's sweep is unaffected. assert.equal(await fullSweepDue(youtube), true); });