import { test } from "node:test"; import assert from "node:assert/strict"; import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs"; import os from "node:os"; import path from "node:path"; import type { DownloadOutcomeRecord } from "../lib/downloadOutcome"; import type { ChannelConfig } from "../lib/channelConfig"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test controller/archiveOrgImport.test.ts // // The bulk import with everything injected: a scripted archive.org, a fake // per-file download, a sleeper that only records. getPaths()/getSettings() // memoize, so the env is set before anything imports them. Every name here is // invented. const ROOT = mkdtempSync(path.join(os.tmpdir(), "archiveorg-import-")); process.env.TRANSCRIPTS_DIR = ROOT; process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); writeFileSync( process.env.SETTINGS_FILE, JSON.stringify({ minFreeDiskGB: 0, sleepBetweenDownloadsSeconds: 3 }) + "\n", ); mkdirSync(path.join(ROOT, "channels", "c", "data"), { recursive: true }); const { ARCHIVE_ORG_ITEM_GAP_SECONDS, ARCHIVE_ORG_MIN_GAP_SECONDS, archiveOrgGapMs, archiveOrgRestriction, archiveOrgSearchUrl, resolveArchiveOrgImportUrl, runArchiveOrgBatchImport, runArchiveOrgImport, searchArchiveOrgItems, summarizeArchiveOrgBatch, } = await import("./archiveOrgImport"); const { ArchiveOrgClient } = await import("../lib/archiveOrgClient"); const { getPaths } = await import("../lib/paths"); const { archiveOrgVideoId } = await import("../lib/archiveOrgId"); const ITEM = "example-item"; const FILES = ["Alpha-AbC123xyz_9.mp4", "Beta-Def456uvw_8.mp4", "Gamma-Ghi789rst_7.mp4"]; function client(meta: unknown) { return new ArchiveOrgClient( { sleep: async () => {}, fetch: async () => new Response(JSON.stringify(meta), { status: 200 }), }, { minGapMs: 0 }, ); } const MULTI = { metadata: { identifier: ITEM, title: "Example Archive" }, files: [ ...FILES.map((name) => ({ name, source: "original" })), { name: "Alpha-AbC123xyz_9.info.json", source: "original" }, { name: "Alpha-AbC123xyz_9.ogv", source: "derivative", original: FILES[0] }, ], }; const SINGLE = { metadata: { identifier: "example-film", title: "Example Film" }, files: [{ name: "film.mp4", source: "original" }, { name: "film.ogv", source: "derivative" }], }; const CONFIG: ChannelConfig = { handling: "transcribe", platform: "archiveorg" } as ChannelConfig; function outcome(status: DownloadOutcomeRecord["status"], failureClass?: DownloadOutcomeRecord["failureClass"]): DownloadOutcomeRecord { return { videoId: "x", status, startedAt: "2026-01-01T00:00:00.000Z", finishedAt: "2026-01-01T00:00:01.000Z", attempts: [], ...(failureClass ? { failureClass } : {}), } as DownloadOutcomeRecord; } test("one URL: an item with several media files is refused, naming the way to choose", async () => { const r = await resolveArchiveOrgImportUrl(`https://archive.org/details/${ITEM}`, { client: client(MULTI) }); assert.equal(r.ok, false); assert.match(!r.ok ? r.error : "", /holds 3 media files.*import-archive-org/s); }); test("one URL: a file of many is that file; the only file of an item is the item", async () => { const r = await resolveArchiveOrgImportUrl( `https://archive.org/details/${ITEM}/${encodeURIComponent(FILES[1])}`, { client: client(MULTI) }, ); assert.ok(r.ok); assert.equal(r.ok && r.id, archiveOrgVideoId({ identifier: ITEM, file: FILES[1] })); assert.equal(r.ok && r.url, `https://archive.org/details/${ITEM}/Beta-Def456uvw_8.mp4`); const whole = await resolveArchiveOrgImportUrl("https://archive.org/embed/example-film", { client: client(SINGLE) }); assert.deepEqual(whole, { ok: true, url: "https://archive.org/details/example-film", id: "example-film", identifier: "example-film" }); const byFile = await resolveArchiveOrgImportUrl("https://archive.org/details/example-film/film.mp4", { client: client(SINGLE) }); assert.equal(byFile.ok && byFile.id, "example-film"); const missing = await resolveArchiveOrgImportUrl(`https://archive.org/details/${ITEM}/nope.mp4`, { client: client(MULTI) }); assert.equal(missing.ok, false); }); test("the gap is the configured pause, floored, plus up to half again", () => { assert.equal(archiveOrgGapMs(0, 0), ARCHIVE_ORG_MIN_GAP_SECONDS * 1000); assert.equal(archiveOrgGapMs(3, 0), ARCHIVE_ORG_MIN_GAP_SECONDS * 1000); assert.equal(archiveOrgGapMs(30, 0), 30_000); assert.equal(archiveOrgGapMs(30, 1), 45_000); }); test("bulk: chosen files one at a time, paced, skipping what is on disk", async () => { const paths = getPaths(); // Beta is already downloaded (a transcript on disk). const betaId = archiveOrgVideoId({ identifier: ITEM, file: FILES[1] }); mkdirSync(path.join(ROOT, "channels", "c", "data", betaId), { recursive: true }); writeFileSync(path.join(ROOT, "channels", "c", "data", betaId, "transcript.json"), "{}"); const urls: string[] = []; const sleeps: number[] = []; const imported: string[] = []; const result = await runArchiveOrgImport({ paths, slug: "c", channelConfig: CONFIG, identifier: ITEM, selection: { match: "\\.mp4$" }, onLog: () => {}, signal: new AbortController().signal, deps: { client: client(MULTI), downloadOne: async (o) => { urls.push(o.videoUrl); return outcome("ok"); }, sleep: async (ms) => { sleeps.push(ms); }, random: () => 0, onImported: (id) => imported.push(id), }, }); assert.deepEqual(urls, [ `https://archive.org/details/${ITEM}/Alpha-AbC123xyz_9.mp4`, `https://archive.org/details/${ITEM}/Gamma-Ghi789rst_7.mp4`, ]); // One gap, between the two fetches — none before the first, none for the skip. assert.deepEqual(sleeps, [ARCHIVE_ORG_MIN_GAP_SECONDS * 1000]); assert.deepEqual(result.imported, [FILES[0], FILES[2]]); assert.deepEqual(result.skipped, [FILES[1]]); assert.equal(imported.length, 2); assert.equal(result.stopped, undefined); }); test("bulk: a rate-limited file stops the batch at once", async () => { let n = 0; const result = await runArchiveOrgImport({ paths: getPaths(), slug: "c", channelConfig: CONFIG, identifier: ITEM, selection: { files: [FILES[0], FILES[2]] }, onLog: () => {}, signal: new AbortController().signal, deps: { client: client(MULTI), downloadOne: async () => { n++; return outcome("failed", "rate_limit"); }, sleep: async () => {}, }, }); assert.equal(n, 1); assert.equal(result.rateLimited, true); assert.match(result.stopped ?? "", /rate-limited/); }); test("bulk: three failures in a row stop it; unknown names are reported", async () => { const many = { metadata: { identifier: ITEM }, files: ["a", "b", "c", "d", "e"].map((x) => ({ name: `${x}.mp3`, source: "original" })), }; let n = 0; const result = await runArchiveOrgImport({ paths: getPaths(), slug: "c", channelConfig: CONFIG, identifier: ITEM, selection: { files: ["a.mp3", "b.mp3", "c.mp3", "d.mp3", "zz.mp3"] }, onLog: () => {}, signal: new AbortController().signal, deps: { client: client(many), downloadOne: async () => { n++; throw new Error("boom"); }, sleep: async () => {}, }, }); assert.equal(n, 3); assert.equal(result.failed.length, 3); assert.deepEqual(result.unknown, ["zz.mp3"]); assert.match(result.stopped ?? "", /3 failures in a row/); }); test("bulk: a dry run fetches nothing", async () => { let n = 0; const result = await runArchiveOrgImport({ paths: getPaths(), slug: "c", channelConfig: CONFIG, identifier: ITEM, selection: { match: "." }, onLog: () => {}, signal: new AbortController().signal, dryRun: true, deps: { client: client(MULTI), downloadOne: async () => { n++; return outcome("ok"); }, }, }); assert.equal(n, 0); assert.equal(result.planned, 3); }); // ─── Many items (release 19 A6) ─── // A scripted archive.org that answers by URL: each item's metadata, a search, // and `{}` (no such item) for anything else. Every request is recorded. function routedClient(items: Record, search?: unknown) { const asked: string[] = []; const c = new ArchiveOrgClient( { sleep: async () => {}, fetch: async (url: string) => { asked.push(url); if (url.includes("advancedsearch.php")) return new Response(JSON.stringify(search ?? {}), { status: 200 }); const id = /\/metadata\/([^/?]+)/.exec(url)?.[1] ?? ""; return new Response(JSON.stringify(items[decodeURIComponent(id)] ?? {}), { status: 200 }); }, }, { minGapMs: 0 }, ); return { client: c, asked }; } const one = (identifier: string, extra: Record = {}, file: Record = {}) => ({ metadata: { identifier, title: identifier, ...extra }, files: [{ name: `${identifier}.mp4`, source: "original", ...file }], }); test("restriction: an access-restricted item restricts every file; else a private file only", () => { const whole = archiveOrgRestriction(one("r1", { "access-restricted-item": "true" }) as never); assert.equal(whole.item, true); assert.deepEqual([...whole.files], ["r1.mp4"]); const priv = archiveOrgRestriction({ metadata: { identifier: "r2" }, files: [ { name: "a.mp4", source: "original", private: "true" }, { name: "b.mp4", source: "original" }, ], } as never); assert.equal(priv.item, false); assert.deepEqual([...priv.files], ["a.mp4"]); }); test("search: one request through the client, identifiers in order; the URL sorts and caps", async () => { const url = archiveOrgSearchUrl("collection:example AND mediatype:movies", 50); assert.match(url, /^https:\/\/archive\.org\/advancedsearch\.php\?/); const q = new URL(url).searchParams; assert.equal(q.get("q"), "collection:example AND mediatype:movies"); assert.deepEqual(q.getAll("fl[]"), ["identifier", "title"]); assert.equal(q.get("sort[]"), "identifier asc"); assert.equal(q.get("rows"), "50"); assert.equal(q.get("output"), "json"); const { client: c, asked } = routedClient( {}, { response: { numFound: 7, docs: [{ identifier: "a-1", title: ["A"] }, { identifier: "b-2" }, { title: "no id" }] } }, ); const r = await searchArchiveOrgItems("x", { rows: 9999, client: c }); assert.equal(asked.length, 1); assert.equal(new URL(asked[0]).searchParams.get("rows"), "500"); assert.equal(r.found, 7); assert.deepEqual(r.items, [{ identifier: "a-1", title: "A" }, { identifier: "b-2" }]); }); test("batch dry run: held (on disk or saved), restricted and missing items are told apart; nothing fetched", async () => { const dataDir = path.join(ROOT, "channels", "c", "data"); // h-disk holds a transcript; h-saved has only its saved-video pointer (the // saved-container tier) — both are held. mkdirSync(path.join(dataDir, "h-disk"), { recursive: true }); writeFileSync(path.join(dataDir, "h-disk", "transcript.json"), "{}"); mkdirSync(path.join(dataDir, "h-saved"), { recursive: true }); writeFileSync( path.join(dataDir, "h-saved", "saved-video.json"), JSON.stringify({ storedAt: "2026-01-01T00:00:00Z", dir: "/store/c/h-saved", file: "source.mp4", bytes: 10 }), ); const { client: c } = routedClient({ "h-disk": one("h-disk"), "h-saved": one("h-saved"), "r-item": one("r-item", { "access-restricted-item": true }), "r-file": one("r-file", {}, { private: "true" }), fresh: one("fresh"), }); const sleeps: number[] = []; const lines: string[] = []; let n = 0; const b = await runArchiveOrgBatchImport({ paths: getPaths(), slug: "c", channelConfig: CONFIG, items: ["h-disk", "h-saved", "r-item", "r-file", "gone", "fresh"].map((identifier) => ({ identifier, selection: { match: "." }, })), onLog: (l) => lines.push(l), signal: new AbortController().signal, dryRun: true, deps: { client: c, downloadOne: async () => { n++; return outcome("ok"); }, sleep: async (ms) => { sleeps.push(ms); }, random: () => 0, }, }); assert.equal(n, 0); // The inter-item gap before every item after the first, and no download gap. assert.deepEqual(sleeps, Array(5).fill(ARCHIVE_ORG_ITEM_GAP_SECONDS * 1000)); const s = summarizeArchiveOrgBatch(b); assert.equal(s.dryRun, true); assert.equal(s.items, 5); assert.equal(s.held, 2); assert.deepEqual(s.restricted, [ { identifier: "r-item", files: ["r-item.mp4"] }, { identifier: "r-file", files: ["r-file.mp4"] }, ]); assert.deepEqual(s.missing, ["gone"]); assert.match(lines.join(""), /RESTRICTED r-item/); assert.match(lines.join(""), /would get fresh/); }); test("batch import: a 401 skips the rest of its item; three refused items in a row stop the batch", async () => { const two = (id: string) => ({ metadata: { identifier: id }, files: [`${id}-a.mp4`, `${id}-b.mp4`].map((name) => ({ name, source: "original" })), }); const { client: c } = routedClient({ p1: two("p1"), p2: two("p2"), p3: two("p3"), p4: two("p4") }); const urls: string[] = []; const b = await runArchiveOrgBatchImport({ paths: getPaths(), slug: "c", channelConfig: CONFIG, items: ["p1", "p2", "p3", "p4"].map((identifier) => ({ identifier, selection: { match: "." } })), onLog: () => {}, signal: new AbortController().signal, deps: { client: c, downloadOne: async (o) => { urls.push(o.videoUrl); return { ...outcome("failed"), attempts: [{ n: 1, error: `archive.org answered HTTP 401 for ${o.videoUrl}` }], } as unknown as DownloadOutcomeRecord; }, sleep: async () => {}, }, }); // One attempt per item (its second file skipped as restricted), three items, then the stop. assert.equal(urls.length, 3); assert.equal(b.refusedStorm, true); assert.match(b.stopped ?? "", /refused 3 items in a row/); assert.deepEqual(b.notReached, ["p4"]); assert.deepEqual(b.items[0].restricted, ["p1-a.mp4", "p1-b.mp4"]); assert.equal(b.items[0].failed.length, 0); }); test("batch import: a rate limit stops everything; the rest are not reached", async () => { const { client: c } = routedClient({ q1: one("q1"), q2: one("q2"), q3: one("q3") }); let n = 0; const b = await runArchiveOrgBatchImport({ paths: getPaths(), slug: "c", channelConfig: CONFIG, items: ["q1", "q2", "q3"].map((identifier) => ({ identifier, selection: { match: "." } })), onLog: () => {}, signal: new AbortController().signal, deps: { client: c, downloadOne: async () => { n++; return n === 1 ? outcome("ok") : outcome("failed", "rate_limit"); }, sleep: async () => {}, }, }); assert.equal(n, 2); assert.equal(b.rateLimited, true); assert.deepEqual(b.notReached, ["q3"]); assert.deepEqual(summarizeArchiveOrgBatch(b).imported, 1); });