import { test, after } from "node:test"; import assert from "node:assert/strict"; import { createHash } from "node:crypto"; import { chmod, mkdir, mkdtemp, readFile, readdir, rm, stat, writeFile } from "node:fs/promises"; import os from "node:os"; import path from "node:path"; import type { Paths } from "../lib/paths"; import type { ChannelConfig } from "../lib/channelConfig"; import type { ArchiveOrgDownloadDeps } from "./archiveOrgDownload"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test controller/archiveOrgDownload.test.ts // // The archive.org download ladder with everything around it faked: a scripted // archive.org (metadata, torrent, a mirror's info.json), a torrent fetch and a // direct fetch that write chosen bytes, a duration probe. The checksum check is // the real one, against md5/sha1 computed here. Every download goes through // downloadOneManaged, with a yt-dlp that records any call — and none is made. // Every name is invented. const ROOT = await mkdtemp(path.join(os.tmpdir(), "archiveorg-dl-")); process.env.TRANSCRIPTS_DIR = ROOT; process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); await writeFile(process.env.SETTINGS_FILE, JSON.stringify({ minFreeDiskGB: 0 }) + "\n"); after(() => rm(ROOT, { recursive: true, force: true })); const { downloadOneManaged } = await import("../ytdlp/downloadOneManaged"); const { ArchiveOrgClient } = await import("../lib/archiveOrgClient"); const { ArchiveOrgRequestError } = await import("../lib/archiveOrgClient"); const { archiveOrgDetailsUrl, archiveOrgVideoId } = await import("../lib/archiveOrgId"); const { platformFromMetadata } = await import("../lib/transcripts-server"); const { DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS } = await import("../lib/archiveOrgTorrent"); const YTDLP_CALLED = path.join(ROOT, "ytdlp-called"); const YTDLP = path.join(ROOT, "fake-ytdlp.sh"); await writeFile(YTDLP, `#!/bin/sh\necho "$@" >> "${YTDLP_CALLED}"\nexit 1\n`); await chmod(YTDLP, 0o755); const GOOD = Buffer.from("the right bytes of the recording\n".repeat(40)); // Same length, one byte different: only the checksum can tell. const BAD = Buffer.from(GOOD); BAD[7] ^= 0xff; const md5 = (b: Buffer) => createHash("md5").update(b).digest("hex"); const sha1 = (b: Buffer) => createHash("sha1").update(b).digest("hex"); // ─── A synthetic torrent (the same ten-line encoder as the torrent test) ─── type Enc = number | string | Buffer | Enc[] | { [k: string]: Enc }; function benc(v: Enc): Buffer { if (typeof v === "number") return Buffer.from(`i${v}e`); if (typeof v === "string") return benc(Buffer.from(v, "utf8")); if (Buffer.isBuffer(v)) return Buffer.concat([Buffer.from(`${v.length}:`), v]); if (Array.isArray(v)) return Buffer.concat([Buffer.from("l"), ...v.map(benc), Buffer.from("e")]); const keys = Object.keys(v).sort(); return Buffer.concat([Buffer.from("d"), ...keys.flatMap((k) => [benc(k), benc(v[k])]), Buffer.from("e")]); } function torrentOf(identifier: string, files: string[]): Buffer { return benc({ "url-list": ["https://archive.org/download/"], info: { name: identifier, "piece length": 16384, pieces: Buffer.alloc(20), files: files.map((f) => ({ path: f.split("/"), length: GOOD.length })), }, }); } // ─── The scripted archive.org ─── type Route = { json?: unknown; bytes?: Buffer; status?: number }; function client(routes: Record, seen: string[] = []) { return new ArchiveOrgClient( { sleep: async () => {}, fetch: async (url) => { seen.push(url); const r = routes[url]; if (!r) return new Response("not found", { status: 404 }); if (r.status) return new Response("no", { status: r.status }); if (r.bytes) return new Response(new Uint8Array(r.bytes), { status: 200 }); return new Response(JSON.stringify(r.json), { status: 200 }); }, }, { minGapMs: 0, maxAttempts: 2 }, ); } const AUDIO_ITEM = "example-radio-show"; const AUDIO_META = { metadata: { identifier: AUDIO_ITEM, title: "Example Radio Show", creator: "Example Station", uploader: "someone@example.org", date: "1999-05-04", description: "An invented show.", subject: "radio; example", }, files: [ { name: "show.mp3", source: "original", format: "VBR MP3", size: String(GOOD.length), md5: md5(GOOD), sha1: sha1(GOOD), length: "125.5" }, { name: `${AUDIO_ITEM}_archive.torrent`, source: "metadata" }, ], }; const VIDEO_ITEM = "example-channel-archive"; const VIDEO_FILE = "Clip One-AbC123xyz_9.mp4"; const VIDEO_META = { metadata: { identifier: VIDEO_ITEM, title: "Example Channel Archive", creator: "Example Creator" }, files: [ { name: VIDEO_FILE, source: "original", format: "h.264", size: String(GOOD.length), md5: md5(GOOD) }, { name: "Clip One-AbC123xyz_9.info.json", source: "original" }, { name: "Clip Two-Def456uvw_8.avi", source: "original", format: "AVI", size: "99" }, { name: "Clip Two-Def456uvw_8.mp4", source: "derivative", original: "Clip Two-Def456uvw_8.avi", format: "h.264", size: String(GOOD.length), md5: md5(GOOD) }, { name: `${VIDEO_ITEM}_archive.torrent`, source: "metadata" }, ], }; const MIRROR_INFO = { id: "AbC123xyz_9", extractor_key: "Youtube", title: "Clip One (the original)", upload_date: "20210102", uploader: "Example Uploader", }; const metaUrl = (id: string) => `https://archive.org/metadata/${id}`; const torrentUrl = (id: string) => `https://archive.org/download/${id}/${id}_archive.torrent`; let n = 0; async function download(opts: { url: string; deps: ArchiveOrgDownloadDeps; config?: Partial; }) { const slug = `c${++n}`; const paths = { channelsDir: path.join(ROOT, "channels"), ytdlpBin: YTDLP, ffprobeBin: "ffprobe", ffmpegBin: "ffmpeg", aria2cBin: "aria2c", } as Paths; await mkdir(path.join(paths.channelsDir, slug, "data"), { recursive: true }); const lines: string[] = []; const rec = await downloadOneManaged({ channelSlug: slug, channelConfig: { handling: "transcribe", platform: "archiveorg", audioFormat: "mp3", ...opts.config } as ChannelConfig, paths, videoUrl: opts.url, onLog: (l) => lines.push(l), signal: new AbortController().signal, appendArchive: true, archiveOrgDeps: { settings: DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, probeDuration: async () => 125.25, ...opts.deps, }, }); const channelDir = path.join(paths.channelsDir, slug); return { rec, lines, log: lines.join(""), channelDir }; } const writeTo = (bytes: Buffer) => async (o: { dest: string }) => { await writeFile(o.dest, bytes); }; test("an audio item, no aria2c: a direct download IS the audio — no yt-dlp, a full record", async () => { const seen: string[] = []; const directs: string[] = []; const { rec, log, channelDir } = await download({ url: archiveOrgDetailsUrl({ identifier: AUDIO_ITEM }), deps: { client: client({ [metaUrl(AUDIO_ITEM)]: { json: AUDIO_META } }, seen), aria2cAvailable: async () => false, fetchDirect: async (o) => { directs.push(o.file); await writeFile(o.dest, GOOD); }, }, }); assert.equal(rec.status, "ok"); assert.deepEqual(rec.attempts.map((a) => [a.kind, a.ytdlpExitCode]), [["archiveorg-direct", 0]]); assert.match(log, /fell back to direct download: torrents need aria2c/); assert.deepEqual(directs, ["show.mp3"]); await assert.rejects(stat(YTDLP_CALLED), "yt-dlp was never run"); const dir = path.join(channelDir, "data", AUDIO_ITEM); assert.deepEqual(await readFile(path.join(dir, "audio.mp3")), GOOD); const info = JSON.parse(await readFile(path.join(dir, "metadata.info.json"), "utf8")); assert.equal(info.extractor_key, "ArchiveOrg"); assert.equal(platformFromMetadata(info), "archiveorg"); assert.equal(info.id, AUDIO_ITEM); assert.equal(info.webpage_url, `https://archive.org/details/${AUDIO_ITEM}`); assert.equal(info.title, "Example Radio Show"); assert.equal(info.uploader, "Example Station", "the public credit, never the account's e-mail"); assert.equal(info.duration, 125.25); assert.equal(info.upload_date, "19990504"); assert.deepEqual(info.tags, ["radio", "example"]); assert.equal(Object.keys(info).at(-1), "upload_date"); assert.deepEqual(info.formats.map((f: { url: string }) => f.url), [`https://archive.org/download/${AUDIO_ITEM}/show.mp3`]); const prov = JSON.parse(await readFile(path.join(dir, "archiveorg.json"), "utf8")); assert.equal(prov.identifier, AUDIO_ITEM); const outcome = JSON.parse(await readFile(path.join(dir, "download-outcome.json"), "utf8")); assert.equal(outcome.status, "ok"); assert.equal(await readFile(path.join(channelDir, "archive"), "utf8"), `archiveorg ${AUDIO_ITEM}\n`); assert.ok(!(await readdir(dir)).includes(".archiveorg-fetch"), "the staging dir is gone"); assert.match(await readFile(path.join(dir, "download.log"), "utf8"), /Verified show\.mp3 \(sha1\)/); }); test("a stalled torrent falls back to a direct download, saying why", async () => { const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: VIDEO_FILE }); let torrentCalls = 0; const finalized: string[] = []; const { rec, log } = await download({ url, deps: { client: client({ [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META }, [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, [VIDEO_FILE]) }, }), aria2cAvailable: async () => true, fetchByTorrent: async (o) => { torrentCalls++; assert.equal(o.entry.path, VIDEO_FILE); assert.equal(o.entry.index, 1); return { ok: false, stalled: true, reason: "stalled: no progress in 5 min" }; }, fetchDirect: writeTo(GOOD), finalize: async (o) => { finalized.push(`${o.sourceFilename} persist=${o.persist}`); await writeFile(path.join(o.videoDir, "audio.mp3"), "a"); await rm(path.join(o.videoDir, o.sourceFilename!)); }, }, }); assert.equal(torrentCalls, 1); assert.equal(rec.status, "ok"); assert.match(log, /fell back to direct download: stalled: no progress in 5 min/); assert.deepEqual(rec.attempts.map((a) => [a.kind, a.error ?? null]), [ ["archiveorg-torrent", "stalled: no progress in 5 min"], ["archiveorg-direct", null], ]); assert.deepEqual(finalized, ["source-media.mp4 persist=false"]); assert.equal(rec.videoId, archiveOrgVideoId({ identifier: VIDEO_ITEM, file: VIDEO_FILE })); }); test("a torrent file that fails its checksum falls back once to a direct download", async () => { const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: VIDEO_FILE }); const { rec, log } = await download({ url, deps: { client: client({ [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META }, [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, [VIDEO_FILE]) }, }), aria2cAvailable: async () => true, fetchByTorrent: async (o) => { const file = path.join(o.stagingDir, VIDEO_ITEM, VIDEO_FILE); await mkdir(path.dirname(file), { recursive: true }); await writeFile(file, BAD); return { ok: true, file, seeded: true }; }, fetchDirect: writeTo(GOOD), finalize: async (o) => { await writeFile(path.join(o.videoDir, "audio.mp3"), "a"); }, }, }); assert.equal(rec.status, "ok"); assert.match(log, /fell back to direct download: checksum mismatch \(md5 /); assert.deepEqual(rec.attempts.map((a) => a.kind), ["archiveorg-torrent", "archiveorg-direct"]); assert.match(rec.attempts[0].error ?? "", /^checksum mismatch: md5/); }); test("a second checksum mismatch fails the record; nothing is kept as audio", async () => { const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: VIDEO_FILE }); let directs = 0; const { rec, channelDir } = await download({ url, deps: { client: client({ [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META }, [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, [VIDEO_FILE]) }, }), aria2cAvailable: async () => true, fetchByTorrent: async (o) => { const file = path.join(o.stagingDir, "x.mp4"); await mkdir(o.stagingDir, { recursive: true }); await writeFile(file, BAD); return { ok: true, file, seeded: false }; }, fetchDirect: async (o) => { directs++; await writeFile(o.dest, BAD); }, }, }); assert.equal(directs, 1, "one direct download after the torrent's mismatch, no more"); assert.equal(rec.status, "failed"); assert.equal(rec.failureClass, "per_video"); assert.match(rec.attempts.at(-1)?.error ?? "", /checksum mismatch/); const dir = path.join(channelDir, "data", rec.videoId); const entries = await readdir(dir); assert.ok(!entries.some((e) => e.startsWith("audio."))); await assert.rejects(stat(path.join(channelDir, "archive"))); }); test("direct only (no torrent): a mismatch is retried once, then fails", async () => { let directs = 0; const { rec } = await download({ url: archiveOrgDetailsUrl({ identifier: AUDIO_ITEM }), deps: { client: client({ [metaUrl(AUDIO_ITEM)]: { json: AUDIO_META } }), settings: { ...DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, torrent: false }, fetchDirect: async (o) => { directs++; await writeFile(o.dest, directs === 1 ? BAD : GOOD); }, }, }); assert.equal(directs, 2); assert.equal(rec.status, "ok"); assert.deepEqual(rec.attempts.map((a) => [a.kind, !!a.error]), [ ["archiveorg-direct", true], ["archiveorg-direct", false], ]); }); test("a verified torrent fetch needs no direct download; the record is the mirror's original", async () => { const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: VIDEO_FILE }); const seen: string[] = []; const { rec, channelDir } = await download({ url, deps: { client: client( { [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META }, [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, ["other.mp4", VIDEO_FILE]) }, [`https://archive.org/download/${VIDEO_ITEM}/Clip%20One-AbC123xyz_9.info.json`]: { json: MIRROR_INFO }, }, seen, ), aria2cAvailable: async () => true, fetchByTorrent: async (o) => { assert.equal(o.entry.index, 2); const file = path.join(o.stagingDir, VIDEO_ITEM, VIDEO_FILE); await mkdir(path.dirname(file), { recursive: true }); await writeFile(file, GOOD); return { ok: true, file, seeded: true }; }, fetchDirect: async () => assert.fail("no direct download after a verified torrent"), finalize: async (o) => { await writeFile(path.join(o.videoDir, "audio.mp3"), "a"); }, }, }); assert.equal(rec.status, "ok"); assert.deepEqual(rec.attempts.map((a) => a.kind), ["archiveorg-torrent"]); const info = JSON.parse(await readFile(path.join(channelDir, "data", rec.videoId, "metadata.info.json"), "utf8")); assert.equal(info.title, "Clip One (the original)"); assert.equal(info.upload_date, "20210102"); assert.equal(info.uploader, "Example Uploader"); assert.equal(info.webpage_url, url); assert.equal(info.id, rec.videoId); const prov = JSON.parse(await readFile(path.join(channelDir, "data", rec.videoId, "archiveorg.json"), "utf8")); assert.equal(prov.mirror.id, "AbC123xyz_9"); assert.equal(prov.mirror.from, "info-json"); // One metadata request, one torrent, one info.json. assert.equal(seen.length, 3); }); test("an .avi original is fetched as archive.org's mp4 of it; a file not in the torrent goes direct", async () => { const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: "Clip Two-Def456uvw_8.avi" }); const directs: string[] = []; const { rec, log } = await download({ url, deps: { client: client({ [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META }, [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, [VIDEO_FILE]) }, }), aria2cAvailable: async () => true, fetchByTorrent: async () => assert.fail("not in the torrent"), fetchDirect: async (o) => { directs.push(o.file); await writeFile(o.dest, GOOD); }, finalize: async (o) => { assert.equal(o.sourceFilename, "source-media.mp4"); await writeFile(path.join(o.videoDir, "audio.mp3"), "a"); }, }, }); assert.equal(rec.status, "ok"); assert.deepEqual(directs, ["Clip Two-Def456uvw_8.mp4"]); assert.match(log, /does not carry Clip Two-Def456uvw_8\.mp4/); }); test("archive.org refusing the direct download is a rate limit; the batch backs off", async () => { const { rec } = await download({ url: archiveOrgDetailsUrl({ identifier: AUDIO_ITEM }), deps: { client: client({ [metaUrl(AUDIO_ITEM)]: { json: AUDIO_META } }), aria2cAvailable: async () => false, fetchDirect: async () => { throw new ArchiveOrgRequestError("archive.org did not deliver x after 4 attempts (HTTP 429); stopping", null, true); }, }, }); assert.equal(rec.status, "failed"); assert.equal(rec.failureClass, "rate_limit"); }); test("an item URL of an item with several media files is refused, nothing fetched", async () => { const { rec, log } = await download({ url: archiveOrgDetailsUrl({ identifier: VIDEO_ITEM }), deps: { client: client({ [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META } }), fetchDirect: async () => assert.fail("nothing to fetch"), }, }); assert.equal(rec.status, "failed"); assert.equal(rec.failureClass, "per_video"); assert.match(log, /holds 2 media files/); });