commit 5e314ea46a132ac9413641ae755d8c4f169251fd
parent a1a5dea0f932cc5b0fa49215979bf5567f5bc409
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 16:00:59 -0400
sources: tests for the archive.org torrent ladder
A synthetic torrent (bencode, file index, piece span, path.utf-8,
single-file); the aria2c argument builder and 1.37's progress lines; one
aria2c run against a fake aria2c (complete + seed, stall, failure, cancel
killing the whole process group); the ladder through downloadOneManaged with
a yt-dlp that is never called (no aria2c → direct, stall → direct, torrent
checksum mismatch → direct once, a second mismatch fails, direct-only retry,
a verified torrent, an .avi fetched as its mp4, a rate-limited direct
download); the client's resumable direct download and cached torrent; the
synthesised info.json (ArchiveOrg maps to archiveorg, a mirror's original
kept, the date last). archiveOrgDeps is the managed download's test seam.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
7 files changed, 985 insertions(+), 3 deletions(-)
diff --git a/common/controller/archiveOrgDownload.test.ts b/common/controller/archiveOrgDownload.test.ts
@@ -0,0 +1,422 @@
+import { test, after } from "node:test";
+import assert from "node:assert/strict";
+import { createHash } from "node:crypto";
+import { chmod, mkdir, mkdtemp, readFile, readdir, rm, stat, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import type { ChannelConfig } from "../lib/channelConfig";
+import type { ArchiveOrgDownloadDeps } from "./archiveOrgDownload";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/archiveOrgDownload.test.ts
+//
+// The archive.org download ladder with everything around it faked: a scripted
+// archive.org (metadata, torrent, a mirror's info.json), a torrent fetch and a
+// direct fetch that write chosen bytes, a duration probe. The checksum check is
+// the real one, against md5/sha1 computed here. Every download goes through
+// downloadOneManaged, with a yt-dlp that records any call — and none is made.
+// Every name is invented.
+
+const ROOT = await mkdtemp(path.join(os.tmpdir(), "archiveorg-dl-"));
+process.env.TRANSCRIPTS_DIR = ROOT;
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+await writeFile(process.env.SETTINGS_FILE, JSON.stringify({ minFreeDiskGB: 0 }) + "\n");
+after(() => rm(ROOT, { recursive: true, force: true }));
+
+const { downloadOneManaged } = await import("../ytdlp/downloadOneManaged");
+const { ArchiveOrgClient } = await import("../lib/archiveOrgClient");
+const { ArchiveOrgRequestError } = await import("../lib/archiveOrgClient");
+const { archiveOrgDetailsUrl, archiveOrgVideoId } = await import("../lib/archiveOrgId");
+const { platformFromMetadata } = await import("../lib/transcripts-server");
+const { DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS } = await import("../lib/archiveOrgTorrent");
+
+const YTDLP_CALLED = path.join(ROOT, "ytdlp-called");
+const YTDLP = path.join(ROOT, "fake-ytdlp.sh");
+await writeFile(YTDLP, `#!/bin/sh\necho "$@" >> "${YTDLP_CALLED}"\nexit 1\n`);
+await chmod(YTDLP, 0o755);
+
+const GOOD = Buffer.from("the right bytes of the recording\n".repeat(40));
+// Same length, one byte different: only the checksum can tell.
+const BAD = Buffer.from(GOOD);
+BAD[7] ^= 0xff;
+const md5 = (b: Buffer) => createHash("md5").update(b).digest("hex");
+const sha1 = (b: Buffer) => createHash("sha1").update(b).digest("hex");
+
+// ─── A synthetic torrent (the same ten-line encoder as the torrent test) ───
+type Enc = number | string | Buffer | Enc[] | { [k: string]: Enc };
+function benc(v: Enc): Buffer {
+ if (typeof v === "number") return Buffer.from(`i${v}e`);
+ if (typeof v === "string") return benc(Buffer.from(v, "utf8"));
+ if (Buffer.isBuffer(v)) return Buffer.concat([Buffer.from(`${v.length}:`), v]);
+ if (Array.isArray(v)) return Buffer.concat([Buffer.from("l"), ...v.map(benc), Buffer.from("e")]);
+ const keys = Object.keys(v).sort();
+ return Buffer.concat([Buffer.from("d"), ...keys.flatMap((k) => [benc(k), benc(v[k])]), Buffer.from("e")]);
+}
+function torrentOf(identifier: string, files: string[]): Buffer {
+ return benc({
+ "url-list": ["https://archive.org/download/"],
+ info: {
+ name: identifier,
+ "piece length": 16384,
+ pieces: Buffer.alloc(20),
+ files: files.map((f) => ({ path: f.split("/"), length: GOOD.length })),
+ },
+ });
+}
+
+// ─── The scripted archive.org ───
+type Route = { json?: unknown; bytes?: Buffer; status?: number };
+function client(routes: Record<string, Route>, seen: string[] = []) {
+ return new ArchiveOrgClient(
+ {
+ sleep: async () => {},
+ fetch: async (url) => {
+ seen.push(url);
+ const r = routes[url];
+ if (!r) return new Response("not found", { status: 404 });
+ if (r.status) return new Response("no", { status: r.status });
+ if (r.bytes) return new Response(new Uint8Array(r.bytes), { status: 200 });
+ return new Response(JSON.stringify(r.json), { status: 200 });
+ },
+ },
+ { minGapMs: 0, maxAttempts: 2 },
+ );
+}
+
+const AUDIO_ITEM = "example-radio-show";
+const AUDIO_META = {
+ metadata: {
+ identifier: AUDIO_ITEM,
+ title: "Example Radio Show",
+ creator: "Example Station",
+ uploader: "someone@example.org",
+ date: "1999-05-04",
+ description: "An invented show.",
+ subject: "radio; example",
+ },
+ files: [
+ { name: "show.mp3", source: "original", format: "VBR MP3", size: String(GOOD.length), md5: md5(GOOD), sha1: sha1(GOOD), length: "125.5" },
+ { name: `${AUDIO_ITEM}_archive.torrent`, source: "metadata" },
+ ],
+};
+
+const VIDEO_ITEM = "example-channel-archive";
+const VIDEO_FILE = "Clip One-AbC123xyz_9.mp4";
+const VIDEO_META = {
+ metadata: { identifier: VIDEO_ITEM, title: "Example Channel Archive", creator: "Example Creator" },
+ files: [
+ { name: VIDEO_FILE, source: "original", format: "h.264", size: String(GOOD.length), md5: md5(GOOD) },
+ { name: "Clip One-AbC123xyz_9.info.json", source: "original" },
+ { name: "Clip Two-Def456uvw_8.avi", source: "original", format: "AVI", size: "99" },
+ { name: "Clip Two-Def456uvw_8.mp4", source: "derivative", original: "Clip Two-Def456uvw_8.avi", format: "h.264", size: String(GOOD.length), md5: md5(GOOD) },
+ { name: `${VIDEO_ITEM}_archive.torrent`, source: "metadata" },
+ ],
+};
+const MIRROR_INFO = {
+ id: "AbC123xyz_9",
+ extractor_key: "Youtube",
+ title: "Clip One (the original)",
+ upload_date: "20210102",
+ uploader: "Example Uploader",
+};
+
+const metaUrl = (id: string) => `https://archive.org/metadata/${id}`;
+const torrentUrl = (id: string) => `https://archive.org/download/${id}/${id}_archive.torrent`;
+
+let n = 0;
+async function download(opts: {
+ url: string;
+ deps: ArchiveOrgDownloadDeps;
+ config?: Partial<ChannelConfig>;
+}) {
+ const slug = `c${++n}`;
+ const paths = {
+ channelsDir: path.join(ROOT, "channels"),
+ ytdlpBin: YTDLP,
+ ffprobeBin: "ffprobe",
+ ffmpegBin: "ffmpeg",
+ aria2cBin: "aria2c",
+ } as Paths;
+ await mkdir(path.join(paths.channelsDir, slug, "data"), { recursive: true });
+ const lines: string[] = [];
+ const rec = await downloadOneManaged({
+ channelSlug: slug,
+ channelConfig: { handling: "transcribe", platform: "archiveorg", audioFormat: "mp3", ...opts.config } as ChannelConfig,
+ paths,
+ videoUrl: opts.url,
+ onLog: (l) => lines.push(l),
+ signal: new AbortController().signal,
+ appendArchive: true,
+ archiveOrgDeps: {
+ settings: DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS,
+ probeDuration: async () => 125.25,
+ ...opts.deps,
+ },
+ });
+ const channelDir = path.join(paths.channelsDir, slug);
+ return { rec, lines, log: lines.join(""), channelDir };
+}
+
+const writeTo = (bytes: Buffer) => async (o: { dest: string }) => {
+ await writeFile(o.dest, bytes);
+};
+
+test("an audio item, no aria2c: a direct download IS the audio — no yt-dlp, a full record", async () => {
+ const seen: string[] = [];
+ const directs: string[] = [];
+ const { rec, log, channelDir } = await download({
+ url: archiveOrgDetailsUrl({ identifier: AUDIO_ITEM }),
+ deps: {
+ client: client({ [metaUrl(AUDIO_ITEM)]: { json: AUDIO_META } }, seen),
+ aria2cAvailable: async () => false,
+ fetchDirect: async (o) => {
+ directs.push(o.file);
+ await writeFile(o.dest, GOOD);
+ },
+ },
+ });
+ assert.equal(rec.status, "ok");
+ assert.deepEqual(rec.attempts.map((a) => [a.kind, a.ytdlpExitCode]), [["archiveorg-direct", 0]]);
+ assert.match(log, /fell back to direct download: torrents need aria2c/);
+ assert.deepEqual(directs, ["show.mp3"]);
+ await assert.rejects(stat(YTDLP_CALLED), "yt-dlp was never run");
+ const dir = path.join(channelDir, "data", AUDIO_ITEM);
+ assert.deepEqual(await readFile(path.join(dir, "audio.mp3")), GOOD);
+ const info = JSON.parse(await readFile(path.join(dir, "metadata.info.json"), "utf8"));
+ assert.equal(info.extractor_key, "ArchiveOrg");
+ assert.equal(platformFromMetadata(info), "archiveorg");
+ assert.equal(info.id, AUDIO_ITEM);
+ assert.equal(info.webpage_url, `https://archive.org/details/${AUDIO_ITEM}`);
+ assert.equal(info.title, "Example Radio Show");
+ assert.equal(info.uploader, "Example Station", "the public credit, never the account's e-mail");
+ assert.equal(info.duration, 125.25);
+ assert.equal(info.upload_date, "19990504");
+ assert.deepEqual(info.tags, ["radio", "example"]);
+ assert.equal(Object.keys(info).at(-1), "upload_date");
+ assert.deepEqual(info.formats.map((f: { url: string }) => f.url), [`https://archive.org/download/${AUDIO_ITEM}/show.mp3`]);
+ const prov = JSON.parse(await readFile(path.join(dir, "archiveorg.json"), "utf8"));
+ assert.equal(prov.identifier, AUDIO_ITEM);
+ const outcome = JSON.parse(await readFile(path.join(dir, "download-outcome.json"), "utf8"));
+ assert.equal(outcome.status, "ok");
+ assert.equal(await readFile(path.join(channelDir, "archive"), "utf8"), `archiveorg ${AUDIO_ITEM}\n`);
+ assert.ok(!(await readdir(dir)).includes(".archiveorg-fetch"), "the staging dir is gone");
+ assert.match(await readFile(path.join(dir, "download.log"), "utf8"), /Verified show\.mp3 \(sha1\)/);
+});
+
+test("a stalled torrent falls back to a direct download, saying why", async () => {
+ const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: VIDEO_FILE });
+ let torrentCalls = 0;
+ const finalized: string[] = [];
+ const { rec, log } = await download({
+ url,
+ deps: {
+ client: client({
+ [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META },
+ [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, [VIDEO_FILE]) },
+ }),
+ aria2cAvailable: async () => true,
+ fetchByTorrent: async (o) => {
+ torrentCalls++;
+ assert.equal(o.entry.path, VIDEO_FILE);
+ assert.equal(o.entry.index, 1);
+ return { ok: false, stalled: true, reason: "stalled: no progress in 5 min" };
+ },
+ fetchDirect: writeTo(GOOD),
+ finalize: async (o) => {
+ finalized.push(`${o.sourceFilename} persist=${o.persist}`);
+ await writeFile(path.join(o.videoDir, "audio.mp3"), "a");
+ await rm(path.join(o.videoDir, o.sourceFilename!));
+ },
+ },
+ });
+ assert.equal(torrentCalls, 1);
+ assert.equal(rec.status, "ok");
+ assert.match(log, /fell back to direct download: stalled: no progress in 5 min/);
+ assert.deepEqual(rec.attempts.map((a) => [a.kind, a.error ?? null]), [
+ ["archiveorg-torrent", "stalled: no progress in 5 min"],
+ ["archiveorg-direct", null],
+ ]);
+ assert.deepEqual(finalized, ["source-media.mp4 persist=false"]);
+ assert.equal(rec.videoId, archiveOrgVideoId({ identifier: VIDEO_ITEM, file: VIDEO_FILE }));
+});
+
+test("a torrent file that fails its checksum falls back once to a direct download", async () => {
+ const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: VIDEO_FILE });
+ const { rec, log } = await download({
+ url,
+ deps: {
+ client: client({
+ [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META },
+ [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, [VIDEO_FILE]) },
+ }),
+ aria2cAvailable: async () => true,
+ fetchByTorrent: async (o) => {
+ const file = path.join(o.stagingDir, VIDEO_ITEM, VIDEO_FILE);
+ await mkdir(path.dirname(file), { recursive: true });
+ await writeFile(file, BAD);
+ return { ok: true, file, seeded: true };
+ },
+ fetchDirect: writeTo(GOOD),
+ finalize: async (o) => {
+ await writeFile(path.join(o.videoDir, "audio.mp3"), "a");
+ },
+ },
+ });
+ assert.equal(rec.status, "ok");
+ assert.match(log, /fell back to direct download: checksum mismatch \(md5 /);
+ assert.deepEqual(rec.attempts.map((a) => a.kind), ["archiveorg-torrent", "archiveorg-direct"]);
+ assert.match(rec.attempts[0].error ?? "", /^checksum mismatch: md5/);
+});
+
+test("a second checksum mismatch fails the record; nothing is kept as audio", async () => {
+ const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: VIDEO_FILE });
+ let directs = 0;
+ const { rec, channelDir } = await download({
+ url,
+ deps: {
+ client: client({
+ [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META },
+ [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, [VIDEO_FILE]) },
+ }),
+ aria2cAvailable: async () => true,
+ fetchByTorrent: async (o) => {
+ const file = path.join(o.stagingDir, "x.mp4");
+ await mkdir(o.stagingDir, { recursive: true });
+ await writeFile(file, BAD);
+ return { ok: true, file, seeded: false };
+ },
+ fetchDirect: async (o) => {
+ directs++;
+ await writeFile(o.dest, BAD);
+ },
+ },
+ });
+ assert.equal(directs, 1, "one direct download after the torrent's mismatch, no more");
+ assert.equal(rec.status, "failed");
+ assert.equal(rec.failureClass, "per_video");
+ assert.match(rec.attempts.at(-1)?.error ?? "", /checksum mismatch/);
+ const dir = path.join(channelDir, "data", rec.videoId);
+ const entries = await readdir(dir);
+ assert.ok(!entries.some((e) => e.startsWith("audio.")));
+ await assert.rejects(stat(path.join(channelDir, "archive")));
+});
+
+test("direct only (no torrent): a mismatch is retried once, then fails", async () => {
+ let directs = 0;
+ const { rec } = await download({
+ url: archiveOrgDetailsUrl({ identifier: AUDIO_ITEM }),
+ deps: {
+ client: client({ [metaUrl(AUDIO_ITEM)]: { json: AUDIO_META } }),
+ settings: { ...DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, torrent: false },
+ fetchDirect: async (o) => {
+ directs++;
+ await writeFile(o.dest, directs === 1 ? BAD : GOOD);
+ },
+ },
+ });
+ assert.equal(directs, 2);
+ assert.equal(rec.status, "ok");
+ assert.deepEqual(rec.attempts.map((a) => [a.kind, !!a.error]), [
+ ["archiveorg-direct", true],
+ ["archiveorg-direct", false],
+ ]);
+});
+
+test("a verified torrent fetch needs no direct download; the record is the mirror's original", async () => {
+ const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: VIDEO_FILE });
+ const seen: string[] = [];
+ const { rec, channelDir } = await download({
+ url,
+ deps: {
+ client: client(
+ {
+ [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META },
+ [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, ["other.mp4", VIDEO_FILE]) },
+ [`https://archive.org/download/${VIDEO_ITEM}/Clip%20One-AbC123xyz_9.info.json`]: { json: MIRROR_INFO },
+ },
+ seen,
+ ),
+ aria2cAvailable: async () => true,
+ fetchByTorrent: async (o) => {
+ assert.equal(o.entry.index, 2);
+ const file = path.join(o.stagingDir, VIDEO_ITEM, VIDEO_FILE);
+ await mkdir(path.dirname(file), { recursive: true });
+ await writeFile(file, GOOD);
+ return { ok: true, file, seeded: true };
+ },
+ fetchDirect: async () => assert.fail("no direct download after a verified torrent"),
+ finalize: async (o) => {
+ await writeFile(path.join(o.videoDir, "audio.mp3"), "a");
+ },
+ },
+ });
+ assert.equal(rec.status, "ok");
+ assert.deepEqual(rec.attempts.map((a) => a.kind), ["archiveorg-torrent"]);
+ const info = JSON.parse(await readFile(path.join(channelDir, "data", rec.videoId, "metadata.info.json"), "utf8"));
+ assert.equal(info.title, "Clip One (the original)");
+ assert.equal(info.upload_date, "20210102");
+ assert.equal(info.uploader, "Example Uploader");
+ assert.equal(info.webpage_url, url);
+ assert.equal(info.id, rec.videoId);
+ const prov = JSON.parse(await readFile(path.join(channelDir, "data", rec.videoId, "archiveorg.json"), "utf8"));
+ assert.equal(prov.mirror.id, "AbC123xyz_9");
+ assert.equal(prov.mirror.from, "info-json");
+ // One metadata request, one torrent, one info.json.
+ assert.equal(seen.length, 3);
+});
+
+test("an .avi original is fetched as archive.org's mp4 of it; a file not in the torrent goes direct", async () => {
+ const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: "Clip Two-Def456uvw_8.avi" });
+ const directs: string[] = [];
+ const { rec, log } = await download({
+ url,
+ deps: {
+ client: client({
+ [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META },
+ [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, [VIDEO_FILE]) },
+ }),
+ aria2cAvailable: async () => true,
+ fetchByTorrent: async () => assert.fail("not in the torrent"),
+ fetchDirect: async (o) => {
+ directs.push(o.file);
+ await writeFile(o.dest, GOOD);
+ },
+ finalize: async (o) => {
+ assert.equal(o.sourceFilename, "source-media.mp4");
+ await writeFile(path.join(o.videoDir, "audio.mp3"), "a");
+ },
+ },
+ });
+ assert.equal(rec.status, "ok");
+ assert.deepEqual(directs, ["Clip Two-Def456uvw_8.mp4"]);
+ assert.match(log, /does not carry Clip Two-Def456uvw_8\.mp4/);
+});
+
+test("archive.org refusing the direct download is a rate limit; the batch backs off", async () => {
+ const { rec } = await download({
+ url: archiveOrgDetailsUrl({ identifier: AUDIO_ITEM }),
+ deps: {
+ client: client({ [metaUrl(AUDIO_ITEM)]: { json: AUDIO_META } }),
+ aria2cAvailable: async () => false,
+ fetchDirect: async () => {
+ throw new ArchiveOrgRequestError("archive.org did not deliver x after 4 attempts (HTTP 429); stopping", null, true);
+ },
+ },
+ });
+ assert.equal(rec.status, "failed");
+ assert.equal(rec.failureClass, "rate_limit");
+});
+
+test("an item URL of an item with several media files is refused, nothing fetched", async () => {
+ const { rec, log } = await download({
+ url: archiveOrgDetailsUrl({ identifier: VIDEO_ITEM }),
+ deps: {
+ client: client({ [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META } }),
+ fetchDirect: async () => assert.fail("nothing to fetch"),
+ },
+ });
+ assert.equal(rec.status, "failed");
+ assert.equal(rec.failureClass, "per_video");
+ assert.match(log, /holds 2 media files/);
+});
diff --git a/common/lib/archiveOrg.test.ts b/common/lib/archiveOrg.test.ts
@@ -13,6 +13,12 @@ import {
type ArchiveOrgItemMetadata,
} from "./archiveOrg";
import { archiveOrgVideoId } from "./archiveOrgId";
+import {
+ archiveOrgFetchFile,
+ archiveOrgInfoJson,
+ archiveOrgRecordFile,
+ parseArchiveOrgLength,
+} from "./archiveOrg";
import { platformFromMetadata, summarize } from "./transcripts-server";
// Run with:
@@ -267,3 +273,97 @@ test("platformFromMetadata and summarize know an archive.org record", () => {
const yt = summarize("c", "AbC123xyz_9", { id: "AbC123xyz_9", extractor_key: "Youtube" });
assert.equal("mediaUrl" in yt, false);
});
+
+// ─── The record without yt-dlp (controller/archiveOrgDownload.ts) ───
+
+const FETCH_ITEM: ArchiveOrgItemMetadata = {
+ metadata: {
+ identifier: "example-tapes",
+ title: "Example Tapes",
+ creator: "Example Archivist",
+ uploader: "someone@example.org",
+ publicdate: "2020-02-03 10:11:12",
+ description: ["Part one.", "Part two."],
+ },
+ files: [
+ { name: "tape1.flac", source: "original", length: "61.5" },
+ { name: "tape1.mp3", source: "derivative", original: "tape1.flac", format: "VBR MP3", length: "61.48" },
+ { name: "tape1.ogg", source: "derivative", original: "tape1.flac", format: "Ogg Vorbis" },
+ { name: "talk.avi", source: "original" },
+ { name: "talk.mp4", source: "derivative", original: "talk.avi", format: "h.264" },
+ { name: "film.mp4", source: "original" },
+ { name: "film.ogv", source: "derivative", original: "film.mp4" },
+ ],
+};
+
+test("the file fetched: a common container as is, else archive.org's mp4/mp3 of it", () => {
+ assert.equal(archiveOrgFetchFile(FETCH_ITEM, "film.mp4")?.name, "film.mp4");
+ assert.equal(archiveOrgFetchFile(FETCH_ITEM, "talk.avi")?.name, "talk.mp4");
+ assert.equal(archiveOrgFetchFile(FETCH_ITEM, "tape1.flac")?.name, "tape1.mp3");
+ assert.equal(archiveOrgFetchFile(FETCH_ITEM, "missing.mp4"), null);
+ assert.equal(archiveOrgRecordFile(FETCH_ITEM, "talk.avi"), "talk.avi");
+ assert.equal(archiveOrgRecordFile(FETCH_ITEM, undefined), null, "several media files: none is the item");
+ assert.equal(
+ archiveOrgRecordFile({ metadata: { identifier: "x" }, files: [{ name: "only.mp3", source: "original" }] }, undefined),
+ "only.mp3",
+ );
+});
+
+test("archive.org lengths: seconds or a clock", () => {
+ assert.equal(parseArchiveOrgLength("123.45"), 123.45);
+ assert.equal(parseArchiveOrgLength("02:03"), 123);
+ assert.equal(parseArchiveOrgLength("1:02:03"), 3723);
+ assert.equal(parseArchiveOrgLength("soon"), undefined);
+});
+
+test("the synthesised record: ArchiveOrg, the canonical id, the file's page, formats, the date last", () => {
+ const prov = buildArchiveOrgProvenance({
+ ref: { identifier: "example-tapes", file: "tape1.flac" },
+ item: FETCH_ITEM,
+ fetchedAt: "2026-10-05T00:00:00.000Z",
+ });
+ const fetched = archiveOrgFetchFile(FETCH_ITEM, "tape1.flac")!;
+ const info = archiveOrgInfoJson({ prov, item: FETCH_ITEM, file: "tape1.flac", fetched });
+ assert.equal(info.extractor_key, "ArchiveOrg");
+ assert.equal(platformFromMetadata(info), "archiveorg");
+ assert.equal(info.id, archiveOrgVideoId({ identifier: "example-tapes", file: "tape1.flac" }));
+ assert.equal(info.webpage_url, "https://archive.org/details/example-tapes/tape1.flac");
+ // One file of many: its own name, never the item's title.
+ assert.equal(info.title, "tape1");
+ assert.equal(info.uploader, "Example Archivist");
+ assert.equal(info.duration, 61.48);
+ assert.equal(info.ext, "mp3");
+ assert.equal(info.description, "Part one.\n\nPart two.");
+ assert.equal(info.upload_date, "20200203");
+ assert.equal(Object.keys(info).at(-1), "upload_date");
+ assert.deepEqual(
+ (info.formats as { format_id: string; format_note: string }[]).map((f) => [f.format_id, f.format_note]),
+ [
+ ["tape1.flac", "original"],
+ ["tape1.mp3", "derivative"],
+ ["tape1.ogg", "derivative"],
+ ],
+ );
+ // The player plays a playable one of them.
+ assert.equal(summarize("c", String(info.id), info).mediaUrl, "https://archive.org/download/example-tapes/tape1.mp3");
+});
+
+test("a mirror's record is the original's: title, date, uploader from its uploaded info.json", () => {
+ const item: ArchiveOrgItemMetadata = {
+ metadata: { identifier: "youtube-AbC123xyz_9", title: "Mirror", date: "2024-01-01" },
+ files: [{ name: "Clip.mp4", source: "original" }, { name: "Clip.info.json", source: "original" }],
+ };
+ const prov = buildArchiveOrgProvenance({
+ ref: { identifier: "youtube-AbC123xyz_9" },
+ item,
+ infoJson: { id: "AbC123xyz_9", extractor_key: "Youtube", title: "The Original", upload_date: "20190909", uploader: "Orig Channel" },
+ fetchedAt: "2026-10-05T00:00:00.000Z",
+ });
+ const info = archiveOrgInfoJson({ prov, item, file: "Clip.mp4", fetched: item.files[0], durationSec: 10 });
+ assert.equal(info.id, "youtube-AbC123xyz_9");
+ assert.equal(info.title, "The Original");
+ assert.equal(info.upload_date, "20190909");
+ assert.equal(info.uploader, "Orig Channel");
+ assert.equal(info.duration, 10);
+ assert.equal(info.webpage_url, "https://archive.org/details/youtube-AbC123xyz_9");
+});
diff --git a/common/lib/archiveOrgClient.test.ts b/common/lib/archiveOrgClient.test.ts
@@ -1,5 +1,8 @@
import { test } from "node:test";
import assert from "node:assert/strict";
+import { mkdtemp, readFile, rm, writeFile, stat } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
import {
ARCHIVE_ORG_USER_AGENT,
ArchiveOrgClient,
@@ -129,3 +132,106 @@ test("Retry-After as seconds or as an HTTP date", () => {
assert.equal(parseRetryAfterMs(null, 0), null);
assert.equal(parseRetryAfterMs("soon", 0), null);
});
+
+// ─── A file's bytes (the fallback when a torrent cannot be used) ───
+
+type FileStep = { status: number; body?: string; headers?: Record<string, string>; expectRange?: string | null };
+
+function fileHarness(steps: FileStep[]) {
+ const sleeps: number[] = [];
+ const ranges: (string | null)[] = [];
+ const client = new ArchiveOrgClient(
+ {
+ sleep: async (ms) => {
+ sleeps.push(ms);
+ },
+ random: () => 0.5,
+ fetch: async (url, init) => {
+ const h = new Headers(init.headers);
+ ranges.push(h.get("range"));
+ assert.equal(h.get("user-agent"), ARCHIVE_ORG_USER_AGENT);
+ assert.equal(url, "https://archive.org/download/example-item/Clip%20One.mp4");
+ const s = steps.shift();
+ if (!s) throw new Error("script exhausted");
+ return new Response(s.body ?? "", { status: s.status, headers: s.headers });
+ },
+ },
+ { minGapMs: 0, baseBackoffMs: 1000 },
+ );
+ return { client, sleeps, ranges };
+}
+
+test("downloadFile: a partial on disk is resumed with a Range request and appended", async () => {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "aoc-dl-"));
+ try {
+ const dest = path.join(dir, "Clip One.mp4");
+ await writeFile(`${dest}.part`, "hello ");
+ const { client, ranges } = fileHarness([
+ { status: 206, body: "world", headers: { "content-length": "5" } },
+ ]);
+ const r = await client.downloadFile("example-item", "Clip One.mp4", dest, { expectedSize: 11 });
+ assert.equal(r.bytes, 11);
+ assert.deepEqual(ranges, ["bytes=6-"]);
+ assert.equal(await readFile(dest, "utf8"), "hello world");
+ await assert.rejects(stat(`${dest}.part`));
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("downloadFile: a 503 waits its Retry-After; a cut-off stream resumes where it stopped", async () => {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "aoc-dl-"));
+ try {
+ const dest = path.join(dir, "Clip One.mp4");
+ const { client, sleeps, ranges } = fileHarness([
+ { status: 503, headers: { "retry-after": "7" } },
+ // Says 11 bytes, sends 4: the stream was cut.
+ { status: 200, body: "hell", headers: { "content-length": "11" } },
+ { status: 206, body: "o world", headers: { "content-length": "7" } },
+ ]);
+ const r = await client.downloadFile("example-item", "Clip One.mp4", dest);
+ assert.equal(r.bytes, 11);
+ assert.deepEqual(ranges, [null, null, "bytes=4-"]);
+ assert.equal(sleeps[0], 7000);
+ assert.equal(await readFile(dest, "utf8"), "hello world");
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("downloadFile: a 404 is final; archive.org refusing to the end is a rate limit", async () => {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "aoc-dl-"));
+ try {
+ const dest = path.join(dir, "Clip One.mp4");
+ const gone = fileHarness([{ status: 404 }]);
+ await assert.rejects(
+ gone.client.downloadFile("example-item", "Clip One.mp4", dest),
+ (e: unknown) => e instanceof ArchiveOrgRequestError && e.status === 404 && !e.rateLimited,
+ );
+ const busy = fileHarness([{ status: 429 }, { status: 429 }, { status: 429 }, { status: 429 }]);
+ await assert.rejects(
+ busy.client.downloadFile("example-item", "Clip One.mp4", dest),
+ (e: unknown) => e instanceof ArchiveOrgRequestError && e.rateLimited,
+ );
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("itemTorrent: fetched once, then cached", async () => {
+ let calls = 0;
+ const client = new ArchiveOrgClient(
+ {
+ sleep: async () => {},
+ fetch: async (url) => {
+ calls++;
+ assert.equal(url, "https://archive.org/download/example-item/example-item_archive.torrent");
+ return new Response(new Uint8Array([100, 101]), { status: 200 });
+ },
+ },
+ { minGapMs: 0 },
+ );
+ assert.deepEqual([...(await client.itemTorrent("example-item"))], [100, 101]);
+ await client.itemTorrent("example-item");
+ assert.equal(calls, 1);
+});
diff --git a/common/lib/archiveOrgTorrent-server.test.ts b/common/lib/archiveOrgTorrent-server.test.ts
@@ -0,0 +1,156 @@
+import { test, after } from "node:test";
+import assert from "node:assert/strict";
+import { chmod, mkdtemp, readFile, rm, stat, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { fetchFileByTorrent, aria2cAvailable, __resetAria2cProbeForTest } from "./archiveOrgTorrent-server";
+import { DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, type ParsedTorrent } from "./archiveOrgTorrent";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrgTorrent-server.test.ts
+//
+// One aria2c run against a FAKE aria2c (a node script below) that prints what
+// aria2c 1.37 prints and does what the mode in FAKE_ARIA2C_MODE says: complete
+// (write the file, run the --on-bt-download-complete hook, seed, exit 0),
+// stall (the same progress forever), fail (an error, exit 1), hang (progress,
+// a child process of its own, until killed). Nothing touches the network.
+
+const ROOT = await mkdtemp(path.join(os.tmpdir(), "aria2c-run-"));
+after(() => rm(ROOT, { recursive: true, force: true }));
+
+const FAKE = path.join(ROOT, "fake-aria2c.mjs");
+await writeFile(
+ FAKE,
+ `#!/usr/bin/env node
+import { mkdirSync, writeFileSync } from "node:fs";
+import { spawn, execFileSync } from "node:child_process";
+import path from "node:path";
+const args = process.argv.slice(2);
+if (args[0] === "--version") { console.log("aria2 version 1.37.0"); process.exit(0); }
+const opt = (k) => (args.find((a) => a.startsWith("--" + k + "=")) ?? "").slice(k.length + 3);
+const dir = opt("dir");
+const hook = opt("on-bt-download-complete");
+const mode = process.env.FAKE_ARIA2C_MODE;
+const target = path.join(dir, "example-item", "Clip One.mp4");
+const say = (s) => process.stdout.write(s + "\\n");
+const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
+writeFileSync(path.join(process.env.FAKE_ARIA2C_LOG, "args.json"), JSON.stringify(args));
+say("10/05 15:46:34 [NOTICE] Downloading 1 item(s)");
+if (mode === "fail") { say("10/05 15:46:35 [ERROR] CUID#7 - Download aborted. URI=x"); process.exit(1); }
+if (mode === "stall") { for (;;) { say("[#2089b0 16KiB/40KiB(40%) CN:1 SD:0 DL:0B]"); await sleep(50); } }
+if (mode === "hang") {
+ const c = spawn("sleep", ["30"], { stdio: "ignore" });
+ writeFileSync(path.join(process.env.FAKE_ARIA2C_LOG, "pids.json"), JSON.stringify([process.pid, c.pid]));
+ let n = 0;
+ for (;;) { n++; say("[#2089b0 " + n + "KiB/40KiB(" + n + "%) CN:1 SD:0 DL:1KiB]"); await sleep(50); }
+}
+// complete
+say(" *** Download Progress Summary as of Mon Oct 5 15:46:35 2026 *** ");
+say("[#2089b0 16KiB/40KiB(40%) CN:3 SD:1 DL:16KiB ETA:1s]");
+say("FILE: " + target);
+mkdirSync(path.dirname(target), { recursive: true });
+writeFileSync(target, "x".repeat(40000));
+execFileSync(hook, ["2089b0", "1", target]);
+say("[#2089b0 SEED(0.0) CN:1 SD:0 UL:0B(0B)]");
+await sleep(150);
+say("10/05 15:46:39 [NOTICE] Seeding is over.");
+process.exit(0);
+`,
+);
+await chmod(FAKE, 0o755);
+process.env.FAKE_ARIA2C_LOG = ROOT;
+
+const PARSED: ParsedTorrent = {
+ name: "example-item",
+ pieceLength: 16384,
+ pieceCount: 3,
+ multiFile: true,
+ files: [{ index: 2, path: "Clip One.mp4", length: 40000, offset: 1000 }],
+ webSeeds: ["https://archive.org/download/"],
+ trackers: [],
+};
+
+function run(mode: string, extra: { signal?: AbortSignal; stallMs?: number } = {}) {
+ process.env.FAKE_ARIA2C_MODE = mode;
+ const lines: string[] = [];
+ const stagingDir = path.join(ROOT, `staging-${mode}`);
+ const p = fetchFileByTorrent({
+ aria2cBin: FAKE,
+ torrent: Buffer.from("d4:infod4:name1:xee"),
+ parsed: PARSED,
+ entry: PARSED.files[0],
+ stagingDir,
+ settings: { ...DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, seedMinutes: 1 },
+ userAgent: "test",
+ onLog: (l) => lines.push(l),
+ signal: extra.signal ?? new AbortController().signal,
+ tickMs: 20,
+ progressLogMs: 0,
+ killGraceMs: 300,
+ ...(extra.stallMs ? { stallMs: extra.stallMs } : {}),
+ });
+ return { p, lines, stagingDir };
+}
+
+function alive(pid: number): boolean {
+ try {
+ process.kill(pid, 0);
+ return true;
+ } catch {
+ return false;
+ }
+}
+
+test("aria2cAvailable: present is true, a missing binary is false", async () => {
+ __resetAria2cProbeForTest();
+ assert.equal(await aria2cAvailable(FAKE), true);
+ assert.equal(await aria2cAvailable(path.join(ROOT, "no-such-aria2c")), false);
+});
+
+test("complete: the file, the hook, seeding, a human progress line", async () => {
+ const { p, lines, stagingDir } = run("complete");
+ const res = await p;
+ assert.equal(res.ok, true);
+ assert.equal(res.ok && res.file, path.join(stagingDir, "example-item", "Clip One.mp4"));
+ assert.equal(res.ok && res.seeded, true);
+ assert.equal((await stat(path.join(stagingDir, "example-item", "Clip One.mp4"))).size, 40000);
+ assert.ok(lines.some((l) => l === "torrent: Clip One.mp4 (1 of 3 pieces, peers 3 (1 seeding), web seed yes)\n"), lines.join(""));
+ assert.ok(lines.some((l) => /seeding 1 min or to ratio 1, whichever comes first/.test(l)));
+ assert.ok(lines.some((l) => /aria2c: \[NOTICE\] Seeding is over\./.test(l)));
+ const args = JSON.parse(await readFile(path.join(ROOT, "args.json"), "utf8")) as string[];
+ assert.ok(args.includes("--select-file=2"));
+ assert.ok(args.includes(`--stop-with-process=${process.pid}`));
+});
+
+test("stall: no progress for the stall time stops aria2c, flagged stalled", async () => {
+ const { p } = run("stall", { stallMs: 300 });
+ const res = await p;
+ assert.equal(res.ok, false);
+ assert.equal(!res.ok && res.stalled, true);
+ assert.match(!res.ok ? res.reason : "", /stalled: no progress/);
+});
+
+test("fail: aria2c's error is the reason", async () => {
+ const res = await run("fail").p;
+ assert.equal(res.ok, false);
+ assert.match(!res.ok ? res.reason : "", /aria2c exited 1: .*Download aborted/);
+});
+
+test("cancel: the whole process group is killed, the result says cancelled", async () => {
+ const ctl = new AbortController();
+ const { p } = run("hang", { signal: ctl.signal });
+ // Let it start and record its pids.
+ let pids: number[] = [];
+ for (let i = 0; i < 100 && pids.length === 0; i++) {
+ await new Promise((r) => setTimeout(r, 20));
+ pids = await readFile(path.join(ROOT, "pids.json"), "utf8").then((t) => JSON.parse(t), () => []);
+ }
+ assert.equal(pids.length, 2);
+ ctl.abort();
+ const res = await p;
+ assert.equal(res.ok, false);
+ assert.equal(!res.ok && res.cancelled, true);
+ await new Promise((r) => setTimeout(r, 100));
+ assert.equal(alive(pids[0]), false, "aria2c is gone");
+ assert.equal(alive(pids[1]), false, "its child is gone with it");
+});
diff --git a/common/lib/archiveOrgTorrent-server.ts b/common/lib/archiveOrgTorrent-server.ts
@@ -83,6 +83,8 @@ export type TorrentFetchOpts = {
killGraceMs?: number;
// How long past --seed-time a seeding aria2c may run before it is stopped.
seedGraceMs?: number;
+ // Overrides settings.stallMinutes (tests).
+ stallMs?: number;
now?: () => number;
};
@@ -116,7 +118,7 @@ export async function fetchFileByTorrent(opts: TorrentFetchOpts): Promise<Torren
const tickMs = opts.tickMs ?? 2_000;
const progressLogMs = opts.progressLogMs ?? 30_000;
const killGraceMs = opts.killGraceMs ?? 10_000;
- const stallMs = Math.max(1, opts.settings.stallMinutes) * 60_000;
+ const stallMs = opts.stallMs ?? Math.max(1, opts.settings.stallMinutes) * 60_000;
const seedMs = Math.max(0, opts.settings.seedMinutes) * 60_000;
const seedGraceMs = opts.seedGraceMs ?? 5 * 60_000;
const log = (s: string) => opts.onLog(s.endsWith("\n") ? s : `${s}\n`);
diff --git a/common/lib/archiveOrgTorrent.test.ts b/common/lib/archiveOrgTorrent.test.ts
@@ -0,0 +1,190 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { bdecode, BencodeError } from "./bencode";
+import {
+ DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS,
+ aria2cArgs,
+ findTorrentFile,
+ parseAria2cSize,
+ parseAria2cStatusLine,
+ parseTorrent,
+ torrentFilePieces,
+ torrentFileSegments,
+ torrentProgressLine,
+} from "./archiveOrgTorrent";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrgTorrent.test.ts
+//
+// A synthetic torrent, built here by a ten-line encoder, shaped like
+// archive.org's `<identifier>_archive.torrent`: name = identifier, a files
+// list, archive.org web seeds. Every name is invented.
+
+type Enc = number | string | Buffer | Enc[] | { [k: string]: Enc };
+function benc(v: Enc): Buffer {
+ if (typeof v === "number") return Buffer.from(`i${v}e`);
+ if (typeof v === "string") return benc(Buffer.from(v, "utf8"));
+ if (Buffer.isBuffer(v)) return Buffer.concat([Buffer.from(`${v.length}:`), v]);
+ if (Array.isArray(v)) return Buffer.concat([Buffer.from("l"), ...v.map(benc), Buffer.from("e")]);
+ const keys = Object.keys(v).sort();
+ return Buffer.concat([Buffer.from("d"), ...keys.flatMap((k) => [benc(k), benc(v[k])]), Buffer.from("e")]);
+}
+
+const PIECE = 16384;
+const FILES = [
+ { path: ["example-item_meta.xml"], length: 1000 },
+ { path: ["Clip One.mp4"], length: 40000 },
+ { path: ["sub dir", "Clip Two [AbC123xyz_9].mp4"], length: 50000 },
+ { path: ["Clip One.info.json"], length: 2000 },
+];
+const total = FILES.reduce((n, f) => n + f.length, 0);
+const TORRENT = benc({
+ announce: "http://bt1.archive.org:6969/announce",
+ "announce-list": [["http://bt1.archive.org:6969/announce"], ["http://bt2.archive.org:6969/announce"]],
+ "url-list": ["https://archive.org/download/", "http://ia000000.us.archive.org/0/items/"],
+ info: {
+ name: "example-item",
+ "piece length": PIECE,
+ pieces: Buffer.alloc(20 * Math.ceil(total / PIECE)),
+ files: FILES,
+ },
+});
+
+test("bencode: integers, strings, lists, dictionaries", () => {
+ assert.equal(bdecode(Buffer.from("i-42e")), -42);
+ assert.deepEqual(bdecode(Buffer.from("4:spam")), Buffer.from("spam"));
+ const v = bdecode(Buffer.from("d3:bari1e3:fool1:a1:bee")) as Record<string, unknown>;
+ assert.equal(v.bar, 1);
+ assert.deepEqual((v.foo as Buffer[]).map((b) => b.toString()), ["a", "b"]);
+});
+
+test("bencode: malformed input throws, naming the offset", () => {
+ for (const bad of ["i12", "5:abc", "l1:a", "d1:ai1e", "x", "i1ei2e", "i1x2e"]) {
+ assert.throws(() => bdecode(Buffer.from(bad)), BencodeError, bad);
+ }
+});
+
+test("a torrent: name, pieces, files with 1-based indexes and offsets, web seeds, trackers", () => {
+ const t = parseTorrent(TORRENT)!;
+ assert.equal(t.name, "example-item");
+ assert.equal(t.pieceLength, PIECE);
+ assert.equal(t.pieceCount, Math.ceil(total / PIECE));
+ assert.equal(t.multiFile, true);
+ assert.deepEqual(
+ t.files.map((f) => [f.index, f.path, f.offset]),
+ [
+ [1, "example-item_meta.xml", 0],
+ [2, "Clip One.mp4", 1000],
+ [3, "sub dir/Clip Two [AbC123xyz_9].mp4", 41000],
+ [4, "Clip One.info.json", 91000],
+ ],
+ );
+ assert.equal(t.webSeeds.length, 2);
+ assert.deepEqual(t.trackers, ["http://bt1.archive.org:6969/announce", "http://bt2.archive.org:6969/announce"]);
+});
+
+test("the file's index is found by its exact path in the item; absent is null", () => {
+ const t = parseTorrent(TORRENT)!;
+ assert.equal(findTorrentFile(t, "sub dir/Clip Two [AbC123xyz_9].mp4")?.index, 3);
+ assert.equal(findTorrentFile(t, "Clip One.mp4")?.index, 2);
+ assert.equal(findTorrentFile(t, "clip one.mp4"), null);
+ assert.equal(findTorrentFile(t, "Added Later.mp4"), null);
+ const two = findTorrentFile(t, "sub dir/Clip Two [AbC123xyz_9].mp4")!;
+ assert.deepEqual(torrentFileSegments(t, two), ["example-item", "sub dir", "Clip Two [AbC123xyz_9].mp4"]);
+ // 41000..90999 spans pieces 2..5 of 16 KiB.
+ assert.deepEqual(torrentFilePieces(t, two), { first: 2, last: 5, count: 4 });
+});
+
+test("path.utf-8 wins over path; not a torrent is null", () => {
+ const t = parseTorrent(
+ benc({
+ info: {
+ name: "x",
+ "piece length": PIECE,
+ pieces: Buffer.alloc(20),
+ files: [{ path: ["mangled"], "path.utf-8": ["Café.mp3"], length: 10 }],
+ },
+ }),
+ )!;
+ assert.equal(t.files[0].path, "Café.mp3");
+ assert.equal(parseTorrent(Buffer.from("not bencode")), null);
+ assert.equal(parseTorrent(benc({ announce: "x" })), null);
+});
+
+test("a single-file torrent writes <dir>/<name>", () => {
+ const t = parseTorrent(benc({ info: { name: "solo.mp3", "piece length": PIECE, pieces: Buffer.alloc(20), length: 99 } }))!;
+ assert.equal(t.multiFile, false);
+ assert.deepEqual(torrentFileSegments(t, t.files[0]), ["solo.mp3"]);
+});
+
+test("aria2c arguments: one file, seeding bounds, politeness, the hook, no word-splitting", () => {
+ const args = aria2cArgs({
+ torrentPath: "/s/item.torrent",
+ dir: "/s/dir with space",
+ fileIndex: 3,
+ settings: { ...DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, maxUploadKiBps: 200 },
+ onCompleteHook: "/s/on-complete.sh",
+ userAgent: "Example UA (+https://example.org)",
+ parentPid: 1234,
+ });
+ assert.ok(args.includes("--dir=/s/dir with space"));
+ assert.ok(args.includes("--select-file=3"));
+ assert.ok(args.includes("--seed-time=10"));
+ assert.ok(args.includes("--seed-ratio=1.0"));
+ assert.ok(args.includes("--bt-max-peers=30"));
+ assert.ok(args.includes("--max-overall-download-limit=0"));
+ assert.ok(args.includes("--max-overall-upload-limit=200K"));
+ assert.ok(args.includes("--bt-remove-unselected-file=true"));
+ assert.ok(args.includes("--follow-torrent=mem"));
+ assert.ok(args.includes("--file-allocation=none"));
+ assert.ok(args.includes("--max-connection-per-server=1"));
+ assert.ok(args.includes("--on-bt-download-complete=/s/on-complete.sh"));
+ assert.ok(args.includes("--user-agent=Example UA (+https://example.org)"));
+ assert.ok(args.includes("--stop-with-process=1234"));
+ assert.equal(args.at(-1), "--torrent-file=/s/item.torrent");
+ assert.ok(args.every((a) => a.startsWith("--")));
+ // No seeding when the operator says so.
+ const none = aria2cArgs({
+ torrentPath: "t",
+ dir: "d",
+ fileIndex: 1,
+ settings: { ...DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, seedMinutes: 0, seedRatio: 0 },
+ onCompleteHook: "h",
+ userAgent: "u",
+ });
+ assert.ok(none.includes("--seed-time=0"));
+ assert.ok(!none.some((a) => a.startsWith("--stop-with-process")));
+});
+
+test("aria2c progress lines, as 1.37 prints them", () => {
+ assert.equal(parseAria2cSize("16KiB"), 16384);
+ assert.equal(parseAria2cSize("2.8MiB"), Math.round(2.8 * 1024 * 1024));
+ assert.equal(parseAria2cSize("0B"), 0);
+ assert.equal(parseAria2cSize("lots"), null);
+ assert.deepEqual(parseAria2cStatusLine("[#09cba8 608KiB/2.8MiB(20%) CN:1 DL:302KiB ETA:7s]"), {
+ seeding: false,
+ completedBytes: 608 * 1024,
+ totalBytes: Math.round(2.8 * 1024 * 1024),
+ percent: 20,
+ connections: 1,
+ });
+ const bt = parseAria2cStatusLine("[#2089b0 1.1MiB/33MiB(3%) CN:5 SD:2 DL:115KiB UL:4KiB(16KiB) ETA:4m48s]");
+ assert.equal(bt?.seeding, false);
+ assert.equal(bt && !bt.seeding ? bt.seeders : -1, 2);
+ assert.equal(bt?.connections, 5);
+ const seed = parseAria2cStatusLine("[#2089b0 SEED(0.4) CN:2 SD:0 UL:12KiB(4.9MiB)]");
+ assert.equal(seed?.seeding, true);
+ assert.equal(seed && seed.seeding ? seed.ratio : -1, 0.4);
+ assert.equal(parseAria2cStatusLine("FILE: out/blob.bin"), null);
+ assert.equal(parseAria2cStatusLine("10/05 15:46:39 [NOTICE] Download complete: out/blob.bin"), null);
+});
+
+test("the human line: pieces of the file's own span, peers, web seed", () => {
+ const line = torrentProgressLine({
+ file: "Clip One.mp4",
+ status: { seeding: false, completedBytes: 50, totalBytes: 100, percent: 50, connections: 4, seeders: 1 },
+ pieces: 9,
+ webSeed: true,
+ });
+ assert.equal(line, "torrent: Clip One.mp4 (4 of 9 pieces, peers 4 (1 seeding), web seed yes)");
+});
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -64,7 +64,10 @@ import {
type MetadataScanEntry,
} from "../controller/metadataScanStore";
import { detectPlatform, type Platform } from "../lib/platform";
-import { downloadArchiveOrgManaged } from "../controller/archiveOrgDownload";
+import {
+ downloadArchiveOrgManaged,
+ type ArchiveOrgDownloadDeps,
+} from "../controller/archiveOrgDownload";
import { finalizeAppExtraction } from "./finalizeAppExtraction";
import { probeMediaDurationSec } from "./ffprobeDuration";
import {
@@ -186,6 +189,9 @@ export type ManagedDownloadOpts = {
// "video_720" = the ≤720p H.264 selector (downloadFormat.ts). Never touches
// the audio-only selector above.
persistFormatPreset?: SourceVideoQuality;
+ // Test seam: what the archive.org download (controller/archiveOrgDownload.ts)
+ // talks to. Every production caller passes none.
+ archiveOrgDeps?: ArchiveOrgDownloadDeps;
};
// When `reuseInfoJson` is true, the real download reuses the metadata the
@@ -648,7 +654,7 @@ export async function downloadOneManaged(
// checksums, and the record is built from the item's metadata
// (controller/archiveOrgDownload.ts). Same log, same outcome sidecar.
if (detectPlatform(opts.videoUrl) === "archiveorg") {
- return await downloadArchiveOrgManaged(opts);
+ return await downloadArchiveOrgManaged(opts, opts.archiveOrgDeps);
}
return await runManagedDownload(opts, channelDir, startedAt, canonicalId);
} finally {