Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 5e314ea46a132ac9413641ae755d8c4f169251fd
parent a1a5dea0f932cc5b0fa49215979bf5567f5bc409
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 16:00:59 -0400

sources: tests for the archive.org torrent ladder

A synthetic torrent (bencode, file index, piece span, path.utf-8,
single-file); the aria2c argument builder and 1.37's progress lines; one
aria2c run against a fake aria2c (complete + seed, stall, failure, cancel
killing the whole process group); the ladder through downloadOneManaged with
a yt-dlp that is never called (no aria2c → direct, stall → direct, torrent
checksum mismatch → direct once, a second mismatch fails, direct-only retry,
a verified torrent, an .avi fetched as its mp4, a rate-limited direct
download); the client's resumable direct download and cached torrent; the
synthesised info.json (ArchiveOrg maps to archiveorg, a mirror's original
kept, the date last). archiveOrgDeps is the managed download's test seam.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/controller/archiveOrgDownload.test.ts | 422+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/archiveOrg.test.ts | 100+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/archiveOrgClient.test.ts | 106+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archiveOrgTorrent-server.test.ts | 156+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/archiveOrgTorrent-server.ts | 4+++-
Acommon/lib/archiveOrgTorrent.test.ts | 190+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/ytdlp/downloadOneManaged.ts | 10++++++++--
7 files changed, 985 insertions(+), 3 deletions(-)

diff --git a/common/controller/archiveOrgDownload.test.ts b/common/controller/archiveOrgDownload.test.ts @@ -0,0 +1,422 @@ +import { test, after } from "node:test"; +import assert from "node:assert/strict"; +import { createHash } from "node:crypto"; +import { chmod, mkdir, mkdtemp, readFile, readdir, rm, stat, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import type { ChannelConfig } from "../lib/channelConfig"; +import type { ArchiveOrgDownloadDeps } from "./archiveOrgDownload"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/archiveOrgDownload.test.ts +// +// The archive.org download ladder with everything around it faked: a scripted +// archive.org (metadata, torrent, a mirror's info.json), a torrent fetch and a +// direct fetch that write chosen bytes, a duration probe. The checksum check is +// the real one, against md5/sha1 computed here. Every download goes through +// downloadOneManaged, with a yt-dlp that records any call — and none is made. +// Every name is invented. + +const ROOT = await mkdtemp(path.join(os.tmpdir(), "archiveorg-dl-")); +process.env.TRANSCRIPTS_DIR = ROOT; +process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); +await writeFile(process.env.SETTINGS_FILE, JSON.stringify({ minFreeDiskGB: 0 }) + "\n"); +after(() => rm(ROOT, { recursive: true, force: true })); + +const { downloadOneManaged } = await import("../ytdlp/downloadOneManaged"); +const { ArchiveOrgClient } = await import("../lib/archiveOrgClient"); +const { ArchiveOrgRequestError } = await import("../lib/archiveOrgClient"); +const { archiveOrgDetailsUrl, archiveOrgVideoId } = await import("../lib/archiveOrgId"); +const { platformFromMetadata } = await import("../lib/transcripts-server"); +const { DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS } = await import("../lib/archiveOrgTorrent"); + +const YTDLP_CALLED = path.join(ROOT, "ytdlp-called"); +const YTDLP = path.join(ROOT, "fake-ytdlp.sh"); +await writeFile(YTDLP, `#!/bin/sh\necho "$@" >> "${YTDLP_CALLED}"\nexit 1\n`); +await chmod(YTDLP, 0o755); + +const GOOD = Buffer.from("the right bytes of the recording\n".repeat(40)); +// Same length, one byte different: only the checksum can tell. +const BAD = Buffer.from(GOOD); +BAD[7] ^= 0xff; +const md5 = (b: Buffer) => createHash("md5").update(b).digest("hex"); +const sha1 = (b: Buffer) => createHash("sha1").update(b).digest("hex"); + +// ─── A synthetic torrent (the same ten-line encoder as the torrent test) ─── +type Enc = number | string | Buffer | Enc[] | { [k: string]: Enc }; +function benc(v: Enc): Buffer { + if (typeof v === "number") return Buffer.from(`i${v}e`); + if (typeof v === "string") return benc(Buffer.from(v, "utf8")); + if (Buffer.isBuffer(v)) return Buffer.concat([Buffer.from(`${v.length}:`), v]); + if (Array.isArray(v)) return Buffer.concat([Buffer.from("l"), ...v.map(benc), Buffer.from("e")]); + const keys = Object.keys(v).sort(); + return Buffer.concat([Buffer.from("d"), ...keys.flatMap((k) => [benc(k), benc(v[k])]), Buffer.from("e")]); +} +function torrentOf(identifier: string, files: string[]): Buffer { + return benc({ + "url-list": ["https://archive.org/download/"], + info: { + name: identifier, + "piece length": 16384, + pieces: Buffer.alloc(20), + files: files.map((f) => ({ path: f.split("/"), length: GOOD.length })), + }, + }); +} + +// ─── The scripted archive.org ─── +type Route = { json?: unknown; bytes?: Buffer; status?: number }; +function client(routes: Record<string, Route>, seen: string[] = []) { + return new ArchiveOrgClient( + { + sleep: async () => {}, + fetch: async (url) => { + seen.push(url); + const r = routes[url]; + if (!r) return new Response("not found", { status: 404 }); + if (r.status) return new Response("no", { status: r.status }); + if (r.bytes) return new Response(new Uint8Array(r.bytes), { status: 200 }); + return new Response(JSON.stringify(r.json), { status: 200 }); + }, + }, + { minGapMs: 0, maxAttempts: 2 }, + ); +} + +const AUDIO_ITEM = "example-radio-show"; +const AUDIO_META = { + metadata: { + identifier: AUDIO_ITEM, + title: "Example Radio Show", + creator: "Example Station", + uploader: "someone@example.org", + date: "1999-05-04", + description: "An invented show.", + subject: "radio; example", + }, + files: [ + { name: "show.mp3", source: "original", format: "VBR MP3", size: String(GOOD.length), md5: md5(GOOD), sha1: sha1(GOOD), length: "125.5" }, + { name: `${AUDIO_ITEM}_archive.torrent`, source: "metadata" }, + ], +}; + +const VIDEO_ITEM = "example-channel-archive"; +const VIDEO_FILE = "Clip One-AbC123xyz_9.mp4"; +const VIDEO_META = { + metadata: { identifier: VIDEO_ITEM, title: "Example Channel Archive", creator: "Example Creator" }, + files: [ + { name: VIDEO_FILE, source: "original", format: "h.264", size: String(GOOD.length), md5: md5(GOOD) }, + { name: "Clip One-AbC123xyz_9.info.json", source: "original" }, + { name: "Clip Two-Def456uvw_8.avi", source: "original", format: "AVI", size: "99" }, + { name: "Clip Two-Def456uvw_8.mp4", source: "derivative", original: "Clip Two-Def456uvw_8.avi", format: "h.264", size: String(GOOD.length), md5: md5(GOOD) }, + { name: `${VIDEO_ITEM}_archive.torrent`, source: "metadata" }, + ], +}; +const MIRROR_INFO = { + id: "AbC123xyz_9", + extractor_key: "Youtube", + title: "Clip One (the original)", + upload_date: "20210102", + uploader: "Example Uploader", +}; + +const metaUrl = (id: string) => `https://archive.org/metadata/${id}`; +const torrentUrl = (id: string) => `https://archive.org/download/${id}/${id}_archive.torrent`; + +let n = 0; +async function download(opts: { + url: string; + deps: ArchiveOrgDownloadDeps; + config?: Partial<ChannelConfig>; +}) { + const slug = `c${++n}`; + const paths = { + channelsDir: path.join(ROOT, "channels"), + ytdlpBin: YTDLP, + ffprobeBin: "ffprobe", + ffmpegBin: "ffmpeg", + aria2cBin: "aria2c", + } as Paths; + await mkdir(path.join(paths.channelsDir, slug, "data"), { recursive: true }); + const lines: string[] = []; + const rec = await downloadOneManaged({ + channelSlug: slug, + channelConfig: { handling: "transcribe", platform: "archiveorg", audioFormat: "mp3", ...opts.config } as ChannelConfig, + paths, + videoUrl: opts.url, + onLog: (l) => lines.push(l), + signal: new AbortController().signal, + appendArchive: true, + archiveOrgDeps: { + settings: DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, + probeDuration: async () => 125.25, + ...opts.deps, + }, + }); + const channelDir = path.join(paths.channelsDir, slug); + return { rec, lines, log: lines.join(""), channelDir }; +} + +const writeTo = (bytes: Buffer) => async (o: { dest: string }) => { + await writeFile(o.dest, bytes); +}; + +test("an audio item, no aria2c: a direct download IS the audio — no yt-dlp, a full record", async () => { + const seen: string[] = []; + const directs: string[] = []; + const { rec, log, channelDir } = await download({ + url: archiveOrgDetailsUrl({ identifier: AUDIO_ITEM }), + deps: { + client: client({ [metaUrl(AUDIO_ITEM)]: { json: AUDIO_META } }, seen), + aria2cAvailable: async () => false, + fetchDirect: async (o) => { + directs.push(o.file); + await writeFile(o.dest, GOOD); + }, + }, + }); + assert.equal(rec.status, "ok"); + assert.deepEqual(rec.attempts.map((a) => [a.kind, a.ytdlpExitCode]), [["archiveorg-direct", 0]]); + assert.match(log, /fell back to direct download: torrents need aria2c/); + assert.deepEqual(directs, ["show.mp3"]); + await assert.rejects(stat(YTDLP_CALLED), "yt-dlp was never run"); + const dir = path.join(channelDir, "data", AUDIO_ITEM); + assert.deepEqual(await readFile(path.join(dir, "audio.mp3")), GOOD); + const info = JSON.parse(await readFile(path.join(dir, "metadata.info.json"), "utf8")); + assert.equal(info.extractor_key, "ArchiveOrg"); + assert.equal(platformFromMetadata(info), "archiveorg"); + assert.equal(info.id, AUDIO_ITEM); + assert.equal(info.webpage_url, `https://archive.org/details/${AUDIO_ITEM}`); + assert.equal(info.title, "Example Radio Show"); + assert.equal(info.uploader, "Example Station", "the public credit, never the account's e-mail"); + assert.equal(info.duration, 125.25); + assert.equal(info.upload_date, "19990504"); + assert.deepEqual(info.tags, ["radio", "example"]); + assert.equal(Object.keys(info).at(-1), "upload_date"); + assert.deepEqual(info.formats.map((f: { url: string }) => f.url), [`https://archive.org/download/${AUDIO_ITEM}/show.mp3`]); + const prov = JSON.parse(await readFile(path.join(dir, "archiveorg.json"), "utf8")); + assert.equal(prov.identifier, AUDIO_ITEM); + const outcome = JSON.parse(await readFile(path.join(dir, "download-outcome.json"), "utf8")); + assert.equal(outcome.status, "ok"); + assert.equal(await readFile(path.join(channelDir, "archive"), "utf8"), `archiveorg ${AUDIO_ITEM}\n`); + assert.ok(!(await readdir(dir)).includes(".archiveorg-fetch"), "the staging dir is gone"); + assert.match(await readFile(path.join(dir, "download.log"), "utf8"), /Verified show\.mp3 \(sha1\)/); +}); + +test("a stalled torrent falls back to a direct download, saying why", async () => { + const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: VIDEO_FILE }); + let torrentCalls = 0; + const finalized: string[] = []; + const { rec, log } = await download({ + url, + deps: { + client: client({ + [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META }, + [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, [VIDEO_FILE]) }, + }), + aria2cAvailable: async () => true, + fetchByTorrent: async (o) => { + torrentCalls++; + assert.equal(o.entry.path, VIDEO_FILE); + assert.equal(o.entry.index, 1); + return { ok: false, stalled: true, reason: "stalled: no progress in 5 min" }; + }, + fetchDirect: writeTo(GOOD), + finalize: async (o) => { + finalized.push(`${o.sourceFilename} persist=${o.persist}`); + await writeFile(path.join(o.videoDir, "audio.mp3"), "a"); + await rm(path.join(o.videoDir, o.sourceFilename!)); + }, + }, + }); + assert.equal(torrentCalls, 1); + assert.equal(rec.status, "ok"); + assert.match(log, /fell back to direct download: stalled: no progress in 5 min/); + assert.deepEqual(rec.attempts.map((a) => [a.kind, a.error ?? null]), [ + ["archiveorg-torrent", "stalled: no progress in 5 min"], + ["archiveorg-direct", null], + ]); + assert.deepEqual(finalized, ["source-media.mp4 persist=false"]); + assert.equal(rec.videoId, archiveOrgVideoId({ identifier: VIDEO_ITEM, file: VIDEO_FILE })); +}); + +test("a torrent file that fails its checksum falls back once to a direct download", async () => { + const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: VIDEO_FILE }); + const { rec, log } = await download({ + url, + deps: { + client: client({ + [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META }, + [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, [VIDEO_FILE]) }, + }), + aria2cAvailable: async () => true, + fetchByTorrent: async (o) => { + const file = path.join(o.stagingDir, VIDEO_ITEM, VIDEO_FILE); + await mkdir(path.dirname(file), { recursive: true }); + await writeFile(file, BAD); + return { ok: true, file, seeded: true }; + }, + fetchDirect: writeTo(GOOD), + finalize: async (o) => { + await writeFile(path.join(o.videoDir, "audio.mp3"), "a"); + }, + }, + }); + assert.equal(rec.status, "ok"); + assert.match(log, /fell back to direct download: checksum mismatch \(md5 /); + assert.deepEqual(rec.attempts.map((a) => a.kind), ["archiveorg-torrent", "archiveorg-direct"]); + assert.match(rec.attempts[0].error ?? "", /^checksum mismatch: md5/); +}); + +test("a second checksum mismatch fails the record; nothing is kept as audio", async () => { + const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: VIDEO_FILE }); + let directs = 0; + const { rec, channelDir } = await download({ + url, + deps: { + client: client({ + [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META }, + [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, [VIDEO_FILE]) }, + }), + aria2cAvailable: async () => true, + fetchByTorrent: async (o) => { + const file = path.join(o.stagingDir, "x.mp4"); + await mkdir(o.stagingDir, { recursive: true }); + await writeFile(file, BAD); + return { ok: true, file, seeded: false }; + }, + fetchDirect: async (o) => { + directs++; + await writeFile(o.dest, BAD); + }, + }, + }); + assert.equal(directs, 1, "one direct download after the torrent's mismatch, no more"); + assert.equal(rec.status, "failed"); + assert.equal(rec.failureClass, "per_video"); + assert.match(rec.attempts.at(-1)?.error ?? "", /checksum mismatch/); + const dir = path.join(channelDir, "data", rec.videoId); + const entries = await readdir(dir); + assert.ok(!entries.some((e) => e.startsWith("audio."))); + await assert.rejects(stat(path.join(channelDir, "archive"))); +}); + +test("direct only (no torrent): a mismatch is retried once, then fails", async () => { + let directs = 0; + const { rec } = await download({ + url: archiveOrgDetailsUrl({ identifier: AUDIO_ITEM }), + deps: { + client: client({ [metaUrl(AUDIO_ITEM)]: { json: AUDIO_META } }), + settings: { ...DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, torrent: false }, + fetchDirect: async (o) => { + directs++; + await writeFile(o.dest, directs === 1 ? BAD : GOOD); + }, + }, + }); + assert.equal(directs, 2); + assert.equal(rec.status, "ok"); + assert.deepEqual(rec.attempts.map((a) => [a.kind, !!a.error]), [ + ["archiveorg-direct", true], + ["archiveorg-direct", false], + ]); +}); + +test("a verified torrent fetch needs no direct download; the record is the mirror's original", async () => { + const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: VIDEO_FILE }); + const seen: string[] = []; + const { rec, channelDir } = await download({ + url, + deps: { + client: client( + { + [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META }, + [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, ["other.mp4", VIDEO_FILE]) }, + [`https://archive.org/download/${VIDEO_ITEM}/Clip%20One-AbC123xyz_9.info.json`]: { json: MIRROR_INFO }, + }, + seen, + ), + aria2cAvailable: async () => true, + fetchByTorrent: async (o) => { + assert.equal(o.entry.index, 2); + const file = path.join(o.stagingDir, VIDEO_ITEM, VIDEO_FILE); + await mkdir(path.dirname(file), { recursive: true }); + await writeFile(file, GOOD); + return { ok: true, file, seeded: true }; + }, + fetchDirect: async () => assert.fail("no direct download after a verified torrent"), + finalize: async (o) => { + await writeFile(path.join(o.videoDir, "audio.mp3"), "a"); + }, + }, + }); + assert.equal(rec.status, "ok"); + assert.deepEqual(rec.attempts.map((a) => a.kind), ["archiveorg-torrent"]); + const info = JSON.parse(await readFile(path.join(channelDir, "data", rec.videoId, "metadata.info.json"), "utf8")); + assert.equal(info.title, "Clip One (the original)"); + assert.equal(info.upload_date, "20210102"); + assert.equal(info.uploader, "Example Uploader"); + assert.equal(info.webpage_url, url); + assert.equal(info.id, rec.videoId); + const prov = JSON.parse(await readFile(path.join(channelDir, "data", rec.videoId, "archiveorg.json"), "utf8")); + assert.equal(prov.mirror.id, "AbC123xyz_9"); + assert.equal(prov.mirror.from, "info-json"); + // One metadata request, one torrent, one info.json. + assert.equal(seen.length, 3); +}); + +test("an .avi original is fetched as archive.org's mp4 of it; a file not in the torrent goes direct", async () => { + const url = archiveOrgDetailsUrl({ identifier: VIDEO_ITEM, file: "Clip Two-Def456uvw_8.avi" }); + const directs: string[] = []; + const { rec, log } = await download({ + url, + deps: { + client: client({ + [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META }, + [torrentUrl(VIDEO_ITEM)]: { bytes: torrentOf(VIDEO_ITEM, [VIDEO_FILE]) }, + }), + aria2cAvailable: async () => true, + fetchByTorrent: async () => assert.fail("not in the torrent"), + fetchDirect: async (o) => { + directs.push(o.file); + await writeFile(o.dest, GOOD); + }, + finalize: async (o) => { + assert.equal(o.sourceFilename, "source-media.mp4"); + await writeFile(path.join(o.videoDir, "audio.mp3"), "a"); + }, + }, + }); + assert.equal(rec.status, "ok"); + assert.deepEqual(directs, ["Clip Two-Def456uvw_8.mp4"]); + assert.match(log, /does not carry Clip Two-Def456uvw_8\.mp4/); +}); + +test("archive.org refusing the direct download is a rate limit; the batch backs off", async () => { + const { rec } = await download({ + url: archiveOrgDetailsUrl({ identifier: AUDIO_ITEM }), + deps: { + client: client({ [metaUrl(AUDIO_ITEM)]: { json: AUDIO_META } }), + aria2cAvailable: async () => false, + fetchDirect: async () => { + throw new ArchiveOrgRequestError("archive.org did not deliver x after 4 attempts (HTTP 429); stopping", null, true); + }, + }, + }); + assert.equal(rec.status, "failed"); + assert.equal(rec.failureClass, "rate_limit"); +}); + +test("an item URL of an item with several media files is refused, nothing fetched", async () => { + const { rec, log } = await download({ + url: archiveOrgDetailsUrl({ identifier: VIDEO_ITEM }), + deps: { + client: client({ [metaUrl(VIDEO_ITEM)]: { json: VIDEO_META } }), + fetchDirect: async () => assert.fail("nothing to fetch"), + }, + }); + assert.equal(rec.status, "failed"); + assert.equal(rec.failureClass, "per_video"); + assert.match(log, /holds 2 media files/); +}); diff --git a/common/lib/archiveOrg.test.ts b/common/lib/archiveOrg.test.ts @@ -13,6 +13,12 @@ import { type ArchiveOrgItemMetadata, } from "./archiveOrg"; import { archiveOrgVideoId } from "./archiveOrgId"; +import { + archiveOrgFetchFile, + archiveOrgInfoJson, + archiveOrgRecordFile, + parseArchiveOrgLength, +} from "./archiveOrg"; import { platformFromMetadata, summarize } from "./transcripts-server"; // Run with: @@ -267,3 +273,97 @@ test("platformFromMetadata and summarize know an archive.org record", () => { const yt = summarize("c", "AbC123xyz_9", { id: "AbC123xyz_9", extractor_key: "Youtube" }); assert.equal("mediaUrl" in yt, false); }); + +// ─── The record without yt-dlp (controller/archiveOrgDownload.ts) ─── + +const FETCH_ITEM: ArchiveOrgItemMetadata = { + metadata: { + identifier: "example-tapes", + title: "Example Tapes", + creator: "Example Archivist", + uploader: "someone@example.org", + publicdate: "2020-02-03 10:11:12", + description: ["Part one.", "Part two."], + }, + files: [ + { name: "tape1.flac", source: "original", length: "61.5" }, + { name: "tape1.mp3", source: "derivative", original: "tape1.flac", format: "VBR MP3", length: "61.48" }, + { name: "tape1.ogg", source: "derivative", original: "tape1.flac", format: "Ogg Vorbis" }, + { name: "talk.avi", source: "original" }, + { name: "talk.mp4", source: "derivative", original: "talk.avi", format: "h.264" }, + { name: "film.mp4", source: "original" }, + { name: "film.ogv", source: "derivative", original: "film.mp4" }, + ], +}; + +test("the file fetched: a common container as is, else archive.org's mp4/mp3 of it", () => { + assert.equal(archiveOrgFetchFile(FETCH_ITEM, "film.mp4")?.name, "film.mp4"); + assert.equal(archiveOrgFetchFile(FETCH_ITEM, "talk.avi")?.name, "talk.mp4"); + assert.equal(archiveOrgFetchFile(FETCH_ITEM, "tape1.flac")?.name, "tape1.mp3"); + assert.equal(archiveOrgFetchFile(FETCH_ITEM, "missing.mp4"), null); + assert.equal(archiveOrgRecordFile(FETCH_ITEM, "talk.avi"), "talk.avi"); + assert.equal(archiveOrgRecordFile(FETCH_ITEM, undefined), null, "several media files: none is the item"); + assert.equal( + archiveOrgRecordFile({ metadata: { identifier: "x" }, files: [{ name: "only.mp3", source: "original" }] }, undefined), + "only.mp3", + ); +}); + +test("archive.org lengths: seconds or a clock", () => { + assert.equal(parseArchiveOrgLength("123.45"), 123.45); + assert.equal(parseArchiveOrgLength("02:03"), 123); + assert.equal(parseArchiveOrgLength("1:02:03"), 3723); + assert.equal(parseArchiveOrgLength("soon"), undefined); +}); + +test("the synthesised record: ArchiveOrg, the canonical id, the file's page, formats, the date last", () => { + const prov = buildArchiveOrgProvenance({ + ref: { identifier: "example-tapes", file: "tape1.flac" }, + item: FETCH_ITEM, + fetchedAt: "2026-10-05T00:00:00.000Z", + }); + const fetched = archiveOrgFetchFile(FETCH_ITEM, "tape1.flac")!; + const info = archiveOrgInfoJson({ prov, item: FETCH_ITEM, file: "tape1.flac", fetched }); + assert.equal(info.extractor_key, "ArchiveOrg"); + assert.equal(platformFromMetadata(info), "archiveorg"); + assert.equal(info.id, archiveOrgVideoId({ identifier: "example-tapes", file: "tape1.flac" })); + assert.equal(info.webpage_url, "https://archive.org/details/example-tapes/tape1.flac"); + // One file of many: its own name, never the item's title. + assert.equal(info.title, "tape1"); + assert.equal(info.uploader, "Example Archivist"); + assert.equal(info.duration, 61.48); + assert.equal(info.ext, "mp3"); + assert.equal(info.description, "Part one.\n\nPart two."); + assert.equal(info.upload_date, "20200203"); + assert.equal(Object.keys(info).at(-1), "upload_date"); + assert.deepEqual( + (info.formats as { format_id: string; format_note: string }[]).map((f) => [f.format_id, f.format_note]), + [ + ["tape1.flac", "original"], + ["tape1.mp3", "derivative"], + ["tape1.ogg", "derivative"], + ], + ); + // The player plays a playable one of them. + assert.equal(summarize("c", String(info.id), info).mediaUrl, "https://archive.org/download/example-tapes/tape1.mp3"); +}); + +test("a mirror's record is the original's: title, date, uploader from its uploaded info.json", () => { + const item: ArchiveOrgItemMetadata = { + metadata: { identifier: "youtube-AbC123xyz_9", title: "Mirror", date: "2024-01-01" }, + files: [{ name: "Clip.mp4", source: "original" }, { name: "Clip.info.json", source: "original" }], + }; + const prov = buildArchiveOrgProvenance({ + ref: { identifier: "youtube-AbC123xyz_9" }, + item, + infoJson: { id: "AbC123xyz_9", extractor_key: "Youtube", title: "The Original", upload_date: "20190909", uploader: "Orig Channel" }, + fetchedAt: "2026-10-05T00:00:00.000Z", + }); + const info = archiveOrgInfoJson({ prov, item, file: "Clip.mp4", fetched: item.files[0], durationSec: 10 }); + assert.equal(info.id, "youtube-AbC123xyz_9"); + assert.equal(info.title, "The Original"); + assert.equal(info.upload_date, "20190909"); + assert.equal(info.uploader, "Orig Channel"); + assert.equal(info.duration, 10); + assert.equal(info.webpage_url, "https://archive.org/details/youtube-AbC123xyz_9"); +}); diff --git a/common/lib/archiveOrgClient.test.ts b/common/lib/archiveOrgClient.test.ts @@ -1,5 +1,8 @@ import { test } from "node:test"; import assert from "node:assert/strict"; +import { mkdtemp, readFile, rm, writeFile, stat } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; import { ARCHIVE_ORG_USER_AGENT, ArchiveOrgClient, @@ -129,3 +132,106 @@ test("Retry-After as seconds or as an HTTP date", () => { assert.equal(parseRetryAfterMs(null, 0), null); assert.equal(parseRetryAfterMs("soon", 0), null); }); + +// ─── A file's bytes (the fallback when a torrent cannot be used) ─── + +type FileStep = { status: number; body?: string; headers?: Record<string, string>; expectRange?: string | null }; + +function fileHarness(steps: FileStep[]) { + const sleeps: number[] = []; + const ranges: (string | null)[] = []; + const client = new ArchiveOrgClient( + { + sleep: async (ms) => { + sleeps.push(ms); + }, + random: () => 0.5, + fetch: async (url, init) => { + const h = new Headers(init.headers); + ranges.push(h.get("range")); + assert.equal(h.get("user-agent"), ARCHIVE_ORG_USER_AGENT); + assert.equal(url, "https://archive.org/download/example-item/Clip%20One.mp4"); + const s = steps.shift(); + if (!s) throw new Error("script exhausted"); + return new Response(s.body ?? "", { status: s.status, headers: s.headers }); + }, + }, + { minGapMs: 0, baseBackoffMs: 1000 }, + ); + return { client, sleeps, ranges }; +} + +test("downloadFile: a partial on disk is resumed with a Range request and appended", async () => { + const dir = await mkdtemp(path.join(os.tmpdir(), "aoc-dl-")); + try { + const dest = path.join(dir, "Clip One.mp4"); + await writeFile(`${dest}.part`, "hello "); + const { client, ranges } = fileHarness([ + { status: 206, body: "world", headers: { "content-length": "5" } }, + ]); + const r = await client.downloadFile("example-item", "Clip One.mp4", dest, { expectedSize: 11 }); + assert.equal(r.bytes, 11); + assert.deepEqual(ranges, ["bytes=6-"]); + assert.equal(await readFile(dest, "utf8"), "hello world"); + await assert.rejects(stat(`${dest}.part`)); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("downloadFile: a 503 waits its Retry-After; a cut-off stream resumes where it stopped", async () => { + const dir = await mkdtemp(path.join(os.tmpdir(), "aoc-dl-")); + try { + const dest = path.join(dir, "Clip One.mp4"); + const { client, sleeps, ranges } = fileHarness([ + { status: 503, headers: { "retry-after": "7" } }, + // Says 11 bytes, sends 4: the stream was cut. + { status: 200, body: "hell", headers: { "content-length": "11" } }, + { status: 206, body: "o world", headers: { "content-length": "7" } }, + ]); + const r = await client.downloadFile("example-item", "Clip One.mp4", dest); + assert.equal(r.bytes, 11); + assert.deepEqual(ranges, [null, null, "bytes=4-"]); + assert.equal(sleeps[0], 7000); + assert.equal(await readFile(dest, "utf8"), "hello world"); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("downloadFile: a 404 is final; archive.org refusing to the end is a rate limit", async () => { + const dir = await mkdtemp(path.join(os.tmpdir(), "aoc-dl-")); + try { + const dest = path.join(dir, "Clip One.mp4"); + const gone = fileHarness([{ status: 404 }]); + await assert.rejects( + gone.client.downloadFile("example-item", "Clip One.mp4", dest), + (e: unknown) => e instanceof ArchiveOrgRequestError && e.status === 404 && !e.rateLimited, + ); + const busy = fileHarness([{ status: 429 }, { status: 429 }, { status: 429 }, { status: 429 }]); + await assert.rejects( + busy.client.downloadFile("example-item", "Clip One.mp4", dest), + (e: unknown) => e instanceof ArchiveOrgRequestError && e.rateLimited, + ); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("itemTorrent: fetched once, then cached", async () => { + let calls = 0; + const client = new ArchiveOrgClient( + { + sleep: async () => {}, + fetch: async (url) => { + calls++; + assert.equal(url, "https://archive.org/download/example-item/example-item_archive.torrent"); + return new Response(new Uint8Array([100, 101]), { status: 200 }); + }, + }, + { minGapMs: 0 }, + ); + assert.deepEqual([...(await client.itemTorrent("example-item"))], [100, 101]); + await client.itemTorrent("example-item"); + assert.equal(calls, 1); +}); diff --git a/common/lib/archiveOrgTorrent-server.test.ts b/common/lib/archiveOrgTorrent-server.test.ts @@ -0,0 +1,156 @@ +import { test, after } from "node:test"; +import assert from "node:assert/strict"; +import { chmod, mkdtemp, readFile, rm, stat, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { fetchFileByTorrent, aria2cAvailable, __resetAria2cProbeForTest } from "./archiveOrgTorrent-server"; +import { DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, type ParsedTorrent } from "./archiveOrgTorrent"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrgTorrent-server.test.ts +// +// One aria2c run against a FAKE aria2c (a node script below) that prints what +// aria2c 1.37 prints and does what the mode in FAKE_ARIA2C_MODE says: complete +// (write the file, run the --on-bt-download-complete hook, seed, exit 0), +// stall (the same progress forever), fail (an error, exit 1), hang (progress, +// a child process of its own, until killed). Nothing touches the network. + +const ROOT = await mkdtemp(path.join(os.tmpdir(), "aria2c-run-")); +after(() => rm(ROOT, { recursive: true, force: true })); + +const FAKE = path.join(ROOT, "fake-aria2c.mjs"); +await writeFile( + FAKE, + `#!/usr/bin/env node +import { mkdirSync, writeFileSync } from "node:fs"; +import { spawn, execFileSync } from "node:child_process"; +import path from "node:path"; +const args = process.argv.slice(2); +if (args[0] === "--version") { console.log("aria2 version 1.37.0"); process.exit(0); } +const opt = (k) => (args.find((a) => a.startsWith("--" + k + "=")) ?? "").slice(k.length + 3); +const dir = opt("dir"); +const hook = opt("on-bt-download-complete"); +const mode = process.env.FAKE_ARIA2C_MODE; +const target = path.join(dir, "example-item", "Clip One.mp4"); +const say = (s) => process.stdout.write(s + "\\n"); +const sleep = (ms) => new Promise((r) => setTimeout(r, ms)); +writeFileSync(path.join(process.env.FAKE_ARIA2C_LOG, "args.json"), JSON.stringify(args)); +say("10/05 15:46:34 [NOTICE] Downloading 1 item(s)"); +if (mode === "fail") { say("10/05 15:46:35 [ERROR] CUID#7 - Download aborted. URI=x"); process.exit(1); } +if (mode === "stall") { for (;;) { say("[#2089b0 16KiB/40KiB(40%) CN:1 SD:0 DL:0B]"); await sleep(50); } } +if (mode === "hang") { + const c = spawn("sleep", ["30"], { stdio: "ignore" }); + writeFileSync(path.join(process.env.FAKE_ARIA2C_LOG, "pids.json"), JSON.stringify([process.pid, c.pid])); + let n = 0; + for (;;) { n++; say("[#2089b0 " + n + "KiB/40KiB(" + n + "%) CN:1 SD:0 DL:1KiB]"); await sleep(50); } +} +// complete +say(" *** Download Progress Summary as of Mon Oct 5 15:46:35 2026 *** "); +say("[#2089b0 16KiB/40KiB(40%) CN:3 SD:1 DL:16KiB ETA:1s]"); +say("FILE: " + target); +mkdirSync(path.dirname(target), { recursive: true }); +writeFileSync(target, "x".repeat(40000)); +execFileSync(hook, ["2089b0", "1", target]); +say("[#2089b0 SEED(0.0) CN:1 SD:0 UL:0B(0B)]"); +await sleep(150); +say("10/05 15:46:39 [NOTICE] Seeding is over."); +process.exit(0); +`, +); +await chmod(FAKE, 0o755); +process.env.FAKE_ARIA2C_LOG = ROOT; + +const PARSED: ParsedTorrent = { + name: "example-item", + pieceLength: 16384, + pieceCount: 3, + multiFile: true, + files: [{ index: 2, path: "Clip One.mp4", length: 40000, offset: 1000 }], + webSeeds: ["https://archive.org/download/"], + trackers: [], +}; + +function run(mode: string, extra: { signal?: AbortSignal; stallMs?: number } = {}) { + process.env.FAKE_ARIA2C_MODE = mode; + const lines: string[] = []; + const stagingDir = path.join(ROOT, `staging-${mode}`); + const p = fetchFileByTorrent({ + aria2cBin: FAKE, + torrent: Buffer.from("d4:infod4:name1:xee"), + parsed: PARSED, + entry: PARSED.files[0], + stagingDir, + settings: { ...DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, seedMinutes: 1 }, + userAgent: "test", + onLog: (l) => lines.push(l), + signal: extra.signal ?? new AbortController().signal, + tickMs: 20, + progressLogMs: 0, + killGraceMs: 300, + ...(extra.stallMs ? { stallMs: extra.stallMs } : {}), + }); + return { p, lines, stagingDir }; +} + +function alive(pid: number): boolean { + try { + process.kill(pid, 0); + return true; + } catch { + return false; + } +} + +test("aria2cAvailable: present is true, a missing binary is false", async () => { + __resetAria2cProbeForTest(); + assert.equal(await aria2cAvailable(FAKE), true); + assert.equal(await aria2cAvailable(path.join(ROOT, "no-such-aria2c")), false); +}); + +test("complete: the file, the hook, seeding, a human progress line", async () => { + const { p, lines, stagingDir } = run("complete"); + const res = await p; + assert.equal(res.ok, true); + assert.equal(res.ok && res.file, path.join(stagingDir, "example-item", "Clip One.mp4")); + assert.equal(res.ok && res.seeded, true); + assert.equal((await stat(path.join(stagingDir, "example-item", "Clip One.mp4"))).size, 40000); + assert.ok(lines.some((l) => l === "torrent: Clip One.mp4 (1 of 3 pieces, peers 3 (1 seeding), web seed yes)\n"), lines.join("")); + assert.ok(lines.some((l) => /seeding 1 min or to ratio 1, whichever comes first/.test(l))); + assert.ok(lines.some((l) => /aria2c: \[NOTICE\] Seeding is over\./.test(l))); + const args = JSON.parse(await readFile(path.join(ROOT, "args.json"), "utf8")) as string[]; + assert.ok(args.includes("--select-file=2")); + assert.ok(args.includes(`--stop-with-process=${process.pid}`)); +}); + +test("stall: no progress for the stall time stops aria2c, flagged stalled", async () => { + const { p } = run("stall", { stallMs: 300 }); + const res = await p; + assert.equal(res.ok, false); + assert.equal(!res.ok && res.stalled, true); + assert.match(!res.ok ? res.reason : "", /stalled: no progress/); +}); + +test("fail: aria2c's error is the reason", async () => { + const res = await run("fail").p; + assert.equal(res.ok, false); + assert.match(!res.ok ? res.reason : "", /aria2c exited 1: .*Download aborted/); +}); + +test("cancel: the whole process group is killed, the result says cancelled", async () => { + const ctl = new AbortController(); + const { p } = run("hang", { signal: ctl.signal }); + // Let it start and record its pids. + let pids: number[] = []; + for (let i = 0; i < 100 && pids.length === 0; i++) { + await new Promise((r) => setTimeout(r, 20)); + pids = await readFile(path.join(ROOT, "pids.json"), "utf8").then((t) => JSON.parse(t), () => []); + } + assert.equal(pids.length, 2); + ctl.abort(); + const res = await p; + assert.equal(res.ok, false); + assert.equal(!res.ok && res.cancelled, true); + await new Promise((r) => setTimeout(r, 100)); + assert.equal(alive(pids[0]), false, "aria2c is gone"); + assert.equal(alive(pids[1]), false, "its child is gone with it"); +}); diff --git a/common/lib/archiveOrgTorrent-server.ts b/common/lib/archiveOrgTorrent-server.ts @@ -83,6 +83,8 @@ export type TorrentFetchOpts = { killGraceMs?: number; // How long past --seed-time a seeding aria2c may run before it is stopped. seedGraceMs?: number; + // Overrides settings.stallMinutes (tests). + stallMs?: number; now?: () => number; }; @@ -116,7 +118,7 @@ export async function fetchFileByTorrent(opts: TorrentFetchOpts): Promise<Torren const tickMs = opts.tickMs ?? 2_000; const progressLogMs = opts.progressLogMs ?? 30_000; const killGraceMs = opts.killGraceMs ?? 10_000; - const stallMs = Math.max(1, opts.settings.stallMinutes) * 60_000; + const stallMs = opts.stallMs ?? Math.max(1, opts.settings.stallMinutes) * 60_000; const seedMs = Math.max(0, opts.settings.seedMinutes) * 60_000; const seedGraceMs = opts.seedGraceMs ?? 5 * 60_000; const log = (s: string) => opts.onLog(s.endsWith("\n") ? s : `${s}\n`); diff --git a/common/lib/archiveOrgTorrent.test.ts b/common/lib/archiveOrgTorrent.test.ts @@ -0,0 +1,190 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { bdecode, BencodeError } from "./bencode"; +import { + DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, + aria2cArgs, + findTorrentFile, + parseAria2cSize, + parseAria2cStatusLine, + parseTorrent, + torrentFilePieces, + torrentFileSegments, + torrentProgressLine, +} from "./archiveOrgTorrent"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrgTorrent.test.ts +// +// A synthetic torrent, built here by a ten-line encoder, shaped like +// archive.org's `<identifier>_archive.torrent`: name = identifier, a files +// list, archive.org web seeds. Every name is invented. + +type Enc = number | string | Buffer | Enc[] | { [k: string]: Enc }; +function benc(v: Enc): Buffer { + if (typeof v === "number") return Buffer.from(`i${v}e`); + if (typeof v === "string") return benc(Buffer.from(v, "utf8")); + if (Buffer.isBuffer(v)) return Buffer.concat([Buffer.from(`${v.length}:`), v]); + if (Array.isArray(v)) return Buffer.concat([Buffer.from("l"), ...v.map(benc), Buffer.from("e")]); + const keys = Object.keys(v).sort(); + return Buffer.concat([Buffer.from("d"), ...keys.flatMap((k) => [benc(k), benc(v[k])]), Buffer.from("e")]); +} + +const PIECE = 16384; +const FILES = [ + { path: ["example-item_meta.xml"], length: 1000 }, + { path: ["Clip One.mp4"], length: 40000 }, + { path: ["sub dir", "Clip Two [AbC123xyz_9].mp4"], length: 50000 }, + { path: ["Clip One.info.json"], length: 2000 }, +]; +const total = FILES.reduce((n, f) => n + f.length, 0); +const TORRENT = benc({ + announce: "http://bt1.archive.org:6969/announce", + "announce-list": [["http://bt1.archive.org:6969/announce"], ["http://bt2.archive.org:6969/announce"]], + "url-list": ["https://archive.org/download/", "http://ia000000.us.archive.org/0/items/"], + info: { + name: "example-item", + "piece length": PIECE, + pieces: Buffer.alloc(20 * Math.ceil(total / PIECE)), + files: FILES, + }, +}); + +test("bencode: integers, strings, lists, dictionaries", () => { + assert.equal(bdecode(Buffer.from("i-42e")), -42); + assert.deepEqual(bdecode(Buffer.from("4:spam")), Buffer.from("spam")); + const v = bdecode(Buffer.from("d3:bari1e3:fool1:a1:bee")) as Record<string, unknown>; + assert.equal(v.bar, 1); + assert.deepEqual((v.foo as Buffer[]).map((b) => b.toString()), ["a", "b"]); +}); + +test("bencode: malformed input throws, naming the offset", () => { + for (const bad of ["i12", "5:abc", "l1:a", "d1:ai1e", "x", "i1ei2e", "i1x2e"]) { + assert.throws(() => bdecode(Buffer.from(bad)), BencodeError, bad); + } +}); + +test("a torrent: name, pieces, files with 1-based indexes and offsets, web seeds, trackers", () => { + const t = parseTorrent(TORRENT)!; + assert.equal(t.name, "example-item"); + assert.equal(t.pieceLength, PIECE); + assert.equal(t.pieceCount, Math.ceil(total / PIECE)); + assert.equal(t.multiFile, true); + assert.deepEqual( + t.files.map((f) => [f.index, f.path, f.offset]), + [ + [1, "example-item_meta.xml", 0], + [2, "Clip One.mp4", 1000], + [3, "sub dir/Clip Two [AbC123xyz_9].mp4", 41000], + [4, "Clip One.info.json", 91000], + ], + ); + assert.equal(t.webSeeds.length, 2); + assert.deepEqual(t.trackers, ["http://bt1.archive.org:6969/announce", "http://bt2.archive.org:6969/announce"]); +}); + +test("the file's index is found by its exact path in the item; absent is null", () => { + const t = parseTorrent(TORRENT)!; + assert.equal(findTorrentFile(t, "sub dir/Clip Two [AbC123xyz_9].mp4")?.index, 3); + assert.equal(findTorrentFile(t, "Clip One.mp4")?.index, 2); + assert.equal(findTorrentFile(t, "clip one.mp4"), null); + assert.equal(findTorrentFile(t, "Added Later.mp4"), null); + const two = findTorrentFile(t, "sub dir/Clip Two [AbC123xyz_9].mp4")!; + assert.deepEqual(torrentFileSegments(t, two), ["example-item", "sub dir", "Clip Two [AbC123xyz_9].mp4"]); + // 41000..90999 spans pieces 2..5 of 16 KiB. + assert.deepEqual(torrentFilePieces(t, two), { first: 2, last: 5, count: 4 }); +}); + +test("path.utf-8 wins over path; not a torrent is null", () => { + const t = parseTorrent( + benc({ + info: { + name: "x", + "piece length": PIECE, + pieces: Buffer.alloc(20), + files: [{ path: ["mangled"], "path.utf-8": ["Café.mp3"], length: 10 }], + }, + }), + )!; + assert.equal(t.files[0].path, "Café.mp3"); + assert.equal(parseTorrent(Buffer.from("not bencode")), null); + assert.equal(parseTorrent(benc({ announce: "x" })), null); +}); + +test("a single-file torrent writes <dir>/<name>", () => { + const t = parseTorrent(benc({ info: { name: "solo.mp3", "piece length": PIECE, pieces: Buffer.alloc(20), length: 99 } }))!; + assert.equal(t.multiFile, false); + assert.deepEqual(torrentFileSegments(t, t.files[0]), ["solo.mp3"]); +}); + +test("aria2c arguments: one file, seeding bounds, politeness, the hook, no word-splitting", () => { + const args = aria2cArgs({ + torrentPath: "/s/item.torrent", + dir: "/s/dir with space", + fileIndex: 3, + settings: { ...DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, maxUploadKiBps: 200 }, + onCompleteHook: "/s/on-complete.sh", + userAgent: "Example UA (+https://example.org)", + parentPid: 1234, + }); + assert.ok(args.includes("--dir=/s/dir with space")); + assert.ok(args.includes("--select-file=3")); + assert.ok(args.includes("--seed-time=10")); + assert.ok(args.includes("--seed-ratio=1.0")); + assert.ok(args.includes("--bt-max-peers=30")); + assert.ok(args.includes("--max-overall-download-limit=0")); + assert.ok(args.includes("--max-overall-upload-limit=200K")); + assert.ok(args.includes("--bt-remove-unselected-file=true")); + assert.ok(args.includes("--follow-torrent=mem")); + assert.ok(args.includes("--file-allocation=none")); + assert.ok(args.includes("--max-connection-per-server=1")); + assert.ok(args.includes("--on-bt-download-complete=/s/on-complete.sh")); + assert.ok(args.includes("--user-agent=Example UA (+https://example.org)")); + assert.ok(args.includes("--stop-with-process=1234")); + assert.equal(args.at(-1), "--torrent-file=/s/item.torrent"); + assert.ok(args.every((a) => a.startsWith("--"))); + // No seeding when the operator says so. + const none = aria2cArgs({ + torrentPath: "t", + dir: "d", + fileIndex: 1, + settings: { ...DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, seedMinutes: 0, seedRatio: 0 }, + onCompleteHook: "h", + userAgent: "u", + }); + assert.ok(none.includes("--seed-time=0")); + assert.ok(!none.some((a) => a.startsWith("--stop-with-process"))); +}); + +test("aria2c progress lines, as 1.37 prints them", () => { + assert.equal(parseAria2cSize("16KiB"), 16384); + assert.equal(parseAria2cSize("2.8MiB"), Math.round(2.8 * 1024 * 1024)); + assert.equal(parseAria2cSize("0B"), 0); + assert.equal(parseAria2cSize("lots"), null); + assert.deepEqual(parseAria2cStatusLine("[#09cba8 608KiB/2.8MiB(20%) CN:1 DL:302KiB ETA:7s]"), { + seeding: false, + completedBytes: 608 * 1024, + totalBytes: Math.round(2.8 * 1024 * 1024), + percent: 20, + connections: 1, + }); + const bt = parseAria2cStatusLine("[#2089b0 1.1MiB/33MiB(3%) CN:5 SD:2 DL:115KiB UL:4KiB(16KiB) ETA:4m48s]"); + assert.equal(bt?.seeding, false); + assert.equal(bt && !bt.seeding ? bt.seeders : -1, 2); + assert.equal(bt?.connections, 5); + const seed = parseAria2cStatusLine("[#2089b0 SEED(0.4) CN:2 SD:0 UL:12KiB(4.9MiB)]"); + assert.equal(seed?.seeding, true); + assert.equal(seed && seed.seeding ? seed.ratio : -1, 0.4); + assert.equal(parseAria2cStatusLine("FILE: out/blob.bin"), null); + assert.equal(parseAria2cStatusLine("10/05 15:46:39 [NOTICE] Download complete: out/blob.bin"), null); +}); + +test("the human line: pieces of the file's own span, peers, web seed", () => { + const line = torrentProgressLine({ + file: "Clip One.mp4", + status: { seeding: false, completedBytes: 50, totalBytes: 100, percent: 50, connections: 4, seeders: 1 }, + pieces: 9, + webSeed: true, + }); + assert.equal(line, "torrent: Clip One.mp4 (4 of 9 pieces, peers 4 (1 seeding), web seed yes)"); +}); diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts @@ -64,7 +64,10 @@ import { type MetadataScanEntry, } from "../controller/metadataScanStore"; import { detectPlatform, type Platform } from "../lib/platform"; -import { downloadArchiveOrgManaged } from "../controller/archiveOrgDownload"; +import { + downloadArchiveOrgManaged, + type ArchiveOrgDownloadDeps, +} from "../controller/archiveOrgDownload"; import { finalizeAppExtraction } from "./finalizeAppExtraction"; import { probeMediaDurationSec } from "./ffprobeDuration"; import { @@ -186,6 +189,9 @@ export type ManagedDownloadOpts = { // "video_720" = the ≤720p H.264 selector (downloadFormat.ts). Never touches // the audio-only selector above. persistFormatPreset?: SourceVideoQuality; + // Test seam: what the archive.org download (controller/archiveOrgDownload.ts) + // talks to. Every production caller passes none. + archiveOrgDeps?: ArchiveOrgDownloadDeps; }; // When `reuseInfoJson` is true, the real download reuses the metadata the @@ -648,7 +654,7 @@ export async function downloadOneManaged( // checksums, and the record is built from the item's metadata // (controller/archiveOrgDownload.ts). Same log, same outcome sidecar. if (detectPlatform(opts.videoUrl) === "archiveorg") { - return await downloadArchiveOrgManaged(opts); + return await downloadArchiveOrgManaged(opts, opts.archiveOrgDeps); } return await runManagedDownload(opts, channelDir, startedAt, canonicalId); } finally {