commit 0580ebcf23bb5785601785a447db246361b6e61c
parent a38ef94914b2d5ecfc92a5799231fcc96dd28a52
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 15:05:53 -0400
sources: archive.org items are a platform (archiveorg) — ids, detection, provenance sidecar, polite client, import controller
- archiveorg in Platform/PLATFORM_VALUES and the one host table (archive.org,
www.archive.org; not web.archive.org)
- canonical ids: a whole item is its identifier; one file of an item is
<identifier>__<slug>-<fnv32 of the path> (lib/archiveOrgId.ts)
- platformFromMetadata maps yt-dlp's ArchiveOrg extractor; summarize gives the
canonical id and a playable mediaUrl
- PLATFORM_ARGS.archiveorg: --sleep-requests 2, exponential --retry-sleep, and
--parse-metadata keeping webpage_url the fetched file page
- "auto" format on archive.org: the original mp4/mkv/webm, an audio item's mp3
- archiveorg.json sidecar (identifier, file, item fields, torrent, the YouTube
original of a mirror) written by the managed download after the prefetch;
metadata.info.json corrected through patchMetadataInfo
(by: archiveorg-provenance)
- ArchiveOrgClient: one request at a time, 2 s gap, identified, Retry-After +
exponential backoff, stops after 4 attempts, item metadata cached
- controller/archiveOrgImport.ts: one-URL resolution (multi-file items refused),
bulk import of chosen files, jittered gap >= 8 s, skip on disk, stop on a
rate limit or 3 failures in a row
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
23 files changed, 2312 insertions(+), 11 deletions(-)
diff --git a/common/controller/archiveOrgImport.test.ts b/common/controller/archiveOrgImport.test.ts
@@ -0,0 +1,222 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import type { DownloadOutcomeRecord } from "../lib/downloadOutcome";
+import type { ChannelConfig } from "../lib/channelConfig";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/archiveOrgImport.test.ts
+//
+// The bulk import with everything injected: a scripted archive.org, a fake
+// per-file download, a sleeper that only records. getPaths()/getSettings()
+// memoize, so the env is set before anything imports them. Every name here is
+// invented.
+
+const ROOT = mkdtempSync(path.join(os.tmpdir(), "archiveorg-import-"));
+process.env.TRANSCRIPTS_DIR = ROOT;
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+writeFileSync(
+ process.env.SETTINGS_FILE,
+ JSON.stringify({ minFreeDiskGB: 0, sleepBetweenDownloadsSeconds: 3 }) + "\n",
+);
+mkdirSync(path.join(ROOT, "channels", "c", "data"), { recursive: true });
+
+const {
+ ARCHIVE_ORG_MIN_GAP_SECONDS,
+ archiveOrgGapMs,
+ resolveArchiveOrgImportUrl,
+ runArchiveOrgImport,
+} = await import("./archiveOrgImport");
+const { ArchiveOrgClient } = await import("../lib/archiveOrgClient");
+const { getPaths } = await import("../lib/paths");
+const { archiveOrgVideoId } = await import("../lib/archiveOrgId");
+
+const ITEM = "example-item";
+const FILES = ["Alpha-AbC123xyz_9.mp4", "Beta-Def456uvw_8.mp4", "Gamma-Ghi789rst_7.mp4"];
+
+function client(meta: unknown) {
+ return new ArchiveOrgClient(
+ {
+ sleep: async () => {},
+ fetch: async () => new Response(JSON.stringify(meta), { status: 200 }),
+ },
+ { minGapMs: 0 },
+ );
+}
+
+const MULTI = {
+ metadata: { identifier: ITEM, title: "Example Archive" },
+ files: [
+ ...FILES.map((name) => ({ name, source: "original" })),
+ { name: "Alpha-AbC123xyz_9.info.json", source: "original" },
+ { name: "Alpha-AbC123xyz_9.ogv", source: "derivative", original: FILES[0] },
+ ],
+};
+const SINGLE = {
+ metadata: { identifier: "example-film", title: "Example Film" },
+ files: [{ name: "film.mp4", source: "original" }, { name: "film.ogv", source: "derivative" }],
+};
+
+const CONFIG: ChannelConfig = { handling: "transcribe", platform: "archiveorg" } as ChannelConfig;
+
+function outcome(status: DownloadOutcomeRecord["status"], failureClass?: DownloadOutcomeRecord["failureClass"]): DownloadOutcomeRecord {
+ return {
+ videoId: "x",
+ status,
+ startedAt: "2026-01-01T00:00:00.000Z",
+ finishedAt: "2026-01-01T00:00:01.000Z",
+ attempts: [],
+ ...(failureClass ? { failureClass } : {}),
+ } as DownloadOutcomeRecord;
+}
+
+test("one URL: an item with several media files is refused, naming the way to choose", async () => {
+ const r = await resolveArchiveOrgImportUrl(`https://archive.org/details/${ITEM}`, { client: client(MULTI) });
+ assert.equal(r.ok, false);
+ assert.match(!r.ok ? r.error : "", /holds 3 media files.*import-archive-org/s);
+});
+
+test("one URL: a file of many is that file; the only file of an item is the item", async () => {
+ const r = await resolveArchiveOrgImportUrl(
+ `https://archive.org/details/${ITEM}/${encodeURIComponent(FILES[1])}`,
+ { client: client(MULTI) },
+ );
+ assert.ok(r.ok);
+ assert.equal(r.ok && r.id, archiveOrgVideoId({ identifier: ITEM, file: FILES[1] }));
+ assert.equal(r.ok && r.url, `https://archive.org/details/${ITEM}/Beta-Def456uvw_8.mp4`);
+
+ const whole = await resolveArchiveOrgImportUrl("https://archive.org/embed/example-film", { client: client(SINGLE) });
+ assert.deepEqual(whole, { ok: true, url: "https://archive.org/details/example-film", id: "example-film", identifier: "example-film" });
+ const byFile = await resolveArchiveOrgImportUrl("https://archive.org/details/example-film/film.mp4", { client: client(SINGLE) });
+ assert.equal(byFile.ok && byFile.id, "example-film");
+
+ const missing = await resolveArchiveOrgImportUrl(`https://archive.org/details/${ITEM}/nope.mp4`, { client: client(MULTI) });
+ assert.equal(missing.ok, false);
+});
+
+test("the gap is the configured pause, floored, plus up to half again", () => {
+ assert.equal(archiveOrgGapMs(0, 0), ARCHIVE_ORG_MIN_GAP_SECONDS * 1000);
+ assert.equal(archiveOrgGapMs(3, 0), ARCHIVE_ORG_MIN_GAP_SECONDS * 1000);
+ assert.equal(archiveOrgGapMs(30, 0), 30_000);
+ assert.equal(archiveOrgGapMs(30, 1), 45_000);
+});
+
+test("bulk: chosen files one at a time, paced, skipping what is on disk", async () => {
+ const paths = getPaths();
+ // Beta is already downloaded (a transcript on disk).
+ const betaId = archiveOrgVideoId({ identifier: ITEM, file: FILES[1] });
+ mkdirSync(path.join(ROOT, "channels", "c", "data", betaId), { recursive: true });
+ writeFileSync(path.join(ROOT, "channels", "c", "data", betaId, "transcript.json"), "{}");
+
+ const urls: string[] = [];
+ const sleeps: number[] = [];
+ const imported: string[] = [];
+ const result = await runArchiveOrgImport({
+ paths,
+ slug: "c",
+ channelConfig: CONFIG,
+ identifier: ITEM,
+ selection: { match: "\\.mp4$" },
+ onLog: () => {},
+ signal: new AbortController().signal,
+ deps: {
+ client: client(MULTI),
+ downloadOne: async (o) => {
+ urls.push(o.videoUrl);
+ return outcome("ok");
+ },
+ sleep: async (ms) => {
+ sleeps.push(ms);
+ },
+ random: () => 0,
+ onImported: (id) => imported.push(id),
+ },
+ });
+ assert.deepEqual(urls, [
+ `https://archive.org/details/${ITEM}/Alpha-AbC123xyz_9.mp4`,
+ `https://archive.org/details/${ITEM}/Gamma-Ghi789rst_7.mp4`,
+ ]);
+ // One gap, between the two fetches — none before the first, none for the skip.
+ assert.deepEqual(sleeps, [ARCHIVE_ORG_MIN_GAP_SECONDS * 1000]);
+ assert.deepEqual(result.imported, [FILES[0], FILES[2]]);
+ assert.deepEqual(result.skipped, [FILES[1]]);
+ assert.equal(imported.length, 2);
+ assert.equal(result.stopped, undefined);
+});
+
+test("bulk: a rate-limited file stops the batch at once", async () => {
+ let n = 0;
+ const result = await runArchiveOrgImport({
+ paths: getPaths(),
+ slug: "c",
+ channelConfig: CONFIG,
+ identifier: ITEM,
+ selection: { files: [FILES[0], FILES[2]] },
+ onLog: () => {},
+ signal: new AbortController().signal,
+ deps: {
+ client: client(MULTI),
+ downloadOne: async () => {
+ n++;
+ return outcome("failed", "rate_limit");
+ },
+ sleep: async () => {},
+ },
+ });
+ assert.equal(n, 1);
+ assert.match(result.stopped ?? "", /rate-limited/);
+});
+
+test("bulk: three failures in a row stop it; unknown names are reported", async () => {
+ const many = {
+ metadata: { identifier: ITEM },
+ files: ["a", "b", "c", "d", "e"].map((x) => ({ name: `${x}.mp3`, source: "original" })),
+ };
+ let n = 0;
+ const result = await runArchiveOrgImport({
+ paths: getPaths(),
+ slug: "c",
+ channelConfig: CONFIG,
+ identifier: ITEM,
+ selection: { files: ["a.mp3", "b.mp3", "c.mp3", "d.mp3", "zz.mp3"] },
+ onLog: () => {},
+ signal: new AbortController().signal,
+ deps: {
+ client: client(many),
+ downloadOne: async () => {
+ n++;
+ throw new Error("boom");
+ },
+ sleep: async () => {},
+ },
+ });
+ assert.equal(n, 3);
+ assert.equal(result.failed.length, 3);
+ assert.deepEqual(result.unknown, ["zz.mp3"]);
+ assert.match(result.stopped ?? "", /3 failures in a row/);
+});
+
+test("bulk: a dry run fetches nothing", async () => {
+ let n = 0;
+ const result = await runArchiveOrgImport({
+ paths: getPaths(),
+ slug: "c",
+ channelConfig: CONFIG,
+ identifier: ITEM,
+ selection: { match: "." },
+ onLog: () => {},
+ signal: new AbortController().signal,
+ dryRun: true,
+ deps: {
+ client: client(MULTI),
+ downloadOne: async () => {
+ n++;
+ return outcome("ok");
+ },
+ },
+ });
+ assert.equal(n, 0);
+ assert.equal(result.planned, 3);
+});
diff --git a/common/controller/archiveOrgImport.ts b/common/controller/archiveOrgImport.ts
@@ -0,0 +1,354 @@
+// IMPORTING FROM archive.org — one item, one file of an item, or a chosen set
+// of files of one item, into an existing channel.
+//
+// Every download is `downloadOneManaged` (the same managed path every other
+// import takes), fetched by the CANONICAL page of what is imported
+// (lib/archiveOrgId.ts): `https://archive.org/details/<identifier>` for an item
+// holding one media file, `…/details/<identifier>/<file>` for one file of
+// many. yt-dlp's ArchiveOrg extractor resolves either to exactly one record,
+// and the provenance step (lib/archiveOrg-server.ts) runs inside it.
+//
+// AN ITEM WITH SEVERAL MEDIA FILES IS NEVER IMPORTED WHOLE. yt-dlp would treat
+// it as a playlist and write every file into the one pinned `data/<id>/`; the
+// import refuses it and names the way to choose files instead.
+//
+// POLITE (the operator: "be polite to archive.org"):
+// - the item's metadata is asked for once (lib/archiveOrgClient.ts caches it)
+// and before any download, so a typo'd identifier costs one request;
+// - files go one at a time, on the `platform:archiveorg` queue, with a
+// jittered gap between them — the channel's (else the global)
+// `sleepBetweenDownloadsSeconds`, never under ARCHIVE_ORG_MIN_GAP_SECONDS,
+// plus up to half again at random;
+// - a file already downloaded is never fetched again;
+// - a rate-limited download stops the batch at once (the platform's own
+// backoff takes over), and ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES failures in
+// a row stop it too. Re-running the same command resumes: what landed is
+// skipped.
+
+import path from "node:path";
+import type { ChannelConfig } from "../lib/channelConfig";
+import type { Paths } from "../lib/paths";
+import type { DownloadOutcomeRecord } from "../lib/downloadOutcome";
+import {
+ listArchiveOrgMediaFiles,
+ pickArchiveOrgFiles,
+ type ArchiveOrgFileSelection,
+ type ArchiveOrgItemMetadata,
+} from "../lib/archiveOrg";
+import {
+ archiveOrgDetailsUrl,
+ archiveOrgVideoId,
+ parseArchiveOrgUrl,
+} from "../lib/archiveOrgId";
+import { archiveOrgClient, type ArchiveOrgClient } from "../lib/archiveOrgClient";
+import { getSettings } from "../lib/settings";
+import { diskGate } from "../lib/diskSpace";
+import { resolveCookiePolicy } from "../lib/cookiePolicy";
+import { downloadOneManaged, type ManagedDownloadOpts } from "../ytdlp/downloadOneManaged";
+import { destinationExists } from "../ytdlp/runYtdlp";
+import { mergeRosterFile } from "./rosterStore";
+
+export const ARCHIVE_ORG_MIN_GAP_SECONDS = 8;
+export const ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES = 3;
+
+// ─── One URL ───
+
+export type ResolvedArchiveOrgImport =
+ | { ok: true; url: string; id: string; identifier: string; file?: string }
+ | { ok: false; error: string };
+
+// What an archive.org URL imports as. An item URL for an item with ONE media
+// file is the item (id = identifier); a file URL is that file — unless the
+// item holds only that one file, when it is the item too, so the same
+// recording never gets two ids. An item URL for an item with several media
+// files is refused, naming them.
+export async function resolveArchiveOrgImportUrl(
+ url: string,
+ opts: { client?: ArchiveOrgClient; signal?: AbortSignal } = {},
+): Promise<ResolvedArchiveOrgImport> {
+ const ref = parseArchiveOrgUrl(url);
+ if (!ref) {
+ return {
+ ok: false,
+ error: `Not an archive.org item URL: ${url} (expected https://archive.org/details/<identifier>[/<file>])`,
+ };
+ }
+ const client = opts.client ?? archiveOrgClient;
+ let item: ArchiveOrgItemMetadata;
+ try {
+ item = await client.itemMetadata(ref.identifier, opts.signal);
+ } catch (err) {
+ return { ok: false, error: (err as Error).message };
+ }
+ const identifier = item.metadata.identifier || ref.identifier;
+ const media = listArchiveOrgMediaFiles(item).map((f) => f.name);
+ if (ref.file) {
+ if (!item.files.some((f) => f.name === ref.file)) {
+ return { ok: false, error: `archive.org item "${identifier}" has no file "${ref.file}"` };
+ }
+ if (media.length === 1 && media[0] === ref.file) {
+ return { ok: true, url: archiveOrgDetailsUrl({ identifier }), id: identifier, identifier };
+ }
+ const file = ref.file;
+ return {
+ ok: true,
+ url: archiveOrgDetailsUrl({ identifier, file }),
+ id: archiveOrgVideoId({ identifier, file }),
+ identifier,
+ file,
+ };
+ }
+ if (media.length === 0) {
+ return { ok: false, error: `archive.org item "${identifier}" has no media files to import` };
+ }
+ if (media.length > 1) {
+ const shown = media.slice(0, 5).map((n) => `"${n}"`).join(", ");
+ return {
+ ok: false,
+ error:
+ `archive.org item "${identifier}" holds ${media.length} media files (${shown}${media.length > 5 ? ", …" : ""}). ` +
+ `Import one by its file URL (https://archive.org/details/${identifier}/<file>), or several with ` +
+ `pnpm ops import-archive-org --json '{"slug":"<channel>","item":"${identifier}","files":[…]}' (or "match": "<regex>").`,
+ };
+ }
+ return { ok: true, url: archiveOrgDetailsUrl({ identifier }), id: identifier, identifier };
+}
+
+// ─── Many files of one item ───
+
+export type ArchiveOrgImportPlanEntry = {
+ file: string;
+ url: string;
+ id: string;
+ // Already downloaded: skipped, never fetched again.
+ onDisk: boolean;
+};
+
+export type ArchiveOrgImportPlan = {
+ identifier: string;
+ title?: string;
+ mediaFiles: number;
+ entries: ArchiveOrgImportPlanEntry[];
+ // Named in `files` but not a media original of the item.
+ unknown: string[];
+};
+
+export async function planArchiveOrgImport(opts: {
+ identifier: string;
+ selection: ArchiveOrgFileSelection;
+ dataDir: string;
+ handling: ChannelConfig["handling"];
+ client?: ArchiveOrgClient;
+ signal?: AbortSignal;
+}): Promise<ArchiveOrgImportPlan> {
+ const client = opts.client ?? archiveOrgClient;
+ const item = await client.itemMetadata(opts.identifier, opts.signal);
+ const identifier = item.metadata.identifier || opts.identifier;
+ const media = listArchiveOrgMediaFiles(item);
+ const { picked, unknown } = pickArchiveOrgFiles(item, opts.selection);
+ const entries: ArchiveOrgImportPlanEntry[] = [];
+ for (const file of picked) {
+ // An item of one media file imports as the item (resolveArchiveOrgImportUrl).
+ const whole = media.length === 1;
+ const id = whole ? identifier : archiveOrgVideoId({ identifier, file });
+ const url = whole ? archiveOrgDetailsUrl({ identifier }) : archiveOrgDetailsUrl({ identifier, file });
+ entries.push({
+ file,
+ url,
+ id,
+ onDisk: await destinationExists(opts.dataDir, id, opts.handling),
+ });
+ }
+ const title = item.metadata.title;
+ return {
+ identifier,
+ ...(typeof title === "string" ? { title } : {}),
+ mediaFiles: media.length,
+ entries,
+ unknown,
+ };
+}
+
+// The gap before the next file: the configured pause, floored, plus up to
+// half again at random so a batch never settles into a fixed beat.
+export function archiveOrgGapMs(sleepBetweenDownloadsSeconds: number, random: number): number {
+ const base = Math.max(ARCHIVE_ORG_MIN_GAP_SECONDS, sleepBetweenDownloadsSeconds || 0);
+ return Math.round(base * (1 + 0.5 * Math.min(1, Math.max(0, random))) * 1000);
+}
+
+function isOk(rec: DownloadOutcomeRecord): boolean {
+ return rec.status.startsWith("ok");
+}
+
+export type ArchiveOrgImportResult = {
+ identifier: string;
+ planned: number;
+ imported: string[];
+ skipped: string[];
+ failed: { file: string; error: string }[];
+ unknown: string[];
+ // Why the batch ended before its last file, when it did.
+ stopped?: string;
+};
+
+export type ArchiveOrgImportDeps = {
+ client?: ArchiveOrgClient;
+ downloadOne?: (opts: ManagedDownloadOpts) => Promise<DownloadOutcomeRecord>;
+ sleep?: (ms: number, signal: AbortSignal) => Promise<void>;
+ random?: () => number;
+ // After each imported file (the editor revalidates its pages).
+ onImported?: (id: string) => void;
+};
+
+function abortableSleep(ms: number, signal: AbortSignal): Promise<void> {
+ return new Promise((resolve) => {
+ if (signal.aborted) return resolve();
+ const t = setTimeout(done, ms);
+ function done() {
+ clearTimeout(t);
+ signal.removeEventListener("abort", done);
+ resolve();
+ }
+ signal.addEventListener("abort", done, { once: true });
+ });
+}
+
+export async function runArchiveOrgImport(opts: {
+ paths: Paths;
+ slug: string;
+ channelConfig: ChannelConfig;
+ identifier: string;
+ selection: ArchiveOrgFileSelection;
+ onLog: (line: string) => void;
+ signal: AbortSignal;
+ drainSignal?: AbortSignal;
+ dryRun?: boolean;
+ deps?: ArchiveOrgImportDeps;
+}): Promise<ArchiveOrgImportResult> {
+ const deps = opts.deps ?? {};
+ const downloadOne = deps.downloadOne ?? downloadOneManaged;
+ const sleep = deps.sleep ?? abortableSleep;
+ const random = deps.random ?? Math.random;
+ const settings = getSettings();
+ const dataDir = path.join(opts.paths.channelsDir, opts.slug, "data");
+ const log = opts.onLog;
+
+ const plan = await planArchiveOrgImport({
+ identifier: opts.identifier,
+ selection: opts.selection,
+ dataDir,
+ handling: opts.channelConfig.handling,
+ client: deps.client,
+ signal: opts.signal,
+ });
+ const result: ArchiveOrgImportResult = {
+ identifier: plan.identifier,
+ planned: plan.entries.length,
+ imported: [],
+ skipped: [],
+ failed: [],
+ unknown: plan.unknown,
+ };
+ log(
+ `archive.org item ${plan.identifier}${plan.title ? ` ("${plan.title}")` : ""}: ` +
+ `${plan.mediaFiles} media files, ${plan.entries.length} chosen, ` +
+ `${plan.entries.filter((e) => e.onDisk).length} already downloaded.\n`,
+ );
+ if (plan.unknown.length > 0) {
+ log(`Not media files of the item (ignored): ${plan.unknown.map((n) => JSON.stringify(n)).join(", ")}\n`);
+ }
+ if (opts.dryRun) {
+ for (const e of plan.entries) log(` ${e.onDisk ? "on disk " : "would get"} ${e.id} ${e.file}\n`);
+ result.skipped = plan.entries.filter((e) => e.onDisk).map((e) => e.file);
+ return result;
+ }
+
+ const sleepSeconds =
+ opts.channelConfig.sleepBetweenDownloadsSeconds ?? settings.sleepBetweenDownloadsSeconds;
+ let fetched = 0;
+ let consecutiveFailures = 0;
+ for (const entry of plan.entries) {
+ if (opts.signal.aborted) {
+ result.stopped = "cancelled";
+ break;
+ }
+ if (opts.drainSignal?.aborted) {
+ result.stopped = "drained";
+ break;
+ }
+ if (entry.onDisk || (await destinationExists(dataDir, entry.id, opts.channelConfig.handling))) {
+ result.skipped.push(entry.file);
+ continue;
+ }
+ const gate = await diskGate(opts.paths, settings, { dir: dataDir });
+ if (!gate.ok) {
+ result.stopped = gate.message;
+ log(`Stopping: ${gate.message}.\n`);
+ break;
+ }
+ if (fetched > 0) {
+ const gap = archiveOrgGapMs(sleepSeconds, random());
+ log(`Waiting ${(gap / 1000).toFixed(1)}s before the next file (archive.org pacing)...\n`);
+ await sleep(gap, opts.signal);
+ if (opts.signal.aborted) {
+ result.stopped = "cancelled";
+ break;
+ }
+ }
+ fetched++;
+ log(`[${fetched}] ${entry.file} → data/${entry.id}/\n`);
+ let rec: DownloadOutcomeRecord | null = null;
+ let error = "";
+ try {
+ rec = await downloadOne({
+ channelSlug: opts.slug,
+ channelConfig: opts.channelConfig,
+ paths: opts.paths,
+ videoUrl: entry.url,
+ onLog: log,
+ signal: opts.signal,
+ cookiePolicy: resolveCookiePolicy(settings, opts.channelConfig),
+ inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback,
+ globalSkipLiveDownloads: settings.skipLiveDownloads,
+ appendArchive: true,
+ });
+ } catch (err) {
+ error = (err as Error).message;
+ }
+ if (rec && isOk(rec)) {
+ consecutiveFailures = 0;
+ result.imported.push(entry.file);
+ await mergeRosterFile(
+ opts.paths,
+ opts.slug,
+ [{ id: entry.id, url: entry.url }],
+ new Date().toISOString(),
+ "import",
+ ).catch(() => {
+ /* the download succeeded; a roster write failure must not fail it */
+ });
+ deps.onImported?.(entry.id);
+ continue;
+ }
+ consecutiveFailures++;
+ const why = error || rec?.attempts.at(-1)?.error || rec?.status || "failed";
+ result.failed.push({ file: entry.file, error: why });
+ log(` failed: ${why}\n`);
+ if (rec?.failureClass === "rate_limit") {
+ result.stopped = "archive.org rate-limited the download; stopping (re-run later — files on disk are skipped)";
+ log(`${result.stopped}.\n`);
+ break;
+ }
+ if (consecutiveFailures >= ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES) {
+ result.stopped = `${consecutiveFailures} failures in a row; stopping`;
+ log(`${result.stopped}.\n`);
+ break;
+ }
+ }
+ log(
+ `archive.org import of ${plan.identifier}: ${result.imported.length} imported, ` +
+ `${result.skipped.length} already on disk, ${result.failed.length} failed` +
+ `${result.stopped ? ` — stopped: ${result.stopped}` : ""}.\n`,
+ );
+ return result;
+}
diff --git a/common/lib/archiveOrg-server.test.ts b/common/lib/archiveOrg-server.test.ts
@@ -0,0 +1,113 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, readFile, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { ArchiveOrgClient } from "./archiveOrgClient";
+import {
+ ensureArchiveOrgProvenance,
+ loadArchiveOrgProvenance,
+} from "./archiveOrg-server";
+import { loadMetadataHistory } from "./metadataHistory-server";
+import { archiveOrgDetailsUrl } from "./archiveOrgId";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrg-server.test.ts
+//
+// The provenance step on a temp video dir, against a scripted archive.org.
+// Every name here is invented.
+
+const ITEM = "example-item";
+const FILE = "First Upload-AbC123xyz_9.mp4";
+const URL_FILE = archiveOrgDetailsUrl({ identifier: ITEM, file: FILE });
+
+const ITEM_META = {
+ metadata: { identifier: ITEM, title: "Example Channel Archive", creator: "Example Creator", collection: "community" },
+ files: [
+ { name: FILE, source: "original" },
+ { name: "First Upload-AbC123xyz_9.info.json", source: "original" },
+ { name: "Second-Def456uvw_8.mp4", source: "original" },
+ { name: `${ITEM}_archive.torrent`, source: "metadata" },
+ ],
+};
+
+const INFO = {
+ id: "AbC123xyz_9",
+ extractor_key: "Youtube",
+ title: "First Upload (original)",
+ upload_date: "20230405",
+ uploader: "Example Creator",
+};
+
+function fakeClient(answers: Record<string, unknown>, seen: string[]) {
+ return new ArchiveOrgClient(
+ {
+ sleep: async () => {},
+ fetch: async (url) => {
+ seen.push(url);
+ if (!(url in answers)) return new Response("{}", { status: 404 });
+ return new Response(JSON.stringify(answers[url]), { status: 200 });
+ },
+ },
+ { minGapMs: 0 },
+ );
+}
+
+async function videoDir(info: Record<string, unknown>): Promise<string> {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "archiveorg-prov-"));
+ await writeFile(path.join(dir, "metadata.info.json"), JSON.stringify(info));
+ return dir;
+}
+
+test("writes the sidecar once and corrects the record, through the history", async () => {
+ const seen: string[] = [];
+ const client = fakeClient(
+ {
+ [`https://archive.org/metadata/${ITEM}`]: ITEM_META,
+ [`https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.info.json`]: INFO,
+ },
+ seen,
+ );
+ const dir = await videoDir({
+ id: `${ITEM}/${FILE}`,
+ extractor_key: "ArchiveOrg",
+ title: "Example Channel Archive",
+ webpage_url: `https://archive.org/details/${ITEM}`,
+ upload_date: "20240304",
+ });
+ const lines: string[] = [];
+ const prov = await ensureArchiveOrgProvenance({ videoDir: dir, videoUrl: URL_FILE, client, onLog: (l) => lines.push(l) });
+ assert.equal(prov?.mirror?.id, "AbC123xyz_9");
+ assert.deepEqual(await loadArchiveOrgProvenance(dir), prov);
+ const info = JSON.parse(await readFile(path.join(dir, "metadata.info.json"), "utf8"));
+ assert.equal(info.webpage_url, URL_FILE);
+ assert.equal(info.title, "First Upload (original)");
+ assert.equal(info.upload_date, "20230405");
+ assert.equal(info.id, `${ITEM}/${FILE}`);
+ const history = await loadMetadataHistory(dir);
+ assert.equal(history?.entries.at(-1)?.by, "archiveorg-provenance");
+ assert.equal(seen.length, 2);
+
+ // A second run (the next download of the record) asks nothing.
+ await ensureArchiveOrgProvenance({ videoDir: dir, videoUrl: URL_FILE, client });
+ assert.equal(seen.length, 2);
+});
+
+test("archive.org unreachable: no sidecar, but the page is still the file's", async () => {
+ const seen: string[] = [];
+ const client = fakeClient({}, seen);
+ const dir = await videoDir({ id: `${ITEM}/${FILE}`, webpage_url: `https://archive.org/details/${ITEM}`, title: "Item" });
+ const lines: string[] = [];
+ const prov = await ensureArchiveOrgProvenance({ videoDir: dir, videoUrl: URL_FILE, client, onLog: (l) => lines.push(l) });
+ assert.equal(prov, null);
+ assert.equal(await loadArchiveOrgProvenance(dir), null);
+ const info = JSON.parse(await readFile(path.join(dir, "metadata.info.json"), "utf8"));
+ assert.equal(info.webpage_url, URL_FILE);
+ assert.equal(info.title, "Item");
+ assert.ok(lines.some((l) => /provenance not fetched/.test(l)));
+});
+
+test("a URL that is not archive.org is left alone", async () => {
+ const dir = await videoDir({ id: "x", webpage_url: "https://example.com/x" });
+ assert.equal(await ensureArchiveOrgProvenance({ videoDir: dir, videoUrl: "https://example.com/x" }), null);
+});
diff --git a/common/lib/archiveOrg-server.ts b/common/lib/archiveOrg-server.ts
@@ -0,0 +1,136 @@
+// archive.org PROVENANCE ON DISK — the `archiveorg.json` sidecar, and the step
+// every managed download of an archive.org record runs after yt-dlp writes its
+// metadata (ytdlp/downloadOneManaged.ts, right after the prefetch).
+//
+// THE STEP, `ensureArchiveOrgProvenance`:
+//
+// 1. the sidecar: kept when it is already there for this item and file;
+// otherwise built from the item's metadata API (one cached request per
+// item, lib/archiveOrgClient.ts) and, when the item carries one, the
+// `.info.json` a mirroring tool uploaded beside the file (one request).
+// 2. the record: metadata.info.json corrected from it
+// (lib/archiveOrg.ts archiveOrgMetadataPatch) through `patchMetadataInfo`,
+// so the change is in metadata.history.json as `archiveorg-provenance`.
+//
+// Step 2 needs no network and always runs: a sidecar that could not be fetched
+// (archive.org refusing, the operator offline) still leaves the record's
+// `webpage_url` the file's page, which is what keeps its directory its own.
+// A failure is logged and does not fail the download — the next download of
+// the record, or a re-import, fills it in.
+
+import path from "node:path";
+import { readFile } from "node:fs/promises";
+import {
+ ARCHIVE_ORG_PROVENANCE_FILENAME,
+ archiveOrgMetadataPatch,
+ buildArchiveOrgProvenance,
+ coerceArchiveOrgProvenance,
+ findArchiveOrgInfoJson,
+ type ArchiveOrgProvenance,
+} from "./archiveOrg";
+import {
+ archiveOrgDetailsUrl,
+ parseArchiveOrgUrl,
+ type ArchiveOrgRef,
+} from "./archiveOrgId";
+import { archiveOrgClient, type ArchiveOrgClient } from "./archiveOrgClient";
+import { patchMetadataInfo } from "./metadataHistory-server";
+import { sidecar, sidecarField } from "./sidecar-server";
+
+export const archiveOrgProvenanceSidecar = sidecar(
+ ARCHIVE_ORG_PROVENANCE_FILENAME,
+ sidecarField(coerceArchiveOrgProvenance),
+);
+
+export const {
+ load: loadArchiveOrgProvenance,
+ write: writeArchiveOrgProvenance,
+} = archiveOrgProvenanceSidecar;
+
+// Fetch what the sidecar records: the item's metadata (cached) and, for a
+// mirror, its uploaded info.json.
+export async function fetchArchiveOrgProvenance(
+ ref: ArchiveOrgRef,
+ opts: { client?: ArchiveOrgClient; signal?: AbortSignal; now?: () => Date } = {},
+): Promise<ArchiveOrgProvenance> {
+ const client = opts.client ?? archiveOrgClient;
+ const item = await client.itemMetadata(ref.identifier, opts.signal);
+ const infoName = findArchiveOrgInfoJson(item, ref.file);
+ let infoJson: unknown;
+ if (infoName) {
+ try {
+ infoJson = await client.itemJsonFile(item.metadata.identifier, infoName, opts.signal);
+ } catch {
+ // The names still say whether it is a mirror; the info.json only adds
+ // the original's title and date.
+ infoJson = undefined;
+ }
+ }
+ return buildArchiveOrgProvenance({
+ ref,
+ item,
+ infoJson,
+ fetchedAt: (opts.now?.() ?? new Date()).toISOString(),
+ });
+}
+
+async function readInfo(videoDir: string): Promise<Record<string, unknown> | null> {
+ try {
+ const v = JSON.parse(await readFile(path.join(videoDir, "metadata.info.json"), "utf8"));
+ return v && typeof v === "object" && !Array.isArray(v) ? (v as Record<string, unknown>) : null;
+ } catch {
+ return null;
+ }
+}
+
+export type EnsureArchiveOrgProvenanceOpts = {
+ videoDir: string;
+ // The URL the record was fetched by (a details/embed/download URL).
+ videoUrl: string;
+ onLog?: (line: string) => void;
+ signal?: AbortSignal;
+ client?: ArchiveOrgClient;
+ now?: () => Date;
+};
+
+export async function ensureArchiveOrgProvenance(
+ opts: EnsureArchiveOrgProvenanceOpts,
+): Promise<ArchiveOrgProvenance | null> {
+ const log = opts.onLog ?? (() => {});
+ const ref = parseArchiveOrgUrl(opts.videoUrl);
+ if (!ref) return null;
+ let prov = await loadArchiveOrgProvenance(opts.videoDir);
+ if (prov && (prov.identifier !== ref.identifier || (prov.file ?? "") !== (ref.file ?? ""))) {
+ prov = null;
+ }
+ if (!prov) {
+ try {
+ prov = await fetchArchiveOrgProvenance(ref, opts);
+ await writeArchiveOrgProvenance(opts.videoDir, prov);
+ log(
+ `archive.org provenance: ${prov.identifier}${prov.file ? ` / ${prov.file}` : ""}` +
+ (prov.mirror ? ` — mirror of YouTube ${prov.mirror.id} (from ${prov.mirror.from})` : "") +
+ `.\n`,
+ );
+ } catch (err) {
+ log(`archive.org provenance not fetched (${(err as Error).message}); the record keeps yt-dlp's fields.\n`);
+ }
+ }
+ const info = await readInfo(opts.videoDir);
+ if (!info) return prov;
+ // Without a sidecar only the page is corrected — it is the one field the
+ // directory's name depends on.
+ const patch = prov
+ ? archiveOrgMetadataPatch(prov, info)
+ : info.webpage_url === archiveOrgDetailsUrl(ref)
+ ? {}
+ : { webpage_url: archiveOrgDetailsUrl(ref) };
+ if (Object.keys(patch).length > 0) {
+ try {
+ await patchMetadataInfo(opts.videoDir, patch, { by: "archiveorg-provenance", onLog: log });
+ } catch (err) {
+ log(`Could not correct metadata.info.json from the archive.org provenance: ${(err as Error).message}\n`);
+ }
+ }
+ return prov;
+}
diff --git a/common/lib/archiveOrg.test.ts b/common/lib/archiveOrg.test.ts
@@ -0,0 +1,269 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ archiveOrgCitationLinks,
+ archiveOrgMetadataPatch,
+ archiveOrgPlayableUrl,
+ buildArchiveOrgProvenance,
+ coerceArchiveOrgProvenance,
+ findArchiveOrgInfoJson,
+ listArchiveOrgMediaFiles,
+ parseArchiveOrgItemMetadata,
+ pickArchiveOrgFiles,
+ type ArchiveOrgItemMetadata,
+} from "./archiveOrg";
+import { archiveOrgVideoId } from "./archiveOrgId";
+import { platformFromMetadata, summarize } from "./transcripts-server";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrg.test.ts
+//
+// A synthetic channel-archive item: three videos uploaded by a mirroring
+// tool, each with its yt-dlp info.json, plus archive.org's derivatives. Every
+// name, id and date is invented.
+
+const ITEM = "example-item";
+const V1 = "First Upload-AbC123xyz_9.mp4";
+const V2 = "Second Upload-Def456uvw_8.mp4";
+const V3 = "Third Upload-Ghi789rst_7.mkv";
+
+function item(over: Partial<ArchiveOrgItemMetadata["metadata"]> = {}): ArchiveOrgItemMetadata {
+ return parseArchiveOrgItemMetadata({
+ metadata: {
+ identifier: ITEM,
+ title: "Example Channel Archive",
+ date: "2024-01-02",
+ publicdate: "2024-03-04 05:06:07",
+ creator: "Example Creator",
+ uploader: "someone@example.org",
+ collection: ["opensource_movies", "community"],
+ mediatype: "movies",
+ ...over,
+ },
+ files: [
+ { name: V1, source: "original", format: "MPEG4" },
+ { name: "First Upload-AbC123xyz_9.info.json", source: "original", format: "JSON" },
+ { name: V2, source: "original", format: "MPEG4", title: "Second, as titled on archive.org" },
+ { name: "Second Upload-Def456uvw_8.info.json", source: "original", format: "JSON" },
+ { name: V3, source: "original", format: "Matroska" },
+ { name: "Third Upload-Ghi789rst_7.mp4", source: "derivative", format: "h.264", original: V3 },
+ { name: "Third Upload-Ghi789rst_7.ogv", source: "derivative", format: "Ogg Video", original: V3 },
+ { name: `${ITEM}_archive.torrent`, source: "metadata", format: "Archive BitTorrent" },
+ { name: `${ITEM}_meta.xml`, source: "original", format: "Metadata" },
+ ],
+ })!;
+}
+
+const MIRROR_INFO = {
+ id: "AbC123xyz_9",
+ extractor_key: "Youtube",
+ title: "First Upload (original title)",
+ upload_date: "20230405",
+ uploader: "Example Creator",
+ channel_url: "https://www.youtube.com/channel/UCexample",
+ description: "The original description.",
+};
+
+test("the metadata API's answer: an item, or null for an unknown identifier", () => {
+ assert.equal(parseArchiveOrgItemMetadata({}), null);
+ assert.equal(parseArchiveOrgItemMetadata(null), null);
+ assert.equal(item().files.length, 9);
+});
+
+test("media files are the originals with a media extension", () => {
+ assert.deepEqual(listArchiveOrgMediaFiles(item()).map((f) => f.name), [V1, V2, V3]);
+});
+
+test("a bulk import picks by exact names or by a case-insensitive regex", () => {
+ assert.deepEqual(pickArchiveOrgFiles(item(), { files: [V2, "nope.mp4", `${ITEM}_meta.xml`] }), {
+ picked: [V2],
+ unknown: ["nope.mp4", `${ITEM}_meta.xml`],
+ });
+ assert.deepEqual(pickArchiveOrgFiles(item(), { match: "^(first|third)" }).picked, [V1, V3]);
+ assert.deepEqual(pickArchiveOrgFiles(item(), { match: "\\.mkv$" }).picked, [V3]);
+});
+
+test("a file's info.json is its stem's; a single-media item's is its only one", () => {
+ assert.equal(findArchiveOrgInfoJson(item(), V1), "First Upload-AbC123xyz_9.info.json");
+ assert.equal(findArchiveOrgInfoJson(item(), V3), null);
+ assert.equal(findArchiveOrgInfoJson(item(), undefined), null);
+});
+
+test("provenance of a mirror reads the original from its info.json", () => {
+ const prov = buildArchiveOrgProvenance({
+ ref: { identifier: ITEM, file: V1 },
+ item: item(),
+ infoJson: MIRROR_INFO,
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ assert.equal(prov.identifier, ITEM);
+ assert.equal(prov.file, V1);
+ assert.equal(prov.itemUrl, `https://archive.org/details/${ITEM}`);
+ assert.equal(prov.fileUrl, `https://archive.org/details/${ITEM}/First%20Upload-AbC123xyz_9.mp4`);
+ assert.equal(prov.downloadUrl, `https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.mp4`);
+ assert.equal(prov.torrentUrl, `https://archive.org/download/${ITEM}/${ITEM}_archive.torrent`);
+ assert.deepEqual(prov.item, {
+ title: "Example Channel Archive",
+ date: "2024-01-02",
+ publicDate: "2024-03-04 05:06:07",
+ creator: "Example Creator",
+ collections: ["opensource_movies", "community"],
+ mediatype: "movies",
+ });
+ assert.deepEqual(prov.mirror, {
+ platform: "youtube",
+ id: "AbC123xyz_9",
+ url: "https://www.youtube.com/watch?v=AbC123xyz_9",
+ title: "First Upload (original title)",
+ uploadDate: "20230405",
+ uploader: "Example Creator",
+ channelUrl: "https://www.youtube.com/channel/UCexample",
+ description: "The original description.",
+ from: "info-json",
+ });
+ // The uploading account's e-mail is never kept.
+ assert.equal(JSON.stringify(prov).includes("@"), false);
+ assert.deepEqual(coerceArchiveOrgProvenance(JSON.parse(JSON.stringify(prov))), prov);
+ assert.equal(coerceArchiveOrgProvenance({ identifier: ITEM }), null);
+});
+
+test("without an info.json the mirror is known by name only; a plain item is no mirror", () => {
+ const byName = buildArchiveOrgProvenance({
+ ref: { identifier: ITEM, file: V3 },
+ item: item(),
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ assert.deepEqual(byName.mirror, {
+ platform: "youtube",
+ id: "Ghi789rst_7",
+ url: "https://www.youtube.com/watch?v=Ghi789rst_7",
+ from: "file-name",
+ });
+ const byIdent = buildArchiveOrgProvenance({
+ ref: { identifier: "youtube-Jkl012mno_6" },
+ item: item({ identifier: "youtube-Jkl012mno_6" }),
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ assert.equal(byIdent.mirror?.from, "identifier");
+ const plain = buildArchiveOrgProvenance({
+ ref: { identifier: "example-film" },
+ item: item({ identifier: "example-film" }),
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ assert.equal(plain.mirror, undefined);
+ assert.equal(plain.file, undefined);
+ // A non-YouTube info.json is not a YouTube mirror.
+ const other = buildArchiveOrgProvenance({
+ ref: { identifier: "example-film", file: "film.mp4" },
+ item: item({ identifier: "example-film" }),
+ infoJson: { id: "abc", extractor_key: "Generic" },
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ assert.equal(other.mirror, undefined);
+});
+
+test("the record is corrected: the file's page and title, the original's date", () => {
+ const prov = buildArchiveOrgProvenance({
+ ref: { identifier: ITEM, file: V1 },
+ item: item(),
+ infoJson: MIRROR_INFO,
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ const info = {
+ id: `${ITEM}/${V1}`,
+ title: "Example Channel Archive",
+ webpage_url: `https://archive.org/details/${ITEM}`,
+ upload_date: "20240304",
+ timestamp: 1709528767,
+ uploader: "someone@example.org",
+ };
+ const patch = archiveOrgMetadataPatch(prov, info);
+ assert.deepEqual(patch, {
+ webpage_url: prov.fileUrl,
+ title: "First Upload (original title)",
+ upload_date: "20230405",
+ description: "The original description.",
+ uploader: "Example Creator",
+ timestamp: null,
+ });
+ // Applied, nothing is left to change.
+ assert.deepEqual(archiveOrgMetadataPatch(prov, { ...info, ...patch }), {});
+
+ // One file of many, no mirror: the file's own archive.org title.
+ const v2 = buildArchiveOrgProvenance({
+ ref: { identifier: ITEM, file: V2 },
+ item: item({ creator: undefined }),
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ const p2 = archiveOrgMetadataPatch({ ...v2, mirror: undefined }, info);
+ assert.equal(p2.title, "Second, as titled on archive.org");
+ assert.equal(p2.uploader, null);
+ assert.equal("upload_date" in p2, false);
+});
+
+test("citation links: a mirror cites YouTube at the second, archive.org and the torrent as downloads", () => {
+ const prov = buildArchiveOrgProvenance({
+ ref: { identifier: ITEM, file: V1 },
+ item: item(),
+ infoJson: MIRROR_INFO,
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ assert.deepEqual(archiveOrgCitationLinks({ webpageUrl: prov.fileUrl, provenance: prov, seconds: 75.6 }), {
+ original: { label: "YouTube", url: "https://www.youtube.com/watch?v=AbC123xyz_9&t=75s" },
+ downloads: [
+ { label: "archive.org", url: prov.fileUrl },
+ { label: "torrent", url: `https://archive.org/download/${ITEM}/${ITEM}_archive.torrent` },
+ ],
+ });
+});
+
+test("citation links: a plain record is derived from its page alone", () => {
+ assert.deepEqual(archiveOrgCitationLinks({ webpageUrl: "https://archive.org/details/example-film", seconds: 30 }), {
+ original: { label: "archive.org", url: "https://archive.org/details/example-film" },
+ downloads: [{ label: "torrent", url: "https://archive.org/download/example-film/example-film_archive.torrent" }],
+ });
+ assert.deepEqual(archiveOrgCitationLinks({ webpageUrl: "https://example.com/x" }), { downloads: [] });
+});
+
+test("the player plays a browser-playable file: an mp4 before an mkv original", () => {
+ const formats = [
+ { url: `https://archive.org/download/${ITEM}/Third%20Upload-Ghi789rst_7.mkv`, ext: "mkv", format_note: "original" },
+ { url: `https://archive.org/download/${ITEM}/Third%20Upload-Ghi789rst_7.ogv`, ext: "ogv", format_note: "derivative" },
+ { url: `https://archive.org/download/${ITEM}/Third%20Upload-Ghi789rst_7.mp4`, ext: "mp4", format_note: "derivative" },
+ ];
+ assert.equal(archiveOrgPlayableUrl({ formats }), formats[2].url);
+ assert.equal(
+ archiveOrgPlayableUrl({ id: `${ITEM}/a b.mp3` }),
+ `https://archive.org/download/${ITEM}/a%20b.mp3`,
+ );
+ assert.equal(archiveOrgPlayableUrl({ id: ITEM }), undefined);
+});
+
+test("platformFromMetadata and summarize know an archive.org record", () => {
+ assert.equal(platformFromMetadata({ extractor_key: "ArchiveOrg" }), "archiveorg");
+ assert.equal(platformFromMetadata({ extractor: "archive.org" }), "archiveorg");
+ assert.equal(platformFromMetadata({ extractor_key: "YoutubeWebArchive" }), "youtube");
+ assert.equal(platformFromMetadata({ extractor_key: "Youtube" }), "youtube");
+ assert.equal(
+ platformFromMetadata({ extractor_key: "Generic", webpage_url: `https://archive.org/details/${ITEM}` }),
+ "archiveorg",
+ );
+ assert.equal(platformFromMetadata({ extractor_key: "Generic", webpage_url: "https://example.com/a.mp3" }), "youtube");
+
+ const id = archiveOrgVideoId({ identifier: ITEM, file: V1 });
+ const s = summarize("example-channel", id, {
+ id: `${ITEM}/${V1}`,
+ extractor_key: "ArchiveOrg",
+ title: "First Upload (original title)",
+ upload_date: "20230405",
+ webpage_url: `https://archive.org/details/${ITEM}/First%20Upload-AbC123xyz_9.mp4`,
+ formats: [{ url: `https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.mp4`, ext: "mp4", format_note: "original" }],
+ });
+ assert.equal(s.platform, "archiveorg");
+ assert.equal(s.id, id);
+ assert.equal(s.slug, `example-channel/${id}`);
+ assert.equal(s.mediaUrl, `https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.mp4`);
+ // Other platforms gain no key.
+ const yt = summarize("c", "AbC123xyz_9", { id: "AbC123xyz_9", extractor_key: "Youtube" });
+ assert.equal("mediaUrl" in yt, false);
+});
diff --git a/common/lib/archiveOrg.ts b/common/lib/archiveOrg.ts
@@ -0,0 +1,446 @@
+// archive.org AS A SOURCE — the pure half: what an item's metadata says, the
+// provenance a record keeps, the links a citation derives from it.
+//
+// Pure (no node:, no fetch) so the browser, the export build and the report
+// compose can all use it. The network half — the polite metadata client — is
+// lib/archiveOrgClient.ts; the per-video sidecar and the metadata patch are
+// lib/archiveOrg-server.ts.
+//
+// THE RECORD. A video imported from archive.org keeps yt-dlp's
+// metadata.info.json like every other (extractor_key "ArchiveOrg"), plus one
+// sidecar, `archiveorg.json` (ArchiveOrgProvenance below): the identifier and
+// file, the item's own title/date/creator/collections, the torrent, and — when
+// the item is a mirror of a YouTube upload — the ORIGINAL's id, URL, title,
+// date and uploader, read from the info.json the mirroring tool uploaded beside
+// the media.
+//
+// WEB-ARCHIVE (WARC / Wayback) RECORDS ARE NOT THIS. A captured web page is a
+// different kind of record with its own reader; it would get its own platform
+// and module beside this one, never a branch in it.
+
+import {
+ archiveOrgDetailsUrl,
+ archiveOrgDownloadUrl,
+ archiveOrgTorrentUrl,
+ parseArchiveOrgUrl,
+ youtubeIdFromFileName,
+ youtubeIdFromIdentifier,
+ type ArchiveOrgRef,
+} from "./archiveOrgId";
+
+export const ARCHIVE_ORG_PROVENANCE_FILENAME = "archiveorg.json";
+export const ARCHIVE_ORG_PROVENANCE_VERSION = 1;
+
+// ─── The item, as the metadata API returns it ───
+
+// https://archive.org/metadata/<identifier> — only the fields read here.
+export type ArchiveOrgItemFile = {
+ name: string;
+ // "original" | "derivative" | "metadata"
+ source?: string;
+ // archive.org's format name ("h.264", "MPEG4", "VBR MP3", "JSON", …).
+ format?: string;
+ size?: string;
+ // Seconds, as a string ("123.45") or a clock ("02:03").
+ length?: string;
+ title?: string;
+ // On a derivative: the original it was made from.
+ original?: string;
+};
+
+export type ArchiveOrgItemMetadata = {
+ metadata: {
+ identifier: string;
+ title?: string | string[];
+ date?: string;
+ publicdate?: string;
+ creator?: string | string[];
+ uploader?: string;
+ collection?: string | string[];
+ mediatype?: string;
+ description?: string | string[];
+ };
+ files: ArchiveOrgItemFile[];
+};
+
+function firstString(v: unknown): string | undefined {
+ if (typeof v === "string") return v.trim() || undefined;
+ if (Array.isArray(v)) {
+ for (const x of v) if (typeof x === "string" && x.trim()) return x.trim();
+ }
+ return undefined;
+}
+
+function stringList(v: unknown): string[] {
+ if (typeof v === "string") return v.trim() ? [v.trim()] : [];
+ if (Array.isArray(v)) return v.filter((x): x is string => typeof x === "string" && !!x.trim());
+ return [];
+}
+
+// The metadata API answers `{}` (HTTP 200) for an identifier with no item, and
+// a "dark" item has metadata but no files. Null for anything without both.
+export function parseArchiveOrgItemMetadata(raw: unknown): ArchiveOrgItemMetadata | null {
+ if (!raw || typeof raw !== "object") return null;
+ const r = raw as { metadata?: unknown; files?: unknown };
+ const m = r.metadata as Record<string, unknown> | undefined;
+ if (!m || typeof m !== "object" || typeof m.identifier !== "string") return null;
+ const files = Array.isArray(r.files)
+ ? r.files.filter(
+ (f): f is ArchiveOrgItemFile =>
+ !!f && typeof f === "object" && typeof (f as { name?: unknown }).name === "string",
+ )
+ : [];
+ return { metadata: m as ArchiveOrgItemMetadata["metadata"], files };
+}
+
+// The extensions archive.org serves media under that yt-dlp can download.
+const MEDIA_EXTS = new Set([
+ "mp4", "m4v", "mkv", "webm", "mov", "avi", "mpeg", "mpg", "ogv", "flv", "wmv", "3gp", "ts",
+ "mp3", "m4a", "ogg", "oga", "opus", "flac", "wav", "aac", "wma",
+]);
+
+function extOf(name: string): string {
+ const m = /\.([A-Za-z0-9]{1,8})$/.exec(name);
+ return m ? m[1].toLowerCase() : "";
+}
+
+export function isMediaFileName(name: string): boolean {
+ return MEDIA_EXTS.has(extOf(name));
+}
+
+// The item's media files: ORIGINALS only (a derivative is archive.org's
+// transcode of one, the same recording), in the item's own order.
+export function listArchiveOrgMediaFiles(item: ArchiveOrgItemMetadata): ArchiveOrgItemFile[] {
+ return item.files.filter((f) => f.source === "original" && isMediaFileName(f.name));
+}
+
+// ─── Picking files for a bulk import ───
+
+export type ArchiveOrgFileSelection =
+ | { files: string[] }
+ | { match: string };
+
+export type ArchiveOrgFilePick = {
+ // The chosen media files, in the item's order.
+ picked: string[];
+ // Named in `files` but not a media original of the item.
+ unknown: string[];
+};
+
+// `files` names exact paths; `match` is a case-insensitive regex over each
+// media file's path. Either way only media originals can be picked.
+export function pickArchiveOrgFiles(
+ item: ArchiveOrgItemMetadata,
+ sel: ArchiveOrgFileSelection,
+): ArchiveOrgFilePick {
+ const media = listArchiveOrgMediaFiles(item).map((f) => f.name);
+ if ("files" in sel) {
+ const want = new Set(sel.files);
+ const known = new Set(media);
+ return {
+ picked: media.filter((n) => want.has(n)),
+ unknown: sel.files.filter((n) => !known.has(n)),
+ };
+ }
+ const re = new RegExp(sel.match, "i");
+ return { picked: media.filter((n) => re.test(n)), unknown: [] };
+}
+
+// ─── The mirror's original ───
+
+export type ArchiveOrgMirror = {
+ platform: "youtube";
+ id: string;
+ url: string;
+ title?: string;
+ // YYYYMMDD
+ uploadDate?: string;
+ uploader?: string;
+ channelUrl?: string;
+ description?: string;
+ // Where the original's identity was read from: the uploaded info.json (and
+ // then every field above is the original's own), or only a name.
+ from: "info-json" | "identifier" | "file-name";
+};
+
+// The info.json a mirroring tool uploaded beside a media file: the file's own
+// stem + `.info.json`. For a single-media item, the item's only info.json.
+export function findArchiveOrgInfoJson(
+ item: ArchiveOrgItemMetadata,
+ file: string | undefined,
+): string | null {
+ const names = item.files.map((f) => f.name);
+ const infos = names.filter((n) => n.endsWith(".info.json"));
+ if (file) {
+ const stem = file.replace(/\.[A-Za-z0-9]{1,8}$/, "");
+ if (infos.includes(`${stem}.info.json`)) return `${stem}.info.json`;
+ return null;
+ }
+ return infos.length === 1 ? infos[0] : null;
+}
+
+function youtubeWatchUrl(id: string): string {
+ return `https://www.youtube.com/watch?v=${id}`;
+}
+
+// The original, from an uploaded info.json when there is one (authoritative:
+// the record yt-dlp wrote when it downloaded the upload), else from the names.
+export function archiveOrgMirrorOf(opts: {
+ identifier: string;
+ file?: string;
+ infoJson?: unknown;
+}): ArchiveOrgMirror | null {
+ const info = opts.infoJson as Record<string, unknown> | null | undefined;
+ if (info && typeof info === "object") {
+ const key = String(info.extractor_key ?? info.extractor ?? "");
+ const id = typeof info.id === "string" ? info.id : "";
+ if (/^youtube$/i.test(key) && /^[A-Za-z0-9_-]{11}$/.test(id)) {
+ const str = (k: string) => (typeof info[k] === "string" && (info[k] as string).trim() ? (info[k] as string) : undefined);
+ const date = str("upload_date");
+ return {
+ platform: "youtube",
+ id,
+ url: youtubeWatchUrl(id),
+ title: str("title"),
+ uploadDate: date && /^\d{8}$/.test(date) ? date : undefined,
+ uploader: str("uploader") ?? str("channel"),
+ channelUrl: str("channel_url") ?? str("uploader_url"),
+ description: str("description"),
+ from: "info-json",
+ };
+ }
+ }
+ const fromIdent = youtubeIdFromIdentifier(opts.identifier);
+ if (fromIdent) return { platform: "youtube", id: fromIdent, url: youtubeWatchUrl(fromIdent), from: "identifier" };
+ const fromName = opts.file ? youtubeIdFromFileName(opts.file) : null;
+ if (fromName) return { platform: "youtube", id: fromName, url: youtubeWatchUrl(fromName), from: "file-name" };
+ return null;
+}
+
+// ─── The provenance sidecar ───
+
+export type ArchiveOrgProvenance = {
+ version: number;
+ identifier: string;
+ // The file's path inside the item; absent for a whole item.
+ file?: string;
+ // The item's page, and the file's own page inside it.
+ itemUrl: string;
+ fileUrl?: string;
+ // The media file's bytes, when the record is one file.
+ downloadUrl?: string;
+ // The item's BitTorrent file, when the item lists one.
+ torrentUrl?: string;
+ item: {
+ title?: string;
+ // archive.org's `date` (the content date the uploader gave), and
+ // `publicdate` (when the item went up).
+ date?: string;
+ publicDate?: string;
+ creator?: string;
+ collections: string[];
+ mediatype?: string;
+ };
+ // The file's own title in the item, when it has one.
+ fileTitle?: string;
+ mirror?: ArchiveOrgMirror;
+ fetchedAt: string;
+};
+
+export function buildArchiveOrgProvenance(opts: {
+ ref: ArchiveOrgRef;
+ item: ArchiveOrgItemMetadata;
+ infoJson?: unknown;
+ fetchedAt: string;
+}): ArchiveOrgProvenance {
+ const { ref, item } = opts;
+ const m = item.metadata;
+ const identifier = m.identifier || ref.identifier;
+ const fileEntry = ref.file ? item.files.find((f) => f.name === ref.file) : undefined;
+ const torrentName = `${identifier}_archive.torrent`;
+ const hasTorrent = item.files.some((f) => f.name === torrentName);
+ const mirror = archiveOrgMirrorOf({ identifier, file: ref.file, infoJson: opts.infoJson });
+ // The uploader field is the uploading account's e-mail address; it is never
+ // kept. `creator` is the public credit.
+ const prov: ArchiveOrgProvenance = {
+ version: ARCHIVE_ORG_PROVENANCE_VERSION,
+ identifier,
+ ...(ref.file ? { file: ref.file } : {}),
+ itemUrl: archiveOrgDetailsUrl({ identifier }),
+ ...(ref.file
+ ? {
+ fileUrl: archiveOrgDetailsUrl({ identifier, file: ref.file }),
+ downloadUrl: archiveOrgDownloadUrl(identifier, ref.file),
+ }
+ : {}),
+ ...(hasTorrent ? { torrentUrl: archiveOrgTorrentUrl(identifier) } : {}),
+ item: {
+ title: firstString(m.title),
+ date: firstString(m.date),
+ publicDate: firstString(m.publicdate),
+ creator: firstString(m.creator),
+ collections: stringList(m.collection),
+ mediatype: firstString(m.mediatype),
+ },
+ ...(fileEntry?.title?.trim() ? { fileTitle: fileEntry.title.trim() } : {}),
+ ...(mirror ? { mirror } : {}),
+ fetchedAt: opts.fetchedAt,
+ };
+ return stripUndefinedDeep(prov);
+}
+
+function stripUndefinedDeep<T>(v: T): T {
+ if (Array.isArray(v)) return v.map(stripUndefinedDeep) as T;
+ if (v && typeof v === "object") {
+ const out: Record<string, unknown> = {};
+ for (const [k, x] of Object.entries(v)) if (x !== undefined) out[k] = stripUndefinedDeep(x);
+ return out as T;
+ }
+ return v;
+}
+
+// The sidecar's shape check: a record or null.
+export function coerceArchiveOrgProvenance(value: unknown): ArchiveOrgProvenance | null {
+ const v = value as Partial<ArchiveOrgProvenance> | null;
+ if (
+ !v ||
+ typeof v !== "object" ||
+ typeof v.identifier !== "string" ||
+ typeof v.itemUrl !== "string" ||
+ typeof v.fetchedAt !== "string" ||
+ !v.item ||
+ typeof v.item !== "object"
+ ) {
+ return null;
+ }
+ const item = v.item as ArchiveOrgProvenance["item"];
+ const mirror = v.mirror;
+ return {
+ ...(v as ArchiveOrgProvenance),
+ item: { ...item, collections: Array.isArray(item.collections) ? item.collections : [] },
+ ...(mirror && (typeof mirror.id !== "string" || typeof mirror.url !== "string")
+ ? { mirror: undefined }
+ : {}),
+ };
+}
+
+// ─── What the record's metadata.info.json is corrected to ───
+
+// yt-dlp describes an entry of a multi-file item with the ITEM's title and
+// page, and a mirror with archive.org's upload date. The record says:
+//
+// webpage_url the page of what was imported (the file's page inside the
+// item). NOT cosmetic: the snapshot renames every video dir to
+// `extractVideoId(webpage_url)` (reconcileVideoDirs.ts), so an
+// entry left with the item's page would be merged into the item.
+// title the original's title (a mirror), else the file's own title,
+// else its file name — never the item's for one file of many.
+// upload_date the original's date (a mirror), with its timestamp dropped
+// description the original's (a mirror), when it had one
+// uploader the original's uploader, else the item's public credit — never
+// the uploading account's e-mail address
+//
+// Returns only the keys that differ from `info`; empty when nothing does.
+export function archiveOrgMetadataPatch(
+ prov: ArchiveOrgProvenance,
+ info: Record<string, unknown>,
+): Record<string, unknown> {
+ const want: Record<string, unknown> = {};
+ want.webpage_url = prov.fileUrl ?? prov.itemUrl;
+ const mirror = prov.mirror;
+ const fileName = prov.file ? (prov.file.split("/").pop() ?? prov.file).replace(/\.[A-Za-z0-9]{1,8}$/, "") : undefined;
+ const title = mirror?.title ?? (prov.file ? (prov.fileTitle ?? fileName) : undefined);
+ if (title) want.title = title;
+ if (mirror?.uploadDate) want.upload_date = mirror.uploadDate;
+ if (mirror?.description) want.description = mirror.description;
+ const uploader = mirror?.uploader ?? prov.item.creator;
+ const current = info.uploader;
+ if (uploader) want.uploader = uploader;
+ else if (typeof current === "string" && current.includes("@")) want.uploader = null;
+ const patch: Record<string, unknown> = {};
+ for (const [k, v] of Object.entries(want)) {
+ if (JSON.stringify(info[k]) !== JSON.stringify(v)) patch[k] = v;
+ }
+ // A mirror's date replaces archive.org's; the timestamp yt-dlp derived from
+ // the item's publicdate would contradict it.
+ if (patch.upload_date !== undefined && typeof info.timestamp === "number") patch.timestamp = null;
+ return patch;
+}
+
+// ─── The links a record derives ───
+
+export type ExternalLink = { label: string; url: string };
+
+// What a citation of an archive.org record links, beside its moment page:
+//
+// original where the recording was first published — the YouTube upload
+// at the cited second for a mirror, else the archive.org page
+// downloads where a reader can fetch the file to check it: the archive.org
+// page (for a mirror, since `original` is YouTube there) and the
+// item's torrent
+//
+// Derived from the record's `webpage_url` alone when there is no provenance
+// (the torrent name is archive.org's fixed `<identifier>_archive.torrent`);
+// the provenance adds the mirror.
+export function archiveOrgCitationLinks(opts: {
+ webpageUrl: string | null | undefined;
+ provenance?: ArchiveOrgProvenance | null;
+ seconds?: number;
+}): { original?: ExternalLink; downloads: ExternalLink[] } {
+ const prov = opts.provenance ?? null;
+ const ref = opts.webpageUrl ? parseArchiveOrgUrl(opts.webpageUrl) : null;
+ const identifier = prov?.identifier ?? ref?.identifier;
+ if (!identifier) return { downloads: [] };
+ const page = prov?.fileUrl ?? prov?.itemUrl ?? archiveOrgDetailsUrl(ref!);
+ const torrent: ExternalLink = {
+ label: "torrent",
+ url: prov ? (prov.torrentUrl ?? archiveOrgTorrentUrl(identifier)) : archiveOrgTorrentUrl(identifier),
+ };
+ const archive: ExternalLink = { label: "archive.org", url: page };
+ const mirror = prov?.mirror;
+ if (mirror) {
+ const secs = Math.max(0, Math.floor(opts.seconds ?? 0));
+ const u = new URL(mirror.url);
+ if (secs > 0) u.searchParams.set("t", `${secs}s`);
+ return { original: { label: "YouTube", url: u.toString() }, downloads: [archive, torrent] };
+ }
+ return { original: archive, downloads: [torrent] };
+}
+
+// ─── Playing it ───
+
+// Browser-playable containers, best first. archive.org transcodes most video
+// originals to an h.264 mp4 derivative, which every browser plays — an .avi or
+// .mpeg original does not.
+const PLAYABLE_EXTS = ["mp4", "m4v", "webm", "m4a", "mp3", "ogg", "oga", "opus"];
+
+type FormatLike = { url?: unknown; ext?: unknown; format_note?: unknown };
+
+// The one file of a record a <video> element can play from archive.org, or
+// undefined. Read from the record's yt-dlp `formats` (each an archive.org
+// download URL; `format_note` is "original" or "derivative"): the playable
+// extension ranked first wins, an original before a derivative of the same
+// extension. A file record with no formats falls back to its own file.
+export function archiveOrgPlayableUrl(meta: {
+ id?: string;
+ formats?: unknown;
+}): string | undefined {
+ const formats = Array.isArray(meta.formats) ? (meta.formats as FormatLike[]) : [];
+ let best: { rank: number; url: string } | null = null;
+ for (const f of formats) {
+ if (typeof f?.url !== "string" || !/^https:\/\/archive\.org\/download\//.test(f.url)) continue;
+ const ext = typeof f.ext === "string" ? f.ext.toLowerCase() : extOf(f.url);
+ const i = PLAYABLE_EXTS.indexOf(ext);
+ if (i < 0) continue;
+ const rank = i * 2 + (f.format_note === "original" ? 0 : 1);
+ if (!best || rank < best.rank) best = { rank, url: f.url };
+ }
+ if (best) return best.url;
+ const id = meta.id ?? "";
+ const slash = id.indexOf("/");
+ if (slash > 0) {
+ const file = id.slice(slash + 1);
+ if (PLAYABLE_EXTS.includes(extOf(file))) return archiveOrgDownloadUrl(id.slice(0, slash), file);
+ }
+ return undefined;
+}
diff --git a/common/lib/archiveOrgClient.test.ts b/common/lib/archiveOrgClient.test.ts
@@ -0,0 +1,131 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ ARCHIVE_ORG_USER_AGENT,
+ ArchiveOrgClient,
+ ArchiveOrgRequestError,
+ parseRetryAfterMs,
+} from "./archiveOrgClient";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrgClient.test.ts
+//
+// THE POLITENESS RULES, ON A FAKE CLOCK. No request leaves the process: the
+// fetch is a script of canned responses, the clock only moves when the client
+// sleeps (or when a "request" takes time), and every sleep is recorded.
+
+type Scripted = { status: number; body?: unknown; headers?: Record<string, string>; takesMs?: number };
+
+function harness(script: Scripted[], opts: ConstructorParameters<typeof ArchiveOrgClient>[1] = {}) {
+ let now = 1_000_000;
+ const sleeps: number[] = [];
+ const calls: { url: string; at: number; ua: string | null }[] = [];
+ const client = new ArchiveOrgClient(
+ {
+ now: () => now,
+ sleep: async (ms) => {
+ sleeps.push(ms);
+ now += ms;
+ },
+ random: () => 0.5, // no jitter
+ fetch: async (url, init) => {
+ const headers = new Headers(init.headers);
+ calls.push({ url, at: now, ua: headers.get("user-agent") });
+ const next = script.shift();
+ if (!next) throw new Error("script exhausted");
+ now += next.takesMs ?? 100;
+ return new Response(next.body === undefined ? "{}" : JSON.stringify(next.body), {
+ status: next.status,
+ headers: next.headers,
+ });
+ },
+ },
+ opts,
+ );
+ return { client, sleeps, calls, advance: (ms: number) => (now += ms) };
+}
+
+const META = { metadata: { identifier: "example-item", title: "Example" }, files: [{ name: "a.mp4", source: "original" }] };
+
+test("an item's metadata is asked for once, identified, and cached", async () => {
+ const h = harness([{ status: 200, body: META }]);
+ const a = await h.client.itemMetadata("example-item");
+ const b = await h.client.itemMetadata("example-item");
+ assert.equal(a, b);
+ assert.equal(h.calls.length, 1);
+ assert.equal(h.calls[0].url, "https://archive.org/metadata/example-item");
+ assert.equal(h.calls[0].ua, ARCHIVE_ORG_USER_AGENT);
+ assert.match(ARCHIVE_ORG_USER_AGENT, /\(\+https?:\/\//);
+});
+
+test("the cache expires", async () => {
+ const h = harness([{ status: 200, body: META }, { status: 200, body: META }], { cacheTtlMs: 1000 });
+ await h.client.itemMetadata("example-item");
+ h.advance(500);
+ await h.client.itemMetadata("example-item");
+ assert.equal(h.calls.length, 1);
+ h.advance(1000);
+ await h.client.itemMetadata("example-item");
+ assert.equal(h.calls.length, 2);
+});
+
+test("requests are one at a time with a gap between them", async () => {
+ const h = harness(
+ [
+ { status: 200, body: { n: 1 }, takesMs: 500 },
+ { status: 200, body: { n: 2 }, takesMs: 500 },
+ { status: 200, body: { n: 3 }, takesMs: 500 },
+ ],
+ { minGapMs: 2000 },
+ );
+ // Started together; served in order, each after the previous ended + the gap.
+ const results = await Promise.all([
+ h.client.getJson("https://archive.org/download/i/1.json"),
+ h.client.getJson("https://archive.org/download/i/2.json"),
+ h.client.getJson("https://archive.org/download/i/3.json"),
+ ]);
+ assert.deepEqual(results, [{ n: 1 }, { n: 2 }, { n: 3 }]);
+ assert.equal(h.calls[1].at - h.calls[0].at, 2500);
+ assert.equal(h.calls[2].at - h.calls[1].at, 2500);
+});
+
+test("a 429 waits for Retry-After, then succeeds", async () => {
+ const h = harness([
+ { status: 429, headers: { "retry-after": "30" } },
+ { status: 200, body: META },
+ ]);
+ await h.client.itemMetadata("example-item");
+ assert.equal(h.calls.length, 2);
+ assert.ok(h.sleeps.includes(30_000), `slept ${h.sleeps}`);
+});
+
+test("a 503 without Retry-After backs off exponentially, and stops after maxAttempts", async () => {
+ const h = harness(
+ [{ status: 503 }, { status: 503 }, { status: 503 }, { status: 503 }, { status: 200, body: META }],
+ { maxAttempts: 4, baseBackoffMs: 5000, minGapMs: 0 },
+ );
+ await assert.rejects(h.client.itemMetadata("example-item"), (err: unknown) => {
+ assert.ok(err instanceof ArchiveOrgRequestError);
+ assert.equal(err.rateLimited, true);
+ assert.match(err.message, /after 4 attempts/);
+ return true;
+ });
+ assert.equal(h.calls.length, 4);
+ // 5 s, 10 s, 20 s between the four attempts; none after the last.
+ assert.deepEqual(h.sleeps, [5000, 10000, 20000]);
+});
+
+test("a 404 is not retried; an empty answer is no item", async () => {
+ const h = harness([{ status: 404 }]);
+ await assert.rejects(h.client.getJson("https://archive.org/download/i/x.json"), /HTTP 404/);
+ assert.equal(h.calls.length, 1);
+ const h2 = harness([{ status: 200, body: {} }]);
+ await assert.rejects(h2.client.itemMetadata("no-such-item"), /has no item "no-such-item"/);
+});
+
+test("Retry-After as seconds or as an HTTP date", () => {
+ assert.equal(parseRetryAfterMs("12", 0), 12_000);
+ assert.equal(parseRetryAfterMs(new Date(60_000).toUTCString(), 0), 60_000);
+ assert.equal(parseRetryAfterMs(null, 0), null);
+ assert.equal(parseRetryAfterMs("soon", 0), null);
+});
diff --git a/common/lib/archiveOrgClient.ts b/common/lib/archiveOrgClient.ts
@@ -0,0 +1,196 @@
+// THE POLITE archive.org CLIENT — every request this app makes to archive.org
+// that is not a yt-dlp spawn (the item metadata API, a mirror's info.json).
+//
+// archive.org is a non-profit serving files off its own disks; the rules here
+// are the operator's "be polite to archive.org", made mechanical:
+//
+// ONE AT A TIME requests are chained: a second caller waits for the first
+// to finish, and then for the gap.
+// A GAP at least `minGapMs` (2 s) between the end of one request
+// and the start of the next, from this process.
+// IDENTIFIED a User-Agent naming the project and its URL.
+// BACKS OFF a 429 or 503 (and 502/504, and a dropped connection) is
+// retried after the server's Retry-After when it sends one,
+// else after an exponential wait (5 s, 10 s, 20 s… capped at
+// 2 min, ±20 % jitter), at most `maxAttempts` times in all;
+// then it stops with ArchiveOrgRequestError, `rateLimited`.
+// ASKS ONCE an item's metadata is cached in memory for `cacheTtlMs`
+// (6 h): a bulk import of 161 files of one item asks for it
+// once, and so does every per-file provenance step after it.
+//
+// The downloads themselves are yt-dlp's (one plain HTTP stream per file, paced
+// by PLATFORM_ARGS.archiveorg) and run on the `platform:archiveorg` job queue,
+// one at a time.
+
+import { PROJECT_NAME, PROJECT_URL } from "./project";
+import {
+ parseArchiveOrgItemMetadata,
+ type ArchiveOrgItemMetadata,
+} from "./archiveOrg";
+import { archiveOrgDownloadUrl, archiveOrgMetadataUrl } from "./archiveOrgId";
+
+export const ARCHIVE_ORG_USER_AGENT = `${PROJECT_NAME} archive.org import (+${PROJECT_URL})`;
+
+export class ArchiveOrgRequestError extends Error {
+ readonly status: number | null;
+ readonly rateLimited: boolean;
+ constructor(message: string, status: number | null, rateLimited: boolean) {
+ super(message);
+ this.name = "ArchiveOrgRequestError";
+ this.status = status;
+ this.rateLimited = rateLimited;
+ }
+}
+
+export type ArchiveOrgClientDeps = {
+ fetch?: (url: string, init: RequestInit) => Promise<Response>;
+ now?: () => number;
+ sleep?: (ms: number, signal?: AbortSignal) => Promise<void>;
+ random?: () => number;
+};
+
+export type ArchiveOrgClientOpts = {
+ minGapMs?: number;
+ maxAttempts?: number;
+ baseBackoffMs?: number;
+ maxBackoffMs?: number;
+ cacheTtlMs?: number;
+ timeoutMs?: number;
+};
+
+const RETRYABLE = new Set([429, 502, 503, 504]);
+
+function defaultSleep(ms: number, signal?: AbortSignal): Promise<void> {
+ return new Promise((resolve, reject) => {
+ if (signal?.aborted) return reject(signal.reason ?? new Error("aborted"));
+ const t = setTimeout(() => {
+ signal?.removeEventListener("abort", onAbort);
+ resolve();
+ }, ms);
+ const onAbort = () => {
+ clearTimeout(t);
+ reject(signal?.reason ?? new Error("aborted"));
+ };
+ signal?.addEventListener("abort", onAbort, { once: true });
+ });
+}
+
+// Retry-After: delta-seconds or an HTTP date. Null when absent or unreadable.
+export function parseRetryAfterMs(value: string | null, nowMs: number): number | null {
+ if (!value) return null;
+ const v = value.trim();
+ if (/^\d+$/.test(v)) return Number(v) * 1000;
+ const at = Date.parse(v);
+ if (Number.isFinite(at)) return Math.max(0, at - nowMs);
+ return null;
+}
+
+export class ArchiveOrgClient {
+ private readonly deps: Required<ArchiveOrgClientDeps>;
+ private readonly opts: Required<ArchiveOrgClientOpts>;
+ private chain: Promise<unknown> = Promise.resolve();
+ private lastDoneAt = -Infinity;
+ private readonly cache = new Map<string, { at: number; value: ArchiveOrgItemMetadata }>();
+ // Requests actually sent (each attempt), for tests and logs.
+ requests = 0;
+
+ constructor(deps: ArchiveOrgClientDeps = {}, opts: ArchiveOrgClientOpts = {}) {
+ this.deps = {
+ fetch: deps.fetch ?? ((url, init) => fetch(url, init)),
+ now: deps.now ?? (() => Date.now()),
+ sleep: deps.sleep ?? defaultSleep,
+ random: deps.random ?? Math.random,
+ };
+ this.opts = {
+ minGapMs: opts.minGapMs ?? 2_000,
+ maxAttempts: opts.maxAttempts ?? 4,
+ baseBackoffMs: opts.baseBackoffMs ?? 5_000,
+ maxBackoffMs: opts.maxBackoffMs ?? 120_000,
+ cacheTtlMs: opts.cacheTtlMs ?? 6 * 60 * 60 * 1000,
+ timeoutMs: opts.timeoutMs ?? 60_000,
+ };
+ }
+
+ // Run `fn` after every earlier request (and its gap) has finished.
+ private serial<T>(fn: () => Promise<T>): Promise<T> {
+ const run = this.chain.then(fn, fn);
+ this.chain = run.catch(() => {});
+ return run;
+ }
+
+ private backoffMs(attempt: number): number {
+ const exp = Math.min(this.opts.maxBackoffMs, this.opts.baseBackoffMs * 2 ** (attempt - 1));
+ const jitter = 1 + (this.deps.random() * 0.4 - 0.2);
+ return Math.round(exp * jitter);
+ }
+
+ // One GET, with the gap before it and the retry policy around it.
+ private async getOnce(url: string, accept: string, signal?: AbortSignal): Promise<Response> {
+ let lastError = "";
+ for (let attempt = 1; attempt <= this.opts.maxAttempts; attempt++) {
+ const wait = this.lastDoneAt + this.opts.minGapMs - this.deps.now();
+ if (wait > 0) await this.deps.sleep(wait, signal);
+ this.requests++;
+ let res: Response | null = null;
+ try {
+ const timeout = AbortSignal.timeout(this.opts.timeoutMs);
+ res = await this.deps.fetch(url, {
+ signal: signal ? AbortSignal.any([signal, timeout]) : timeout,
+ redirect: "follow",
+ headers: { accept, "user-agent": ARCHIVE_ORG_USER_AGENT },
+ });
+ } catch (err) {
+ if (signal?.aborted) throw err;
+ lastError = (err as Error).message;
+ } finally {
+ this.lastDoneAt = this.deps.now();
+ }
+ if (res && res.ok) return res;
+ if (res && !RETRYABLE.has(res.status)) {
+ throw new ArchiveOrgRequestError(
+ `archive.org answered HTTP ${res.status} for ${url}`,
+ res.status,
+ false,
+ );
+ }
+ if (res) lastError = `HTTP ${res.status}`;
+ if (attempt === this.opts.maxAttempts) break;
+ const retryAfter = res ? parseRetryAfterMs(res.headers.get("retry-after"), this.deps.now()) : null;
+ const delay = Math.min(this.opts.maxBackoffMs, retryAfter ?? this.backoffMs(attempt));
+ await this.deps.sleep(delay, signal);
+ }
+ throw new ArchiveOrgRequestError(
+ `archive.org did not answer ${url} after ${this.opts.maxAttempts} attempts (${lastError}); stopping`,
+ null,
+ true,
+ );
+ }
+
+ getJson(url: string, signal?: AbortSignal): Promise<unknown> {
+ return this.serial(async () => {
+ const res = await this.getOnce(url, "application/json", signal);
+ return res.json();
+ });
+ }
+
+ // The item's metadata, once per `cacheTtlMs`.
+ async itemMetadata(identifier: string, signal?: AbortSignal): Promise<ArchiveOrgItemMetadata> {
+ const hit = this.cache.get(identifier);
+ if (hit && this.deps.now() - hit.at < this.opts.cacheTtlMs) return hit.value;
+ const raw = await this.getJson(archiveOrgMetadataUrl(identifier), signal);
+ const item = parseArchiveOrgItemMetadata(raw);
+ if (!item) {
+ throw new ArchiveOrgRequestError(`archive.org has no item "${identifier}"`, 404, false);
+ }
+ this.cache.set(identifier, { at: this.deps.now(), value: item });
+ return item;
+ }
+
+ // A small JSON file inside an item (a mirror's `.info.json`).
+ itemJsonFile(identifier: string, file: string, signal?: AbortSignal): Promise<unknown> {
+ return this.getJson(archiveOrgDownloadUrl(identifier, file), signal);
+ }
+}
+
+// THE process's client: one chain, one gap, one cache for every caller.
+export const archiveOrgClient = new ArchiveOrgClient();
diff --git a/common/lib/archiveOrgId.test.ts b/common/lib/archiveOrgId.test.ts
@@ -0,0 +1,95 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ archiveOrgDetailsUrl,
+ archiveOrgDownloadUrl,
+ archiveOrgTorrentUrl,
+ archiveOrgVideoId,
+ archiveOrgVideoIdFromNativeId,
+ parseArchiveOrgUrl,
+ youtubeIdFromFileName,
+ youtubeIdFromIdentifier,
+} from "./archiveOrgId";
+import { extractVideoId } from "./videoId";
+import { defaultWebpageUrl, detectPlatform, queueKeyForUrl } from "./platform";
+import { dataDirIdForUrl } from "../ytdlp/runYtdlp";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrgId.test.ts
+//
+// Every identifier, file and YouTube id here is invented.
+
+const ITEM = "example-item";
+const FILE = "Example Talk (Part 1)-AbC123xyz_9.mp4";
+
+test("archive.org item hosts are detected; the Wayback Machine is not", () => {
+ assert.equal(detectPlatform(`https://archive.org/details/${ITEM}`), "archiveorg");
+ assert.equal(detectPlatform(`https://www.archive.org/details/${ITEM}`), "archiveorg");
+ assert.equal(detectPlatform("https://web.archive.org/web/2020/https://example.com/"), null);
+ assert.equal(queueKeyForUrl(`https://archive.org/details/${ITEM}`), "platform:archiveorg");
+});
+
+test("a whole item's id is its identifier, from details, embed and download URLs", () => {
+ for (const url of [
+ `https://archive.org/details/${ITEM}`,
+ `https://archive.org/details/${ITEM}/`,
+ `https://archive.org/details/${ITEM}?autoplay=1`,
+ `https://archive.org/embed/${ITEM}`,
+ `https://archive.org/download/${ITEM}`,
+ ]) {
+ assert.equal(extractVideoId(url), ITEM, url);
+ }
+});
+
+test("a file inside an item gets a stable, filesystem-safe, unique id", () => {
+ const url = archiveOrgDetailsUrl({ identifier: ITEM, file: FILE });
+ assert.equal(url, `https://archive.org/details/${ITEM}/Example%20Talk%20(Part%201)-AbC123xyz_9.mp4`);
+ const id = extractVideoId(url)!;
+ assert.match(id, /^example-item__Example-Talk-Part-1-AbC123xyz_9-[0-9a-f]{8}$/);
+ // The same file by every URL form, encoded or not (yt-dlp unquotes `+`).
+ assert.equal(extractVideoId(`https://archive.org/embed/${ITEM}/${encodeURIComponent(FILE)}`), id);
+ assert.equal(extractVideoId(archiveOrgDownloadUrl(ITEM, FILE)), id);
+ assert.equal(
+ extractVideoId(`https://archive.org/details/${ITEM}/Example+Talk+(Part+1)-AbC123xyz_9.mp4`),
+ id,
+ );
+ // yt-dlp's native id for the entry maps to the same id.
+ assert.equal(archiveOrgVideoIdFromNativeId(`${ITEM}/${FILE}`), id);
+ assert.equal(archiveOrgVideoIdFromNativeId(ITEM), ITEM);
+ // The output path pins to it (a safe directory name).
+ assert.equal(dataDirIdForUrl(url), id);
+ // Paths that slug alike still differ.
+ const a = archiveOrgVideoId({ identifier: ITEM, file: "a b.mp4" });
+ const b = archiveOrgVideoId({ identifier: ITEM, file: "a_b.mp4" });
+ const c = archiveOrgVideoId({ identifier: ITEM, file: "a b.mkv" });
+ assert.equal(new Set([a, b, c]).size, 3);
+ // Sub-directories are part of the path.
+ assert.equal(
+ parseArchiveOrgUrl(`https://archive.org/details/${ITEM}/disc1/01%20Intro.mp3`)?.file,
+ "disc1/01 Intro.mp3",
+ );
+});
+
+test("URLs that name no item have no id", () => {
+ assert.equal(parseArchiveOrgUrl("https://archive.org/search?query=x"), null);
+ assert.equal(parseArchiveOrgUrl("https://archive.org/details/"), null);
+ assert.equal(extractVideoId("https://archive.org/search?query=x"), null);
+});
+
+test("the torrent and page URLs", () => {
+ assert.equal(archiveOrgTorrentUrl(ITEM), `https://archive.org/download/${ITEM}/${ITEM}_archive.torrent`);
+ assert.equal(defaultWebpageUrl("archiveorg", ITEM), `https://archive.org/details/${ITEM}`);
+ const fileId = archiveOrgVideoId({ identifier: ITEM, file: FILE });
+ assert.equal(defaultWebpageUrl("archiveorg", fileId), `https://archive.org/details/${ITEM}`);
+});
+
+test("YouTube ids read from mirror names", () => {
+ assert.equal(youtubeIdFromIdentifier("youtube-AbC123xyz_9"), "AbC123xyz_9");
+ assert.equal(youtubeIdFromIdentifier(ITEM), null);
+ assert.equal(youtubeIdFromFileName(FILE), "AbC123xyz_9");
+ assert.equal(youtubeIdFromFileName("Some Title [Zz9-Qq8_Ww7].webm"), "Zz9-Qq8_Ww7");
+ assert.equal(youtubeIdFromFileName("dir/Some Title [Zz9-Qq8_Ww7].info.json"), "Zz9-Qq8_Ww7");
+ // An eleven-letter lowercase word is not an id.
+ assert.equal(youtubeIdFromFileName("interview-performance.mp4"), null);
+ assert.equal(youtubeIdFromFileName("plain.mp4"), null);
+});
diff --git a/common/lib/archiveOrgId.ts b/common/lib/archiveOrgId.ts
@@ -0,0 +1,171 @@
+// archive.org ITEM AND FILE IDENTITY — what an archive.org URL names, and the
+// canonical video id (the `data/<id>/` dir name) it maps to.
+//
+// A leaf with no imports, like lib/videoId.ts (which calls into it): the roster
+// store, the browser and the export build all canonicalize URLs through here.
+//
+// AN ITEM is archive.org's unit of upload: `https://archive.org/details/<identifier>`.
+// An identifier is ASCII letters, digits, `.`, `-` and `_` — already a safe
+// directory name, so a WHOLE ITEM (one that holds a single media file) has the
+// identifier itself as its video id.
+//
+// A FILE INSIDE AN ITEM (a channel-archive item can hold a hundred and sixty
+// videos) is `https://archive.org/details/<identifier>/<file path>` — the form
+// yt-dlp's ArchiveOrg extractor resolves to that one entry. A file path is free
+// text (spaces, brackets, unicode, sub-directories), so its id is
+//
+// <identifier>__<slug of the path without its extension>-<8 hex>
+//
+// where the slug keeps `[A-Za-z0-9_-]` and folds every other run into one `-`
+// (cut to 48 characters), and the hex is a 32-bit FNV-1a hash of the EXACT file
+// path. The slug keeps the id readable; the hash keeps it unique — two paths
+// that slug alike ("a b.mp4", "a_b.mp4", "a b.mkv") still differ. Stable: the
+// same path always gives the same id, so a re-import lands in the same dir.
+//
+// `/embed/` and `/download/` URLs name the same things and canonicalize to the
+// same ids. `web.archive.org` (the Wayback Machine) is NOT an item host and is
+// not handled here.
+
+export type ArchiveOrgRef = {
+ identifier: string;
+ // The file's path inside the item, decoded; absent for the whole item.
+ file?: string;
+};
+
+const IDENTIFIER_RE = /^[A-Za-z0-9][A-Za-z0-9._-]*$/;
+
+// Hosts that serve items. Not `web.archive.org` (Wayback) and not the
+// `iaNNNNNN.us.archive.org` storage nodes a download redirects to.
+export function isArchiveOrgItemHost(host: string): boolean {
+ const h = host.toLowerCase();
+ return h === "archive.org" || h === "www.archive.org";
+}
+
+// yt-dlp reads the path with `unquote_plus`; so do we, so the file we name is
+// the file it resolves.
+function unquotePlus(s: string): string {
+ try {
+ return decodeURIComponent(s.replace(/\+/g, " "));
+ } catch {
+ return s;
+ }
+}
+
+// The item (and file) an archive.org URL names, or null for anything else.
+export function parseArchiveOrgUrl(url: string): ArchiveOrgRef | null {
+ let u: URL;
+ try {
+ u = new URL(url);
+ } catch {
+ return null;
+ }
+ if (!isArchiveOrgItemHost(u.hostname)) return null;
+ const segs = u.pathname.split("/").filter(Boolean);
+ if (segs.length < 2) return null;
+ const kind = segs[0];
+ if (kind !== "details" && kind !== "embed" && kind !== "download") return null;
+ const identifier = unquotePlus(segs[1]);
+ if (!IDENTIFIER_RE.test(identifier)) return null;
+ const rest = segs.slice(2).map(unquotePlus);
+ const file = rest.length > 0 ? rest.join("/") : undefined;
+ // A download URL with no file is the item's file listing, i.e. the item.
+ return file ? { identifier, file } : { identifier };
+}
+
+// FNV-1a, 32 bits, over the UTF-16 code units — dependency-free and the same in
+// every runtime this module is imported into.
+function fnv1a32(s: string): string {
+ let h = 0x811c9dc5;
+ for (let i = 0; i < s.length; i++) {
+ h ^= s.charCodeAt(i);
+ h = Math.imul(h, 0x01000193) >>> 0;
+ }
+ return h.toString(16).padStart(8, "0");
+}
+
+const SLUG_MAX = 48;
+
+function fileSlug(file: string): string {
+ const base = file.replace(/\.[A-Za-z0-9]{1,8}$/, "");
+ const slug = base
+ .replace(/[^A-Za-z0-9_-]+/g, "-")
+ .replace(/-{2,}/g, "-")
+ .replace(/^-+|-+$/g, "")
+ .slice(0, SLUG_MAX)
+ .replace(/-+$/, "");
+ return slug || "file";
+}
+
+// The canonical video id of an item, or of one file inside it.
+export function archiveOrgVideoId(ref: ArchiveOrgRef): string {
+ if (!ref.file) return ref.identifier;
+ return `${ref.identifier}__${fileSlug(ref.file)}-${fnv1a32(ref.file)}`;
+}
+
+// yt-dlp's own id for an archive.org record: the identifier, or
+// `<identifier>/<file path>` for an entry of a multi-file item. The canonical id
+// of that record, or null when it is not one.
+export function archiveOrgVideoIdFromNativeId(
+ nativeId: string | null | undefined,
+): string | null {
+ if (!nativeId) return null;
+ const slash = nativeId.indexOf("/");
+ const identifier = slash < 0 ? nativeId : nativeId.slice(0, slash);
+ if (!IDENTIFIER_RE.test(identifier)) return null;
+ const file = slash < 0 ? "" : nativeId.slice(slash + 1);
+ return archiveOrgVideoId(file ? { identifier, file } : { identifier });
+}
+
+function encodePath(file: string): string {
+ return file.split("/").map(encodeURIComponent).join("/");
+}
+
+// The item page, or the file's own page inside it — the URL a record keeps as
+// its `webpage_url`, and the one every link to it uses.
+export function archiveOrgDetailsUrl(ref: ArchiveOrgRef): string {
+ const base = `https://archive.org/details/${ref.identifier}`;
+ return ref.file ? `${base}/${encodePath(ref.file)}` : base;
+}
+
+// The file's bytes.
+export function archiveOrgDownloadUrl(identifier: string, file: string): string {
+ return `https://archive.org/download/${identifier}/${encodePath(file)}`;
+}
+
+// The item's BitTorrent file. archive.org derives one for every item, named
+// `<identifier>_archive.torrent`; it covers every file in the item.
+export function archiveOrgTorrentUrl(identifier: string): string {
+ return `https://archive.org/download/${identifier}/${identifier}_archive.torrent`;
+}
+
+// The item's metadata API (one JSON document: the item's fields and its file
+// list).
+export function archiveOrgMetadataUrl(identifier: string): string {
+ return `https://archive.org/metadata/${identifier}`;
+}
+
+// THE YOUTUBE ID A MIRROR CARRIES, when it says so in its name.
+//
+// `youtube-<id>` is the identifier tubeup (the usual YouTube → archive.org
+// mirroring tool) gives an item; a file yt-dlp named carries the id as
+// `<title>-<id>.<ext>` or `<title> [<id>].<ext>`. The bracketed form is
+// unambiguous. The dashed one is a guess at the last eleven characters before
+// the extension, so it is only taken when the token is not plain lowercase
+// letters — a real id is random base64 and almost never is, while an English
+// word of eleven letters ("performance") always is.
+const YT_ID = "[A-Za-z0-9_-]{11}";
+
+export function youtubeIdFromIdentifier(identifier: string): string | null {
+ const m = new RegExp(`^youtube-(${YT_ID})$`).exec(identifier);
+ return m ? m[1] : null;
+}
+
+export function youtubeIdFromFileName(file: string): string | null {
+ const name = file.split("/").pop() ?? file;
+ const stem = name.replace(/(\.[A-Za-z0-9]{1,8})+$/, "");
+ const bracket = new RegExp(`\\[(${YT_ID})\\]$`).exec(stem);
+ if (bracket) return bracket[1];
+ const dashed = new RegExp(`(?:^|[-_ ])(${YT_ID})$`).exec(stem);
+ if (dashed && /[A-Z0-9_-]/.test(dashed[1])) return dashed[1];
+ return null;
+}
diff --git a/common/lib/detectPlatform.mjs b/common/lib/detectPlatform.mjs
@@ -24,6 +24,9 @@ export function detectPlatform(url) {
if (host === "x.com" || host.endsWith(".x.com")) return "twitter";
if (host.endsWith("twitter.com")) return "twitter";
if (host === "bsky.app" || host.endsWith(".bsky.app")) return "bluesky";
+ // archive.org ITEMS only. Not web.archive.org (the Wayback Machine's page
+ // captures are a different kind of record) — see lib/archiveOrgId.ts.
+ if (host === "archive.org" || host === "www.archive.org") return "archiveorg";
} catch {
/* fall through */
}
diff --git a/common/lib/metadataHistory.ts b/common/lib/metadataHistory.ts
@@ -60,6 +60,10 @@ export const METADATA_HISTORY_WRITERS = [
// date, description and duration from the channel's RSS feed, for a record
// imported by its enclosure URL, which carries none of them.
"feed-backfill",
+ // An archive.org record corrected from its provenance (lib/archiveOrg.ts
+ // archiveOrgMetadataPatch): the file's page and title instead of the item's,
+ // and a mirror's original title, date and uploader.
+ "archiveorg-provenance",
] as const;
export type MetadataHistoryWriter = (typeof METADATA_HISTORY_WRITERS)[number];
diff --git a/common/lib/momentUrl.ts b/common/lib/momentUrl.ts
@@ -71,8 +71,10 @@ export function platformMomentUrl(
case "twitch":
u.searchParams.set("t", twitchTime(secs));
return u.toString();
- // Rumble / Kick / unknown: the watch page has no dependable start param —
- // return the plain webpage URL rather than an invalid seek.
+ // Rumble / Kick / archive.org / unknown: the watch page has no dependable
+ // start param — return the plain webpage URL rather than an invalid seek.
+ // (archive.org's own player takes none we can rely on; the archive's
+ // viewer plays the file itself and seeks it — PlayerProvider.)
default:
return webpageUrl;
}
@@ -163,8 +165,9 @@ export function viewerMomentBaseUrl(
// The platform base. Unlike platformMomentUrl (which falls back to the bare
// webpage URL), a base MUST be appendable — so only platforms whose time param
// takes raw seconds qualify. Twitch is excluded (its `t` takes an `XhYmZs`
-// token, so appending an integer would be an invalid seek); Rumble/Kick/unknown
-// have no dependable start param at all. Null in every non-appendable case.
+// token, so appending an integer would be an invalid seek); Rumble/Kick/
+// archive.org/unknown have no dependable start param at all. Null in every
+// non-appendable case.
export function platformMomentBaseUrl(
webpageUrl: string | null | undefined,
platform: Platform | null | undefined,
diff --git a/common/lib/platform.ts b/common/lib/platform.ts
@@ -8,6 +8,9 @@ export type Platform =
| "odysee"
| "twitch"
| "kick"
+ // archive.org items (lib/archiveOrgId.ts): a whole item, or one file inside
+ // a multi-file item.
+ | "archiveorg"
| "twitter"
| "bluesky";
@@ -17,6 +20,7 @@ export const PLATFORM_VALUES: ReadonlyArray<Platform> = [
"odysee",
"twitch",
"kick",
+ "archiveorg",
"twitter",
"bluesky",
];
@@ -49,6 +53,13 @@ export function defaultWebpageUrl(platform: Platform, id: string): string {
if (platform === "twitch") return `https://www.twitch.tv/videos/${id}`;
// Kick's canonical id is the VOD UUID; /video/<uuid> resolves to the VOD.
if (platform === "kick") return `https://kick.com/video/${id}`;
+ // A whole item's id IS its identifier. A file's id is a slug + hash the file
+ // path cannot be recovered from, so this is the right page only for a whole
+ // item; every archived record carries its own `webpage_url`.
+ if (platform === "archiveorg") {
+ const file = /^(.+?)__.*-[0-9a-f]{8}$/.exec(id);
+ return `https://archive.org/details/${file ? file[1] : id}`;
+ }
// Social posts: /i/status/<id> resolves without knowing the handle. Bluesky
// has no handle-free permalink, so this is only a last-resort fallback —
// every archived post carries its own canonical `url` (see postPermalink).
diff --git a/common/lib/sidecar-server.test.ts b/common/lib/sidecar-server.test.ts
@@ -6,6 +6,7 @@ import path from "node:path";
import { SIDECAR_FILENAMES, sidecar, sidecarField } from "./sidecar-server";
import { SUB_FILE_RE } from "./videoStatus";
// Every sidecar module, so the enumeration below sees every declaration.
+import "./archiveOrg-server";
import "./attribution-server";
import "./diarization-server";
import "./digest-server";
@@ -41,10 +42,11 @@ async function scratch(): Promise<string> {
return mkdtemp(path.join(os.tmpdir(), "sidecar-"));
}
-test("every declared sidecar filename escapes SUB_FILE_RE, and all ten are declared", () => {
+test("every declared sidecar filename escapes SUB_FILE_RE, and all eleven are declared", () => {
assert.deepEqual([...SIDECAR_FILENAMES].sort(), [
"ai-digest.json",
"ai-digest.overrides.json",
+ "archiveorg.json",
"attribution.json",
"availability.json",
"diarization.json",
diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts
@@ -1,7 +1,9 @@
import { readFile } from "node:fs/promises";
import path from "node:path";
import { formatDate, formatDuration } from "./format";
-import { defaultWebpageUrl } from "./platform";
+import { defaultWebpageUrl, detectPlatform } from "./platform";
+import { archiveOrgPlayableUrl } from "./archiveOrg";
+import { archiveOrgVideoIdFromNativeId } from "./archiveOrgId";
import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts";
import type { MediaType, VideoStat, VideoStatus } from "./stats";
import type { VideoState } from "./availability";
@@ -39,6 +41,10 @@ export type RawMetadata = {
// yt-dlp's coarse kind: "video" | "livestream" | "short" (YouTube). Absent on
// platforms that don't distinguish, where we fall back to is/was_live.
media_type?: string;
+ // Every format the extractor offered. Read only for archive.org, where each
+ // is a plain download URL and one of them is what the player plays
+ // (lib/archiveOrg.ts archiveOrgPlayableUrl).
+ formats?: unknown;
};
// Read and parse a video's metadata.info.json into the typed RawMetadata
@@ -64,12 +70,25 @@ export function loadRawMetadataFromDir(
return loadRawMetadata(path.join(videoDir, "metadata.info.json"));
}
+// The platform a record is from, by yt-dlp's extractor. archive.org's
+// extractor is `ArchiveOrg` (key) / `archive.org` (name) — NOT web.archive.org's
+// `YoutubeWebArchive`, which is a YouTube video.
+//
+// AN UNKNOWN EXTRACTOR: the record's own page decides when its host is one the
+// app knows (lib/detectPlatform.mjs); otherwise "youtube", as it always has
+// been — the generic extractor (a podcast episode imported by its enclosure
+// URL) still lands there, and changing that would relabel records already
+// published.
export function platformFromMetadata(meta: RawMetadata): Platform {
const key = meta.extractor_key ?? meta.extractor ?? "";
if (/^rumble/i.test(key)) return "rumble";
if (/^lbry/i.test(key)) return "odysee";
if (/^twitch/i.test(key)) return "twitch";
if (/^kick/i.test(key)) return "kick";
+ if (/^archive\.?org$/i.test(key)) return "archiveorg";
+ if (/^youtube/i.test(key)) return "youtube";
+ const fromPage = detectPlatform(meta.webpage_url);
+ if (fromPage === "archiveorg") return fromPage;
return "youtube";
}
@@ -97,10 +116,14 @@ export function summarize(
): TranscriptSummary {
const isLivestream = isLivestreamMetadata(meta);
const platform = platformFromMetadata(meta);
+ // archive.org: yt-dlp's id for one file of an item is `<identifier>/<path>`,
+ // which is not a slug; the canonical id (lib/archiveOrgId.ts) is.
const id =
platform === "odysee"
? (meta.webpage_url_basename ?? meta.id ?? videoDir)
- : (meta.id ?? videoDir);
+ : platform === "archiveorg"
+ ? (archiveOrgVideoIdFromNativeId(meta.id) ?? videoDir)
+ : (meta.id ?? videoDir);
const dateFromDir = videoDir.match(/^(\d{8})(?:_|$)/)?.[1];
return {
slug: `${channelSlug}/${id}`,
@@ -118,6 +141,10 @@ export function summarize(
webpageUrl: meta.webpage_url ?? defaultWebpageUrl(platform, id),
// Kick VODs play from a persisted HLS manifest (no iframe embed exists).
hlsUrl: platform === "kick" ? meta.manifest_url : undefined,
+ // archive.org plays the file itself in a native <video>, which seeks.
+ ...(platform === "archiveorg"
+ ? { mediaUrl: archiveOrgPlayableUrl(meta) }
+ : {}),
};
}
diff --git a/common/lib/transcripts.ts b/common/lib/transcripts.ts
@@ -26,6 +26,11 @@ export type TranscriptSummary = {
// HLS master-playlist URL for platforms with no iframe embed (Kick VODs).
// Undefined for every other platform. Played via react-player/file + hls.js.
hlsUrl?: string;
+ // A plain media file URL a native <video> plays and seeks (archive.org: the
+ // record's browser-playable file, lib/archiveOrg.ts). Undefined everywhere
+ // else. New with the archiveorg platform, so no cached or normalized record
+ // predates it and no cache version moved.
+ mediaUrl?: string;
// Curated tag ids (common/lib/curatedTags.ts) — the operator's cross-channel
// vocabulary, NOT the yt-dlp keywords in `tags` above. OMITTED when empty, so
// an untagged corpus's pages stay byte-identical to the ones already on disk
diff --git a/common/lib/videoId.ts b/common/lib/videoId.ts
@@ -1,17 +1,27 @@
// The canonical video id derived from a video URL — the name every video's
-// data/<id>/ dir carries, on every platform. Deliberately a leaf module with no
-// imports at all: it lives here rather than in ytdlp/runYtdlp.ts (its original
-// home, which still re-exports it) so that low-level stores like
+// data/<id>/ dir carries, on every platform. Deliberately a leaf module whose
+// one import (archiveOrgId.ts) is itself a leaf with no imports: it lives here
+// rather than in ytdlp/runYtdlp.ts (its original home, which still re-exports it) so that low-level stores like
// controller/rosterStore.ts can canonicalize a URL without pulling in execa,
// the settings loader and the whole download pipeline.
//
// Canonical is NOT the same as yt-dlp's native extractor id: the two coincide
// on YouTube and diverge everywhere else. See archiveIdForUrl in runYtdlp.ts
// for the native-id resolution that reads metadata.info.json.
+
+import { isArchiveOrgItemHost, parseArchiveOrgUrl, archiveOrgVideoId } from "./archiveOrgId";
+
export function extractVideoId(url: string): string | null {
try {
const u = new URL(url);
const host = u.hostname.toLowerCase();
+ if (isArchiveOrgItemHost(host)) {
+ // A whole item → its identifier; one file inside an item → a stable
+ // `<identifier>__<slug>-<hash>` (lib/archiveOrgId.ts). A URL that names
+ // no item (a search page, a collection listing) has no id.
+ const ref = parseArchiveOrgUrl(url);
+ return ref ? archiveOrgVideoId(ref) : null;
+ }
if (host.endsWith("youtube.com") || host === "youtu.be") {
const v = u.searchParams.get("v");
if (v) return v;
diff --git a/common/ytdlp/channelArgs.test.ts b/common/ytdlp/channelArgs.test.ts
@@ -147,3 +147,20 @@ test("the live pace (the shared state) reaches every channelExtraArgs call", ()
globalThis.__yttAutoQueueState__ = undefined;
}
});
+
+test("archive.org is paced and backs off, and keeps the fetched page as webpage_url", () => {
+ const args = platformArgs("archiveorg");
+ assert.equal(staticSleepRequestsSeconds("archiveorg"), 2);
+ assert.deepEqual(args.slice(0, 2), ["--sleep-requests", "2"]);
+ assert.ok(args.includes("http:exp=2:120"));
+ const i = args.indexOf("--parse-metadata");
+ const [field, re] = [args[i + 1].slice(0, args[i + 1].indexOf(":")), args[i + 1].slice(args[i + 1].indexOf(":") + 1)];
+ assert.equal(field, "original_url");
+ // The regex (Python's syntax, which this subset shares) takes an archive.org
+ // page and nothing else.
+ const js = new RegExp(`^${re.replace("(?P<webpage_url>", "(?<webpage_url>")}$`);
+ assert.equal(js.exec("https://archive.org/details/example-item/a%20b.mp4")?.groups?.webpage_url, "https://archive.org/details/example-item/a%20b.mp4");
+ assert.equal(js.exec("https://www.youtube.com/watch?v=AbC123xyz_9"), null);
+ // No parallel transfer is ever asked for.
+ assert.equal(args.some((a) => /concurrent|downloader|^-N$/.test(a)), false);
+});
diff --git a/common/ytdlp/downloadFormat.test.ts b/common/ytdlp/downloadFormat.test.ts
@@ -1,6 +1,7 @@
import { test } from "node:test";
import assert from "node:assert/strict";
import {
+ ARCHIVE_ORG_AUTO_FORMAT_SELECTOR,
DOWNLOAD_FORMAT_LABELS,
DOWNLOAD_FORMAT_PRESETS,
ORIGINAL_SOURCE_FORMAT_SELECTOR,
@@ -102,3 +103,12 @@ test("sourceVideoQualityForMaxHeight: a cap at or under 720 is video_720, above
assert.equal(sourceVideoQualityForMaxHeight(1080), "original");
assert.equal(sourceVideoQualityForMaxHeight(2160), "original");
});
+
+test("auto on archive.org takes the uploader's original, an audio item's mp3", () => {
+ const sel = resolveDownloadFormatSelector("auto", "archiveorg");
+ assert.equal(sel, ARCHIVE_ORG_AUTO_FORMAT_SELECTOR);
+ assert.match(sel, /^b\[format_note=original\]\[ext=mp4\]\//);
+ assert.ok(sel.split("/").includes("mp3"));
+ // An explicit preset still applies literally.
+ assert.equal(resolveDownloadFormatSelector("bestaudio", "archiveorg"), "bestaudio/worst");
+});
diff --git a/common/ytdlp/downloadFormat.ts b/common/ytdlp/downloadFormat.ts
@@ -4,7 +4,9 @@ import type { Platform } from "../lib/platform";
// an enum rather than a free-form `-f` string so it validates like audioFormat).
// "auto" is platform-aware: Odysee/LBRY only serves the full-length audio in its
// `original` format (every HLS rung is CDN-truncated to a few minutes), so auto
-// prefers `original` there and the historical `bestaudio/worst` everywhere else.
+// prefers `original` there, archive.org gets the uploader's original file
+// (ARCHIVE_ORG_AUTO_FORMAT_SELECTOR), and the historical `bestaudio/worst`
+// everywhere else.
export type DownloadFormatPreset =
| "auto"
| "original"
@@ -126,12 +128,32 @@ export function resolveDownloadFormatSelector(
].join("/");
case "auto":
default:
+ if (platform === "archiveorg") return ARCHIVE_ORG_AUTO_FORMAT_SELECTOR;
return platform === "odysee"
? "original/bestaudio/worst"
: "bestaudio/worst";
}
}
+// archive.org's "auto": the ORIGINAL file the uploader put up, in a common
+// container, and for an audio item its MP3 (or Ogg) — not "bestaudio/worst".
+// yt-dlp knows no codecs for an archive.org format (only its extension and
+// `format_note`, "original" or "derivative"), so `bestaudio` never matches a
+// video file and `worst` would take whichever transcode sorts last. A video
+// original in another container (.avi, .mpeg) is the next rung; a FLAC/WAV
+// original loses to its MP3 derivative, a fraction of the bytes for the same
+// transcript. Anything at all is the last rung.
+export const ARCHIVE_ORG_AUTO_FORMAT_SELECTOR = [
+ "b[format_note=original][ext=mp4]",
+ "b[format_note=original][ext=mkv]",
+ "b[format_note=original][ext=webm]",
+ "b[format_note=original][ext=mp3]",
+ "mp3",
+ "ogg",
+ "b[format_note=original]",
+ "b",
+].join("/");
+
// The override chain mirrors audioFormat: per-run override beats the per-channel
// default beats the global default; "auto" is the baseline when nothing is set.
export function resolveDownloadFormatPreset(opts: {
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -64,6 +64,7 @@ import {
type MetadataScanEntry,
} from "../controller/metadataScanStore";
import { detectPlatform, type Platform } from "../lib/platform";
+import { ensureArchiveOrgProvenance } from "../lib/archiveOrg-server";
import { probeMediaDurationSec } from "./ffprobeDuration";
import {
isShortAudio,
@@ -184,6 +185,9 @@ export type ManagedDownloadOpts = {
// "video_720" = the ≤720p H.264 selector (downloadFormat.ts). Never touches
// the audio-only selector above.
persistFormatPreset?: SourceVideoQuality;
+ // Test seam: the archive.org provenance step (lib/archiveOrg-server.ts).
+ // Every production caller passes none.
+ archiveOrgProvenance?: typeof ensureArchiveOrgProvenance;
};
// When `reuseInfoJson` is true, the real download reuses the metadata the
@@ -926,6 +930,22 @@ async function runManagedDownload(
}
const metaPath = path.join(videoDir, "metadata.info.json");
+ // AN archive.org RECORD IS CORRECTED BEFORE ANYTHING READS IT: its
+ // provenance sidecar written (one cached metadata request per item), and
+ // metadata.info.json given the file's own page and title — yt-dlp writes
+ // the ITEM's for one file of a multi-file item (lib/archiveOrg-server.ts).
+ // Only after a prefetch that succeeded: a refused one is not a record.
+ if (
+ detectPlatform(opts.videoUrl) === "archiveorg" &&
+ attemptSucceeded(attempts.at(-1)?.ytdlpExitCode ?? null)
+ ) {
+ await (opts.archiveOrgProvenance ?? ensureArchiveOrgProvenance)({
+ videoDir,
+ videoUrl: opts.videoUrl,
+ onLog: opts.onLog,
+ signal: opts.signal,
+ });
+ }
const metadata = await loadRawMetadata(metaPath);
// Only wire --load-info-json into the real attempts when we actually have
// the metadata file; a failed prefetch falls through to the legacy path so
@@ -1240,6 +1260,16 @@ async function runManagedDownload(
runAudioCheck,
)
: await runAudioCheck();
+ // The audio-checked pass re-extracted, so the archive.org correction above
+ // is re-applied (from the sidecar; no request).
+ if (canonicalId && detectPlatform(opts.videoUrl) === "archiveorg") {
+ await (opts.archiveOrgProvenance ?? ensureArchiveOrgProvenance)({
+ videoDir: path.join(channelDir, "data", canonicalId),
+ videoUrl: opts.videoUrl,
+ onLog: opts.onLog,
+ signal: opts.signal,
+ });
+ }
primaryRes = {
exitCode: audioOutcome.ytdlpExitCode,
stderrTail: audioOutcome.stderrTail,
diff --git a/common/ytdlp/platformArgs.mjs b/common/ytdlp/platformArgs.mjs
@@ -34,10 +34,34 @@ import { detectPlatform } from "../lib/detectPlatform.mjs";
// (prefetch, subtitles, retries) back to back, and the two channels that 429'd
// were the two most-downloaded. Mirrors rumble's pace; a channel's own
// `ytdlpExtraArgs` still wins because it comes after.
+//
+// archiveorg: be polite to archive.org (a non-profit serving files from its
+// own disks). `--sleep-requests 2` spaces the extractor's requests (the embed
+// page, then the metadata API); `--retry-sleep` turns yt-dlp's immediate
+// retries of a refused request or a dropped transfer into an exponential wait
+// (2 s doubling to 120 s), and the stock retry counts bound how many. yt-dlp
+// downloads an archive.org file as ONE plain HTTP stream (no fragments, no
+// parallel ranges), and nothing here passes `-N` or an external downloader.
+// `--parse-metadata` makes the record's `webpage_url` the URL it was fetched
+// by: for one file of a multi-file item yt-dlp writes the ITEM's page there,
+// and the snapshot renames every video dir to `extractVideoId(webpage_url)`
+// (reconcileVideoDirs.ts) — the file's record would be merged into the item's.
+// The import always fetches by the canonical file page, so the two agree; the
+// regex only takes an archive.org URL, and leaves any other untouched.
/** @type {Readonly<Partial<Record<Platform, readonly string[]>>>} */
export const PLATFORM_ARGS = Object.freeze({
rumble: Object.freeze(["--impersonate", "chrome", "--sleep-requests", "1"]),
youtube: Object.freeze(["--sleep-requests", "1"]),
+ archiveorg: Object.freeze([
+ "--sleep-requests",
+ "2",
+ "--retry-sleep",
+ "http:exp=2:120",
+ "--retry-sleep",
+ "extractor:exp=2:120",
+ "--parse-metadata",
+ "original_url:(?P<webpage_url>https://archive\\.org/(?:details|embed|download)/.+)",
+ ]),
});
/**