Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 0580ebcf23bb5785601785a447db246361b6e61c
parent a38ef94914b2d5ecfc92a5799231fcc96dd28a52
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 15:05:53 -0400

sources: archive.org items are a platform (archiveorg) — ids, detection, provenance sidecar, polite client, import controller

- archiveorg in Platform/PLATFORM_VALUES and the one host table (archive.org,
  www.archive.org; not web.archive.org)
- canonical ids: a whole item is its identifier; one file of an item is
  <identifier>__<slug>-<fnv32 of the path> (lib/archiveOrgId.ts)
- platformFromMetadata maps yt-dlp's ArchiveOrg extractor; summarize gives the
  canonical id and a playable mediaUrl
- PLATFORM_ARGS.archiveorg: --sleep-requests 2, exponential --retry-sleep, and
  --parse-metadata keeping webpage_url the fetched file page
- "auto" format on archive.org: the original mp4/mkv/webm, an audio item's mp3
- archiveorg.json sidecar (identifier, file, item fields, torrent, the YouTube
  original of a mirror) written by the managed download after the prefetch;
  metadata.info.json corrected through patchMetadataInfo
  (by: archiveorg-provenance)
- ArchiveOrgClient: one request at a time, 2 s gap, identified, Retry-After +
  exponential backoff, stops after 4 attempts, item metadata cached
- controller/archiveOrgImport.ts: one-URL resolution (multi-file items refused),
  bulk import of chosen files, jittered gap >= 8 s, skip on disk, stop on a
  rate limit or 3 failures in a row

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/controller/archiveOrgImport.test.ts | 222+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/archiveOrgImport.ts | 354+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archiveOrg-server.test.ts | 113+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archiveOrg-server.ts | 136+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archiveOrg.test.ts | 269+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archiveOrg.ts | 446+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archiveOrgClient.test.ts | 131+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archiveOrgClient.ts | 196+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archiveOrgId.test.ts | 95+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archiveOrgId.ts | 171+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/detectPlatform.mjs | 3+++
Mcommon/lib/metadataHistory.ts | 4++++
Mcommon/lib/momentUrl.ts | 11+++++++----
Mcommon/lib/platform.ts | 11+++++++++++
Mcommon/lib/sidecar-server.test.ts | 4+++-
Mcommon/lib/transcripts-server.ts | 31+++++++++++++++++++++++++++++--
Mcommon/lib/transcripts.ts | 5+++++
Mcommon/lib/videoId.ts | 16+++++++++++++---
Mcommon/ytdlp/channelArgs.test.ts | 17+++++++++++++++++
Mcommon/ytdlp/downloadFormat.test.ts | 10++++++++++
Mcommon/ytdlp/downloadFormat.ts | 24+++++++++++++++++++++++-
Mcommon/ytdlp/downloadOneManaged.ts | 30++++++++++++++++++++++++++++++
Mcommon/ytdlp/platformArgs.mjs | 24++++++++++++++++++++++++
23 files changed, 2312 insertions(+), 11 deletions(-)

diff --git a/common/controller/archiveOrgImport.test.ts b/common/controller/archiveOrgImport.test.ts @@ -0,0 +1,222 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import type { DownloadOutcomeRecord } from "../lib/downloadOutcome"; +import type { ChannelConfig } from "../lib/channelConfig"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/archiveOrgImport.test.ts +// +// The bulk import with everything injected: a scripted archive.org, a fake +// per-file download, a sleeper that only records. getPaths()/getSettings() +// memoize, so the env is set before anything imports them. Every name here is +// invented. + +const ROOT = mkdtempSync(path.join(os.tmpdir(), "archiveorg-import-")); +process.env.TRANSCRIPTS_DIR = ROOT; +process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); +writeFileSync( + process.env.SETTINGS_FILE, + JSON.stringify({ minFreeDiskGB: 0, sleepBetweenDownloadsSeconds: 3 }) + "\n", +); +mkdirSync(path.join(ROOT, "channels", "c", "data"), { recursive: true }); + +const { + ARCHIVE_ORG_MIN_GAP_SECONDS, + archiveOrgGapMs, + resolveArchiveOrgImportUrl, + runArchiveOrgImport, +} = await import("./archiveOrgImport"); +const { ArchiveOrgClient } = await import("../lib/archiveOrgClient"); +const { getPaths } = await import("../lib/paths"); +const { archiveOrgVideoId } = await import("../lib/archiveOrgId"); + +const ITEM = "example-item"; +const FILES = ["Alpha-AbC123xyz_9.mp4", "Beta-Def456uvw_8.mp4", "Gamma-Ghi789rst_7.mp4"]; + +function client(meta: unknown) { + return new ArchiveOrgClient( + { + sleep: async () => {}, + fetch: async () => new Response(JSON.stringify(meta), { status: 200 }), + }, + { minGapMs: 0 }, + ); +} + +const MULTI = { + metadata: { identifier: ITEM, title: "Example Archive" }, + files: [ + ...FILES.map((name) => ({ name, source: "original" })), + { name: "Alpha-AbC123xyz_9.info.json", source: "original" }, + { name: "Alpha-AbC123xyz_9.ogv", source: "derivative", original: FILES[0] }, + ], +}; +const SINGLE = { + metadata: { identifier: "example-film", title: "Example Film" }, + files: [{ name: "film.mp4", source: "original" }, { name: "film.ogv", source: "derivative" }], +}; + +const CONFIG: ChannelConfig = { handling: "transcribe", platform: "archiveorg" } as ChannelConfig; + +function outcome(status: DownloadOutcomeRecord["status"], failureClass?: DownloadOutcomeRecord["failureClass"]): DownloadOutcomeRecord { + return { + videoId: "x", + status, + startedAt: "2026-01-01T00:00:00.000Z", + finishedAt: "2026-01-01T00:00:01.000Z", + attempts: [], + ...(failureClass ? { failureClass } : {}), + } as DownloadOutcomeRecord; +} + +test("one URL: an item with several media files is refused, naming the way to choose", async () => { + const r = await resolveArchiveOrgImportUrl(`https://archive.org/details/${ITEM}`, { client: client(MULTI) }); + assert.equal(r.ok, false); + assert.match(!r.ok ? r.error : "", /holds 3 media files.*import-archive-org/s); +}); + +test("one URL: a file of many is that file; the only file of an item is the item", async () => { + const r = await resolveArchiveOrgImportUrl( + `https://archive.org/details/${ITEM}/${encodeURIComponent(FILES[1])}`, + { client: client(MULTI) }, + ); + assert.ok(r.ok); + assert.equal(r.ok && r.id, archiveOrgVideoId({ identifier: ITEM, file: FILES[1] })); + assert.equal(r.ok && r.url, `https://archive.org/details/${ITEM}/Beta-Def456uvw_8.mp4`); + + const whole = await resolveArchiveOrgImportUrl("https://archive.org/embed/example-film", { client: client(SINGLE) }); + assert.deepEqual(whole, { ok: true, url: "https://archive.org/details/example-film", id: "example-film", identifier: "example-film" }); + const byFile = await resolveArchiveOrgImportUrl("https://archive.org/details/example-film/film.mp4", { client: client(SINGLE) }); + assert.equal(byFile.ok && byFile.id, "example-film"); + + const missing = await resolveArchiveOrgImportUrl(`https://archive.org/details/${ITEM}/nope.mp4`, { client: client(MULTI) }); + assert.equal(missing.ok, false); +}); + +test("the gap is the configured pause, floored, plus up to half again", () => { + assert.equal(archiveOrgGapMs(0, 0), ARCHIVE_ORG_MIN_GAP_SECONDS * 1000); + assert.equal(archiveOrgGapMs(3, 0), ARCHIVE_ORG_MIN_GAP_SECONDS * 1000); + assert.equal(archiveOrgGapMs(30, 0), 30_000); + assert.equal(archiveOrgGapMs(30, 1), 45_000); +}); + +test("bulk: chosen files one at a time, paced, skipping what is on disk", async () => { + const paths = getPaths(); + // Beta is already downloaded (a transcript on disk). + const betaId = archiveOrgVideoId({ identifier: ITEM, file: FILES[1] }); + mkdirSync(path.join(ROOT, "channels", "c", "data", betaId), { recursive: true }); + writeFileSync(path.join(ROOT, "channels", "c", "data", betaId, "transcript.json"), "{}"); + + const urls: string[] = []; + const sleeps: number[] = []; + const imported: string[] = []; + const result = await runArchiveOrgImport({ + paths, + slug: "c", + channelConfig: CONFIG, + identifier: ITEM, + selection: { match: "\\.mp4$" }, + onLog: () => {}, + signal: new AbortController().signal, + deps: { + client: client(MULTI), + downloadOne: async (o) => { + urls.push(o.videoUrl); + return outcome("ok"); + }, + sleep: async (ms) => { + sleeps.push(ms); + }, + random: () => 0, + onImported: (id) => imported.push(id), + }, + }); + assert.deepEqual(urls, [ + `https://archive.org/details/${ITEM}/Alpha-AbC123xyz_9.mp4`, + `https://archive.org/details/${ITEM}/Gamma-Ghi789rst_7.mp4`, + ]); + // One gap, between the two fetches — none before the first, none for the skip. + assert.deepEqual(sleeps, [ARCHIVE_ORG_MIN_GAP_SECONDS * 1000]); + assert.deepEqual(result.imported, [FILES[0], FILES[2]]); + assert.deepEqual(result.skipped, [FILES[1]]); + assert.equal(imported.length, 2); + assert.equal(result.stopped, undefined); +}); + +test("bulk: a rate-limited file stops the batch at once", async () => { + let n = 0; + const result = await runArchiveOrgImport({ + paths: getPaths(), + slug: "c", + channelConfig: CONFIG, + identifier: ITEM, + selection: { files: [FILES[0], FILES[2]] }, + onLog: () => {}, + signal: new AbortController().signal, + deps: { + client: client(MULTI), + downloadOne: async () => { + n++; + return outcome("failed", "rate_limit"); + }, + sleep: async () => {}, + }, + }); + assert.equal(n, 1); + assert.match(result.stopped ?? "", /rate-limited/); +}); + +test("bulk: three failures in a row stop it; unknown names are reported", async () => { + const many = { + metadata: { identifier: ITEM }, + files: ["a", "b", "c", "d", "e"].map((x) => ({ name: `${x}.mp3`, source: "original" })), + }; + let n = 0; + const result = await runArchiveOrgImport({ + paths: getPaths(), + slug: "c", + channelConfig: CONFIG, + identifier: ITEM, + selection: { files: ["a.mp3", "b.mp3", "c.mp3", "d.mp3", "zz.mp3"] }, + onLog: () => {}, + signal: new AbortController().signal, + deps: { + client: client(many), + downloadOne: async () => { + n++; + throw new Error("boom"); + }, + sleep: async () => {}, + }, + }); + assert.equal(n, 3); + assert.equal(result.failed.length, 3); + assert.deepEqual(result.unknown, ["zz.mp3"]); + assert.match(result.stopped ?? "", /3 failures in a row/); +}); + +test("bulk: a dry run fetches nothing", async () => { + let n = 0; + const result = await runArchiveOrgImport({ + paths: getPaths(), + slug: "c", + channelConfig: CONFIG, + identifier: ITEM, + selection: { match: "." }, + onLog: () => {}, + signal: new AbortController().signal, + dryRun: true, + deps: { + client: client(MULTI), + downloadOne: async () => { + n++; + return outcome("ok"); + }, + }, + }); + assert.equal(n, 0); + assert.equal(result.planned, 3); +}); diff --git a/common/controller/archiveOrgImport.ts b/common/controller/archiveOrgImport.ts @@ -0,0 +1,354 @@ +// IMPORTING FROM archive.org — one item, one file of an item, or a chosen set +// of files of one item, into an existing channel. +// +// Every download is `downloadOneManaged` (the same managed path every other +// import takes), fetched by the CANONICAL page of what is imported +// (lib/archiveOrgId.ts): `https://archive.org/details/<identifier>` for an item +// holding one media file, `…/details/<identifier>/<file>` for one file of +// many. yt-dlp's ArchiveOrg extractor resolves either to exactly one record, +// and the provenance step (lib/archiveOrg-server.ts) runs inside it. +// +// AN ITEM WITH SEVERAL MEDIA FILES IS NEVER IMPORTED WHOLE. yt-dlp would treat +// it as a playlist and write every file into the one pinned `data/<id>/`; the +// import refuses it and names the way to choose files instead. +// +// POLITE (the operator: "be polite to archive.org"): +// - the item's metadata is asked for once (lib/archiveOrgClient.ts caches it) +// and before any download, so a typo'd identifier costs one request; +// - files go one at a time, on the `platform:archiveorg` queue, with a +// jittered gap between them — the channel's (else the global) +// `sleepBetweenDownloadsSeconds`, never under ARCHIVE_ORG_MIN_GAP_SECONDS, +// plus up to half again at random; +// - a file already downloaded is never fetched again; +// - a rate-limited download stops the batch at once (the platform's own +// backoff takes over), and ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES failures in +// a row stop it too. Re-running the same command resumes: what landed is +// skipped. + +import path from "node:path"; +import type { ChannelConfig } from "../lib/channelConfig"; +import type { Paths } from "../lib/paths"; +import type { DownloadOutcomeRecord } from "../lib/downloadOutcome"; +import { + listArchiveOrgMediaFiles, + pickArchiveOrgFiles, + type ArchiveOrgFileSelection, + type ArchiveOrgItemMetadata, +} from "../lib/archiveOrg"; +import { + archiveOrgDetailsUrl, + archiveOrgVideoId, + parseArchiveOrgUrl, +} from "../lib/archiveOrgId"; +import { archiveOrgClient, type ArchiveOrgClient } from "../lib/archiveOrgClient"; +import { getSettings } from "../lib/settings"; +import { diskGate } from "../lib/diskSpace"; +import { resolveCookiePolicy } from "../lib/cookiePolicy"; +import { downloadOneManaged, type ManagedDownloadOpts } from "../ytdlp/downloadOneManaged"; +import { destinationExists } from "../ytdlp/runYtdlp"; +import { mergeRosterFile } from "./rosterStore"; + +export const ARCHIVE_ORG_MIN_GAP_SECONDS = 8; +export const ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES = 3; + +// ─── One URL ─── + +export type ResolvedArchiveOrgImport = + | { ok: true; url: string; id: string; identifier: string; file?: string } + | { ok: false; error: string }; + +// What an archive.org URL imports as. An item URL for an item with ONE media +// file is the item (id = identifier); a file URL is that file — unless the +// item holds only that one file, when it is the item too, so the same +// recording never gets two ids. An item URL for an item with several media +// files is refused, naming them. +export async function resolveArchiveOrgImportUrl( + url: string, + opts: { client?: ArchiveOrgClient; signal?: AbortSignal } = {}, +): Promise<ResolvedArchiveOrgImport> { + const ref = parseArchiveOrgUrl(url); + if (!ref) { + return { + ok: false, + error: `Not an archive.org item URL: ${url} (expected https://archive.org/details/<identifier>[/<file>])`, + }; + } + const client = opts.client ?? archiveOrgClient; + let item: ArchiveOrgItemMetadata; + try { + item = await client.itemMetadata(ref.identifier, opts.signal); + } catch (err) { + return { ok: false, error: (err as Error).message }; + } + const identifier = item.metadata.identifier || ref.identifier; + const media = listArchiveOrgMediaFiles(item).map((f) => f.name); + if (ref.file) { + if (!item.files.some((f) => f.name === ref.file)) { + return { ok: false, error: `archive.org item "${identifier}" has no file "${ref.file}"` }; + } + if (media.length === 1 && media[0] === ref.file) { + return { ok: true, url: archiveOrgDetailsUrl({ identifier }), id: identifier, identifier }; + } + const file = ref.file; + return { + ok: true, + url: archiveOrgDetailsUrl({ identifier, file }), + id: archiveOrgVideoId({ identifier, file }), + identifier, + file, + }; + } + if (media.length === 0) { + return { ok: false, error: `archive.org item "${identifier}" has no media files to import` }; + } + if (media.length > 1) { + const shown = media.slice(0, 5).map((n) => `"${n}"`).join(", "); + return { + ok: false, + error: + `archive.org item "${identifier}" holds ${media.length} media files (${shown}${media.length > 5 ? ", …" : ""}). ` + + `Import one by its file URL (https://archive.org/details/${identifier}/<file>), or several with ` + + `pnpm ops import-archive-org --json '{"slug":"<channel>","item":"${identifier}","files":[…]}' (or "match": "<regex>").`, + }; + } + return { ok: true, url: archiveOrgDetailsUrl({ identifier }), id: identifier, identifier }; +} + +// ─── Many files of one item ─── + +export type ArchiveOrgImportPlanEntry = { + file: string; + url: string; + id: string; + // Already downloaded: skipped, never fetched again. + onDisk: boolean; +}; + +export type ArchiveOrgImportPlan = { + identifier: string; + title?: string; + mediaFiles: number; + entries: ArchiveOrgImportPlanEntry[]; + // Named in `files` but not a media original of the item. + unknown: string[]; +}; + +export async function planArchiveOrgImport(opts: { + identifier: string; + selection: ArchiveOrgFileSelection; + dataDir: string; + handling: ChannelConfig["handling"]; + client?: ArchiveOrgClient; + signal?: AbortSignal; +}): Promise<ArchiveOrgImportPlan> { + const client = opts.client ?? archiveOrgClient; + const item = await client.itemMetadata(opts.identifier, opts.signal); + const identifier = item.metadata.identifier || opts.identifier; + const media = listArchiveOrgMediaFiles(item); + const { picked, unknown } = pickArchiveOrgFiles(item, opts.selection); + const entries: ArchiveOrgImportPlanEntry[] = []; + for (const file of picked) { + // An item of one media file imports as the item (resolveArchiveOrgImportUrl). + const whole = media.length === 1; + const id = whole ? identifier : archiveOrgVideoId({ identifier, file }); + const url = whole ? archiveOrgDetailsUrl({ identifier }) : archiveOrgDetailsUrl({ identifier, file }); + entries.push({ + file, + url, + id, + onDisk: await destinationExists(opts.dataDir, id, opts.handling), + }); + } + const title = item.metadata.title; + return { + identifier, + ...(typeof title === "string" ? { title } : {}), + mediaFiles: media.length, + entries, + unknown, + }; +} + +// The gap before the next file: the configured pause, floored, plus up to +// half again at random so a batch never settles into a fixed beat. +export function archiveOrgGapMs(sleepBetweenDownloadsSeconds: number, random: number): number { + const base = Math.max(ARCHIVE_ORG_MIN_GAP_SECONDS, sleepBetweenDownloadsSeconds || 0); + return Math.round(base * (1 + 0.5 * Math.min(1, Math.max(0, random))) * 1000); +} + +function isOk(rec: DownloadOutcomeRecord): boolean { + return rec.status.startsWith("ok"); +} + +export type ArchiveOrgImportResult = { + identifier: string; + planned: number; + imported: string[]; + skipped: string[]; + failed: { file: string; error: string }[]; + unknown: string[]; + // Why the batch ended before its last file, when it did. + stopped?: string; +}; + +export type ArchiveOrgImportDeps = { + client?: ArchiveOrgClient; + downloadOne?: (opts: ManagedDownloadOpts) => Promise<DownloadOutcomeRecord>; + sleep?: (ms: number, signal: AbortSignal) => Promise<void>; + random?: () => number; + // After each imported file (the editor revalidates its pages). + onImported?: (id: string) => void; +}; + +function abortableSleep(ms: number, signal: AbortSignal): Promise<void> { + return new Promise((resolve) => { + if (signal.aborted) return resolve(); + const t = setTimeout(done, ms); + function done() { + clearTimeout(t); + signal.removeEventListener("abort", done); + resolve(); + } + signal.addEventListener("abort", done, { once: true }); + }); +} + +export async function runArchiveOrgImport(opts: { + paths: Paths; + slug: string; + channelConfig: ChannelConfig; + identifier: string; + selection: ArchiveOrgFileSelection; + onLog: (line: string) => void; + signal: AbortSignal; + drainSignal?: AbortSignal; + dryRun?: boolean; + deps?: ArchiveOrgImportDeps; +}): Promise<ArchiveOrgImportResult> { + const deps = opts.deps ?? {}; + const downloadOne = deps.downloadOne ?? downloadOneManaged; + const sleep = deps.sleep ?? abortableSleep; + const random = deps.random ?? Math.random; + const settings = getSettings(); + const dataDir = path.join(opts.paths.channelsDir, opts.slug, "data"); + const log = opts.onLog; + + const plan = await planArchiveOrgImport({ + identifier: opts.identifier, + selection: opts.selection, + dataDir, + handling: opts.channelConfig.handling, + client: deps.client, + signal: opts.signal, + }); + const result: ArchiveOrgImportResult = { + identifier: plan.identifier, + planned: plan.entries.length, + imported: [], + skipped: [], + failed: [], + unknown: plan.unknown, + }; + log( + `archive.org item ${plan.identifier}${plan.title ? ` ("${plan.title}")` : ""}: ` + + `${plan.mediaFiles} media files, ${plan.entries.length} chosen, ` + + `${plan.entries.filter((e) => e.onDisk).length} already downloaded.\n`, + ); + if (plan.unknown.length > 0) { + log(`Not media files of the item (ignored): ${plan.unknown.map((n) => JSON.stringify(n)).join(", ")}\n`); + } + if (opts.dryRun) { + for (const e of plan.entries) log(` ${e.onDisk ? "on disk " : "would get"} ${e.id} ${e.file}\n`); + result.skipped = plan.entries.filter((e) => e.onDisk).map((e) => e.file); + return result; + } + + const sleepSeconds = + opts.channelConfig.sleepBetweenDownloadsSeconds ?? settings.sleepBetweenDownloadsSeconds; + let fetched = 0; + let consecutiveFailures = 0; + for (const entry of plan.entries) { + if (opts.signal.aborted) { + result.stopped = "cancelled"; + break; + } + if (opts.drainSignal?.aborted) { + result.stopped = "drained"; + break; + } + if (entry.onDisk || (await destinationExists(dataDir, entry.id, opts.channelConfig.handling))) { + result.skipped.push(entry.file); + continue; + } + const gate = await diskGate(opts.paths, settings, { dir: dataDir }); + if (!gate.ok) { + result.stopped = gate.message; + log(`Stopping: ${gate.message}.\n`); + break; + } + if (fetched > 0) { + const gap = archiveOrgGapMs(sleepSeconds, random()); + log(`Waiting ${(gap / 1000).toFixed(1)}s before the next file (archive.org pacing)...\n`); + await sleep(gap, opts.signal); + if (opts.signal.aborted) { + result.stopped = "cancelled"; + break; + } + } + fetched++; + log(`[${fetched}] ${entry.file} → data/${entry.id}/\n`); + let rec: DownloadOutcomeRecord | null = null; + let error = ""; + try { + rec = await downloadOne({ + channelSlug: opts.slug, + channelConfig: opts.channelConfig, + paths: opts.paths, + videoUrl: entry.url, + onLog: log, + signal: opts.signal, + cookiePolicy: resolveCookiePolicy(settings, opts.channelConfig), + inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback, + globalSkipLiveDownloads: settings.skipLiveDownloads, + appendArchive: true, + }); + } catch (err) { + error = (err as Error).message; + } + if (rec && isOk(rec)) { + consecutiveFailures = 0; + result.imported.push(entry.file); + await mergeRosterFile( + opts.paths, + opts.slug, + [{ id: entry.id, url: entry.url }], + new Date().toISOString(), + "import", + ).catch(() => { + /* the download succeeded; a roster write failure must not fail it */ + }); + deps.onImported?.(entry.id); + continue; + } + consecutiveFailures++; + const why = error || rec?.attempts.at(-1)?.error || rec?.status || "failed"; + result.failed.push({ file: entry.file, error: why }); + log(` failed: ${why}\n`); + if (rec?.failureClass === "rate_limit") { + result.stopped = "archive.org rate-limited the download; stopping (re-run later — files on disk are skipped)"; + log(`${result.stopped}.\n`); + break; + } + if (consecutiveFailures >= ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES) { + result.stopped = `${consecutiveFailures} failures in a row; stopping`; + log(`${result.stopped}.\n`); + break; + } + } + log( + `archive.org import of ${plan.identifier}: ${result.imported.length} imported, ` + + `${result.skipped.length} already on disk, ${result.failed.length} failed` + + `${result.stopped ? ` — stopped: ${result.stopped}` : ""}.\n`, + ); + return result; +} diff --git a/common/lib/archiveOrg-server.test.ts b/common/lib/archiveOrg-server.test.ts @@ -0,0 +1,113 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, readFile, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { ArchiveOrgClient } from "./archiveOrgClient"; +import { + ensureArchiveOrgProvenance, + loadArchiveOrgProvenance, +} from "./archiveOrg-server"; +import { loadMetadataHistory } from "./metadataHistory-server"; +import { archiveOrgDetailsUrl } from "./archiveOrgId"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrg-server.test.ts +// +// The provenance step on a temp video dir, against a scripted archive.org. +// Every name here is invented. + +const ITEM = "example-item"; +const FILE = "First Upload-AbC123xyz_9.mp4"; +const URL_FILE = archiveOrgDetailsUrl({ identifier: ITEM, file: FILE }); + +const ITEM_META = { + metadata: { identifier: ITEM, title: "Example Channel Archive", creator: "Example Creator", collection: "community" }, + files: [ + { name: FILE, source: "original" }, + { name: "First Upload-AbC123xyz_9.info.json", source: "original" }, + { name: "Second-Def456uvw_8.mp4", source: "original" }, + { name: `${ITEM}_archive.torrent`, source: "metadata" }, + ], +}; + +const INFO = { + id: "AbC123xyz_9", + extractor_key: "Youtube", + title: "First Upload (original)", + upload_date: "20230405", + uploader: "Example Creator", +}; + +function fakeClient(answers: Record<string, unknown>, seen: string[]) { + return new ArchiveOrgClient( + { + sleep: async () => {}, + fetch: async (url) => { + seen.push(url); + if (!(url in answers)) return new Response("{}", { status: 404 }); + return new Response(JSON.stringify(answers[url]), { status: 200 }); + }, + }, + { minGapMs: 0 }, + ); +} + +async function videoDir(info: Record<string, unknown>): Promise<string> { + const dir = await mkdtemp(path.join(os.tmpdir(), "archiveorg-prov-")); + await writeFile(path.join(dir, "metadata.info.json"), JSON.stringify(info)); + return dir; +} + +test("writes the sidecar once and corrects the record, through the history", async () => { + const seen: string[] = []; + const client = fakeClient( + { + [`https://archive.org/metadata/${ITEM}`]: ITEM_META, + [`https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.info.json`]: INFO, + }, + seen, + ); + const dir = await videoDir({ + id: `${ITEM}/${FILE}`, + extractor_key: "ArchiveOrg", + title: "Example Channel Archive", + webpage_url: `https://archive.org/details/${ITEM}`, + upload_date: "20240304", + }); + const lines: string[] = []; + const prov = await ensureArchiveOrgProvenance({ videoDir: dir, videoUrl: URL_FILE, client, onLog: (l) => lines.push(l) }); + assert.equal(prov?.mirror?.id, "AbC123xyz_9"); + assert.deepEqual(await loadArchiveOrgProvenance(dir), prov); + const info = JSON.parse(await readFile(path.join(dir, "metadata.info.json"), "utf8")); + assert.equal(info.webpage_url, URL_FILE); + assert.equal(info.title, "First Upload (original)"); + assert.equal(info.upload_date, "20230405"); + assert.equal(info.id, `${ITEM}/${FILE}`); + const history = await loadMetadataHistory(dir); + assert.equal(history?.entries.at(-1)?.by, "archiveorg-provenance"); + assert.equal(seen.length, 2); + + // A second run (the next download of the record) asks nothing. + await ensureArchiveOrgProvenance({ videoDir: dir, videoUrl: URL_FILE, client }); + assert.equal(seen.length, 2); +}); + +test("archive.org unreachable: no sidecar, but the page is still the file's", async () => { + const seen: string[] = []; + const client = fakeClient({}, seen); + const dir = await videoDir({ id: `${ITEM}/${FILE}`, webpage_url: `https://archive.org/details/${ITEM}`, title: "Item" }); + const lines: string[] = []; + const prov = await ensureArchiveOrgProvenance({ videoDir: dir, videoUrl: URL_FILE, client, onLog: (l) => lines.push(l) }); + assert.equal(prov, null); + assert.equal(await loadArchiveOrgProvenance(dir), null); + const info = JSON.parse(await readFile(path.join(dir, "metadata.info.json"), "utf8")); + assert.equal(info.webpage_url, URL_FILE); + assert.equal(info.title, "Item"); + assert.ok(lines.some((l) => /provenance not fetched/.test(l))); +}); + +test("a URL that is not archive.org is left alone", async () => { + const dir = await videoDir({ id: "x", webpage_url: "https://example.com/x" }); + assert.equal(await ensureArchiveOrgProvenance({ videoDir: dir, videoUrl: "https://example.com/x" }), null); +}); diff --git a/common/lib/archiveOrg-server.ts b/common/lib/archiveOrg-server.ts @@ -0,0 +1,136 @@ +// archive.org PROVENANCE ON DISK — the `archiveorg.json` sidecar, and the step +// every managed download of an archive.org record runs after yt-dlp writes its +// metadata (ytdlp/downloadOneManaged.ts, right after the prefetch). +// +// THE STEP, `ensureArchiveOrgProvenance`: +// +// 1. the sidecar: kept when it is already there for this item and file; +// otherwise built from the item's metadata API (one cached request per +// item, lib/archiveOrgClient.ts) and, when the item carries one, the +// `.info.json` a mirroring tool uploaded beside the file (one request). +// 2. the record: metadata.info.json corrected from it +// (lib/archiveOrg.ts archiveOrgMetadataPatch) through `patchMetadataInfo`, +// so the change is in metadata.history.json as `archiveorg-provenance`. +// +// Step 2 needs no network and always runs: a sidecar that could not be fetched +// (archive.org refusing, the operator offline) still leaves the record's +// `webpage_url` the file's page, which is what keeps its directory its own. +// A failure is logged and does not fail the download — the next download of +// the record, or a re-import, fills it in. + +import path from "node:path"; +import { readFile } from "node:fs/promises"; +import { + ARCHIVE_ORG_PROVENANCE_FILENAME, + archiveOrgMetadataPatch, + buildArchiveOrgProvenance, + coerceArchiveOrgProvenance, + findArchiveOrgInfoJson, + type ArchiveOrgProvenance, +} from "./archiveOrg"; +import { + archiveOrgDetailsUrl, + parseArchiveOrgUrl, + type ArchiveOrgRef, +} from "./archiveOrgId"; +import { archiveOrgClient, type ArchiveOrgClient } from "./archiveOrgClient"; +import { patchMetadataInfo } from "./metadataHistory-server"; +import { sidecar, sidecarField } from "./sidecar-server"; + +export const archiveOrgProvenanceSidecar = sidecar( + ARCHIVE_ORG_PROVENANCE_FILENAME, + sidecarField(coerceArchiveOrgProvenance), +); + +export const { + load: loadArchiveOrgProvenance, + write: writeArchiveOrgProvenance, +} = archiveOrgProvenanceSidecar; + +// Fetch what the sidecar records: the item's metadata (cached) and, for a +// mirror, its uploaded info.json. +export async function fetchArchiveOrgProvenance( + ref: ArchiveOrgRef, + opts: { client?: ArchiveOrgClient; signal?: AbortSignal; now?: () => Date } = {}, +): Promise<ArchiveOrgProvenance> { + const client = opts.client ?? archiveOrgClient; + const item = await client.itemMetadata(ref.identifier, opts.signal); + const infoName = findArchiveOrgInfoJson(item, ref.file); + let infoJson: unknown; + if (infoName) { + try { + infoJson = await client.itemJsonFile(item.metadata.identifier, infoName, opts.signal); + } catch { + // The names still say whether it is a mirror; the info.json only adds + // the original's title and date. + infoJson = undefined; + } + } + return buildArchiveOrgProvenance({ + ref, + item, + infoJson, + fetchedAt: (opts.now?.() ?? new Date()).toISOString(), + }); +} + +async function readInfo(videoDir: string): Promise<Record<string, unknown> | null> { + try { + const v = JSON.parse(await readFile(path.join(videoDir, "metadata.info.json"), "utf8")); + return v && typeof v === "object" && !Array.isArray(v) ? (v as Record<string, unknown>) : null; + } catch { + return null; + } +} + +export type EnsureArchiveOrgProvenanceOpts = { + videoDir: string; + // The URL the record was fetched by (a details/embed/download URL). + videoUrl: string; + onLog?: (line: string) => void; + signal?: AbortSignal; + client?: ArchiveOrgClient; + now?: () => Date; +}; + +export async function ensureArchiveOrgProvenance( + opts: EnsureArchiveOrgProvenanceOpts, +): Promise<ArchiveOrgProvenance | null> { + const log = opts.onLog ?? (() => {}); + const ref = parseArchiveOrgUrl(opts.videoUrl); + if (!ref) return null; + let prov = await loadArchiveOrgProvenance(opts.videoDir); + if (prov && (prov.identifier !== ref.identifier || (prov.file ?? "") !== (ref.file ?? ""))) { + prov = null; + } + if (!prov) { + try { + prov = await fetchArchiveOrgProvenance(ref, opts); + await writeArchiveOrgProvenance(opts.videoDir, prov); + log( + `archive.org provenance: ${prov.identifier}${prov.file ? ` / ${prov.file}` : ""}` + + (prov.mirror ? ` — mirror of YouTube ${prov.mirror.id} (from ${prov.mirror.from})` : "") + + `.\n`, + ); + } catch (err) { + log(`archive.org provenance not fetched (${(err as Error).message}); the record keeps yt-dlp's fields.\n`); + } + } + const info = await readInfo(opts.videoDir); + if (!info) return prov; + // Without a sidecar only the page is corrected — it is the one field the + // directory's name depends on. + const patch = prov + ? archiveOrgMetadataPatch(prov, info) + : info.webpage_url === archiveOrgDetailsUrl(ref) + ? {} + : { webpage_url: archiveOrgDetailsUrl(ref) }; + if (Object.keys(patch).length > 0) { + try { + await patchMetadataInfo(opts.videoDir, patch, { by: "archiveorg-provenance", onLog: log }); + } catch (err) { + log(`Could not correct metadata.info.json from the archive.org provenance: ${(err as Error).message}\n`); + } + } + return prov; +} diff --git a/common/lib/archiveOrg.test.ts b/common/lib/archiveOrg.test.ts @@ -0,0 +1,269 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + archiveOrgCitationLinks, + archiveOrgMetadataPatch, + archiveOrgPlayableUrl, + buildArchiveOrgProvenance, + coerceArchiveOrgProvenance, + findArchiveOrgInfoJson, + listArchiveOrgMediaFiles, + parseArchiveOrgItemMetadata, + pickArchiveOrgFiles, + type ArchiveOrgItemMetadata, +} from "./archiveOrg"; +import { archiveOrgVideoId } from "./archiveOrgId"; +import { platformFromMetadata, summarize } from "./transcripts-server"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrg.test.ts +// +// A synthetic channel-archive item: three videos uploaded by a mirroring +// tool, each with its yt-dlp info.json, plus archive.org's derivatives. Every +// name, id and date is invented. + +const ITEM = "example-item"; +const V1 = "First Upload-AbC123xyz_9.mp4"; +const V2 = "Second Upload-Def456uvw_8.mp4"; +const V3 = "Third Upload-Ghi789rst_7.mkv"; + +function item(over: Partial<ArchiveOrgItemMetadata["metadata"]> = {}): ArchiveOrgItemMetadata { + return parseArchiveOrgItemMetadata({ + metadata: { + identifier: ITEM, + title: "Example Channel Archive", + date: "2024-01-02", + publicdate: "2024-03-04 05:06:07", + creator: "Example Creator", + uploader: "someone@example.org", + collection: ["opensource_movies", "community"], + mediatype: "movies", + ...over, + }, + files: [ + { name: V1, source: "original", format: "MPEG4" }, + { name: "First Upload-AbC123xyz_9.info.json", source: "original", format: "JSON" }, + { name: V2, source: "original", format: "MPEG4", title: "Second, as titled on archive.org" }, + { name: "Second Upload-Def456uvw_8.info.json", source: "original", format: "JSON" }, + { name: V3, source: "original", format: "Matroska" }, + { name: "Third Upload-Ghi789rst_7.mp4", source: "derivative", format: "h.264", original: V3 }, + { name: "Third Upload-Ghi789rst_7.ogv", source: "derivative", format: "Ogg Video", original: V3 }, + { name: `${ITEM}_archive.torrent`, source: "metadata", format: "Archive BitTorrent" }, + { name: `${ITEM}_meta.xml`, source: "original", format: "Metadata" }, + ], + })!; +} + +const MIRROR_INFO = { + id: "AbC123xyz_9", + extractor_key: "Youtube", + title: "First Upload (original title)", + upload_date: "20230405", + uploader: "Example Creator", + channel_url: "https://www.youtube.com/channel/UCexample", + description: "The original description.", +}; + +test("the metadata API's answer: an item, or null for an unknown identifier", () => { + assert.equal(parseArchiveOrgItemMetadata({}), null); + assert.equal(parseArchiveOrgItemMetadata(null), null); + assert.equal(item().files.length, 9); +}); + +test("media files are the originals with a media extension", () => { + assert.deepEqual(listArchiveOrgMediaFiles(item()).map((f) => f.name), [V1, V2, V3]); +}); + +test("a bulk import picks by exact names or by a case-insensitive regex", () => { + assert.deepEqual(pickArchiveOrgFiles(item(), { files: [V2, "nope.mp4", `${ITEM}_meta.xml`] }), { + picked: [V2], + unknown: ["nope.mp4", `${ITEM}_meta.xml`], + }); + assert.deepEqual(pickArchiveOrgFiles(item(), { match: "^(first|third)" }).picked, [V1, V3]); + assert.deepEqual(pickArchiveOrgFiles(item(), { match: "\\.mkv$" }).picked, [V3]); +}); + +test("a file's info.json is its stem's; a single-media item's is its only one", () => { + assert.equal(findArchiveOrgInfoJson(item(), V1), "First Upload-AbC123xyz_9.info.json"); + assert.equal(findArchiveOrgInfoJson(item(), V3), null); + assert.equal(findArchiveOrgInfoJson(item(), undefined), null); +}); + +test("provenance of a mirror reads the original from its info.json", () => { + const prov = buildArchiveOrgProvenance({ + ref: { identifier: ITEM, file: V1 }, + item: item(), + infoJson: MIRROR_INFO, + fetchedAt: "2026-01-01T00:00:00.000Z", + }); + assert.equal(prov.identifier, ITEM); + assert.equal(prov.file, V1); + assert.equal(prov.itemUrl, `https://archive.org/details/${ITEM}`); + assert.equal(prov.fileUrl, `https://archive.org/details/${ITEM}/First%20Upload-AbC123xyz_9.mp4`); + assert.equal(prov.downloadUrl, `https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.mp4`); + assert.equal(prov.torrentUrl, `https://archive.org/download/${ITEM}/${ITEM}_archive.torrent`); + assert.deepEqual(prov.item, { + title: "Example Channel Archive", + date: "2024-01-02", + publicDate: "2024-03-04 05:06:07", + creator: "Example Creator", + collections: ["opensource_movies", "community"], + mediatype: "movies", + }); + assert.deepEqual(prov.mirror, { + platform: "youtube", + id: "AbC123xyz_9", + url: "https://www.youtube.com/watch?v=AbC123xyz_9", + title: "First Upload (original title)", + uploadDate: "20230405", + uploader: "Example Creator", + channelUrl: "https://www.youtube.com/channel/UCexample", + description: "The original description.", + from: "info-json", + }); + // The uploading account's e-mail is never kept. + assert.equal(JSON.stringify(prov).includes("@"), false); + assert.deepEqual(coerceArchiveOrgProvenance(JSON.parse(JSON.stringify(prov))), prov); + assert.equal(coerceArchiveOrgProvenance({ identifier: ITEM }), null); +}); + +test("without an info.json the mirror is known by name only; a plain item is no mirror", () => { + const byName = buildArchiveOrgProvenance({ + ref: { identifier: ITEM, file: V3 }, + item: item(), + fetchedAt: "2026-01-01T00:00:00.000Z", + }); + assert.deepEqual(byName.mirror, { + platform: "youtube", + id: "Ghi789rst_7", + url: "https://www.youtube.com/watch?v=Ghi789rst_7", + from: "file-name", + }); + const byIdent = buildArchiveOrgProvenance({ + ref: { identifier: "youtube-Jkl012mno_6" }, + item: item({ identifier: "youtube-Jkl012mno_6" }), + fetchedAt: "2026-01-01T00:00:00.000Z", + }); + assert.equal(byIdent.mirror?.from, "identifier"); + const plain = buildArchiveOrgProvenance({ + ref: { identifier: "example-film" }, + item: item({ identifier: "example-film" }), + fetchedAt: "2026-01-01T00:00:00.000Z", + }); + assert.equal(plain.mirror, undefined); + assert.equal(plain.file, undefined); + // A non-YouTube info.json is not a YouTube mirror. + const other = buildArchiveOrgProvenance({ + ref: { identifier: "example-film", file: "film.mp4" }, + item: item({ identifier: "example-film" }), + infoJson: { id: "abc", extractor_key: "Generic" }, + fetchedAt: "2026-01-01T00:00:00.000Z", + }); + assert.equal(other.mirror, undefined); +}); + +test("the record is corrected: the file's page and title, the original's date", () => { + const prov = buildArchiveOrgProvenance({ + ref: { identifier: ITEM, file: V1 }, + item: item(), + infoJson: MIRROR_INFO, + fetchedAt: "2026-01-01T00:00:00.000Z", + }); + const info = { + id: `${ITEM}/${V1}`, + title: "Example Channel Archive", + webpage_url: `https://archive.org/details/${ITEM}`, + upload_date: "20240304", + timestamp: 1709528767, + uploader: "someone@example.org", + }; + const patch = archiveOrgMetadataPatch(prov, info); + assert.deepEqual(patch, { + webpage_url: prov.fileUrl, + title: "First Upload (original title)", + upload_date: "20230405", + description: "The original description.", + uploader: "Example Creator", + timestamp: null, + }); + // Applied, nothing is left to change. + assert.deepEqual(archiveOrgMetadataPatch(prov, { ...info, ...patch }), {}); + + // One file of many, no mirror: the file's own archive.org title. + const v2 = buildArchiveOrgProvenance({ + ref: { identifier: ITEM, file: V2 }, + item: item({ creator: undefined }), + fetchedAt: "2026-01-01T00:00:00.000Z", + }); + const p2 = archiveOrgMetadataPatch({ ...v2, mirror: undefined }, info); + assert.equal(p2.title, "Second, as titled on archive.org"); + assert.equal(p2.uploader, null); + assert.equal("upload_date" in p2, false); +}); + +test("citation links: a mirror cites YouTube at the second, archive.org and the torrent as downloads", () => { + const prov = buildArchiveOrgProvenance({ + ref: { identifier: ITEM, file: V1 }, + item: item(), + infoJson: MIRROR_INFO, + fetchedAt: "2026-01-01T00:00:00.000Z", + }); + assert.deepEqual(archiveOrgCitationLinks({ webpageUrl: prov.fileUrl, provenance: prov, seconds: 75.6 }), { + original: { label: "YouTube", url: "https://www.youtube.com/watch?v=AbC123xyz_9&t=75s" }, + downloads: [ + { label: "archive.org", url: prov.fileUrl }, + { label: "torrent", url: `https://archive.org/download/${ITEM}/${ITEM}_archive.torrent` }, + ], + }); +}); + +test("citation links: a plain record is derived from its page alone", () => { + assert.deepEqual(archiveOrgCitationLinks({ webpageUrl: "https://archive.org/details/example-film", seconds: 30 }), { + original: { label: "archive.org", url: "https://archive.org/details/example-film" }, + downloads: [{ label: "torrent", url: "https://archive.org/download/example-film/example-film_archive.torrent" }], + }); + assert.deepEqual(archiveOrgCitationLinks({ webpageUrl: "https://example.com/x" }), { downloads: [] }); +}); + +test("the player plays a browser-playable file: an mp4 before an mkv original", () => { + const formats = [ + { url: `https://archive.org/download/${ITEM}/Third%20Upload-Ghi789rst_7.mkv`, ext: "mkv", format_note: "original" }, + { url: `https://archive.org/download/${ITEM}/Third%20Upload-Ghi789rst_7.ogv`, ext: "ogv", format_note: "derivative" }, + { url: `https://archive.org/download/${ITEM}/Third%20Upload-Ghi789rst_7.mp4`, ext: "mp4", format_note: "derivative" }, + ]; + assert.equal(archiveOrgPlayableUrl({ formats }), formats[2].url); + assert.equal( + archiveOrgPlayableUrl({ id: `${ITEM}/a b.mp3` }), + `https://archive.org/download/${ITEM}/a%20b.mp3`, + ); + assert.equal(archiveOrgPlayableUrl({ id: ITEM }), undefined); +}); + +test("platformFromMetadata and summarize know an archive.org record", () => { + assert.equal(platformFromMetadata({ extractor_key: "ArchiveOrg" }), "archiveorg"); + assert.equal(platformFromMetadata({ extractor: "archive.org" }), "archiveorg"); + assert.equal(platformFromMetadata({ extractor_key: "YoutubeWebArchive" }), "youtube"); + assert.equal(platformFromMetadata({ extractor_key: "Youtube" }), "youtube"); + assert.equal( + platformFromMetadata({ extractor_key: "Generic", webpage_url: `https://archive.org/details/${ITEM}` }), + "archiveorg", + ); + assert.equal(platformFromMetadata({ extractor_key: "Generic", webpage_url: "https://example.com/a.mp3" }), "youtube"); + + const id = archiveOrgVideoId({ identifier: ITEM, file: V1 }); + const s = summarize("example-channel", id, { + id: `${ITEM}/${V1}`, + extractor_key: "ArchiveOrg", + title: "First Upload (original title)", + upload_date: "20230405", + webpage_url: `https://archive.org/details/${ITEM}/First%20Upload-AbC123xyz_9.mp4`, + formats: [{ url: `https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.mp4`, ext: "mp4", format_note: "original" }], + }); + assert.equal(s.platform, "archiveorg"); + assert.equal(s.id, id); + assert.equal(s.slug, `example-channel/${id}`); + assert.equal(s.mediaUrl, `https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.mp4`); + // Other platforms gain no key. + const yt = summarize("c", "AbC123xyz_9", { id: "AbC123xyz_9", extractor_key: "Youtube" }); + assert.equal("mediaUrl" in yt, false); +}); diff --git a/common/lib/archiveOrg.ts b/common/lib/archiveOrg.ts @@ -0,0 +1,446 @@ +// archive.org AS A SOURCE — the pure half: what an item's metadata says, the +// provenance a record keeps, the links a citation derives from it. +// +// Pure (no node:, no fetch) so the browser, the export build and the report +// compose can all use it. The network half — the polite metadata client — is +// lib/archiveOrgClient.ts; the per-video sidecar and the metadata patch are +// lib/archiveOrg-server.ts. +// +// THE RECORD. A video imported from archive.org keeps yt-dlp's +// metadata.info.json like every other (extractor_key "ArchiveOrg"), plus one +// sidecar, `archiveorg.json` (ArchiveOrgProvenance below): the identifier and +// file, the item's own title/date/creator/collections, the torrent, and — when +// the item is a mirror of a YouTube upload — the ORIGINAL's id, URL, title, +// date and uploader, read from the info.json the mirroring tool uploaded beside +// the media. +// +// WEB-ARCHIVE (WARC / Wayback) RECORDS ARE NOT THIS. A captured web page is a +// different kind of record with its own reader; it would get its own platform +// and module beside this one, never a branch in it. + +import { + archiveOrgDetailsUrl, + archiveOrgDownloadUrl, + archiveOrgTorrentUrl, + parseArchiveOrgUrl, + youtubeIdFromFileName, + youtubeIdFromIdentifier, + type ArchiveOrgRef, +} from "./archiveOrgId"; + +export const ARCHIVE_ORG_PROVENANCE_FILENAME = "archiveorg.json"; +export const ARCHIVE_ORG_PROVENANCE_VERSION = 1; + +// ─── The item, as the metadata API returns it ─── + +// https://archive.org/metadata/<identifier> — only the fields read here. +export type ArchiveOrgItemFile = { + name: string; + // "original" | "derivative" | "metadata" + source?: string; + // archive.org's format name ("h.264", "MPEG4", "VBR MP3", "JSON", …). + format?: string; + size?: string; + // Seconds, as a string ("123.45") or a clock ("02:03"). + length?: string; + title?: string; + // On a derivative: the original it was made from. + original?: string; +}; + +export type ArchiveOrgItemMetadata = { + metadata: { + identifier: string; + title?: string | string[]; + date?: string; + publicdate?: string; + creator?: string | string[]; + uploader?: string; + collection?: string | string[]; + mediatype?: string; + description?: string | string[]; + }; + files: ArchiveOrgItemFile[]; +}; + +function firstString(v: unknown): string | undefined { + if (typeof v === "string") return v.trim() || undefined; + if (Array.isArray(v)) { + for (const x of v) if (typeof x === "string" && x.trim()) return x.trim(); + } + return undefined; +} + +function stringList(v: unknown): string[] { + if (typeof v === "string") return v.trim() ? [v.trim()] : []; + if (Array.isArray(v)) return v.filter((x): x is string => typeof x === "string" && !!x.trim()); + return []; +} + +// The metadata API answers `{}` (HTTP 200) for an identifier with no item, and +// a "dark" item has metadata but no files. Null for anything without both. +export function parseArchiveOrgItemMetadata(raw: unknown): ArchiveOrgItemMetadata | null { + if (!raw || typeof raw !== "object") return null; + const r = raw as { metadata?: unknown; files?: unknown }; + const m = r.metadata as Record<string, unknown> | undefined; + if (!m || typeof m !== "object" || typeof m.identifier !== "string") return null; + const files = Array.isArray(r.files) + ? r.files.filter( + (f): f is ArchiveOrgItemFile => + !!f && typeof f === "object" && typeof (f as { name?: unknown }).name === "string", + ) + : []; + return { metadata: m as ArchiveOrgItemMetadata["metadata"], files }; +} + +// The extensions archive.org serves media under that yt-dlp can download. +const MEDIA_EXTS = new Set([ + "mp4", "m4v", "mkv", "webm", "mov", "avi", "mpeg", "mpg", "ogv", "flv", "wmv", "3gp", "ts", + "mp3", "m4a", "ogg", "oga", "opus", "flac", "wav", "aac", "wma", +]); + +function extOf(name: string): string { + const m = /\.([A-Za-z0-9]{1,8})$/.exec(name); + return m ? m[1].toLowerCase() : ""; +} + +export function isMediaFileName(name: string): boolean { + return MEDIA_EXTS.has(extOf(name)); +} + +// The item's media files: ORIGINALS only (a derivative is archive.org's +// transcode of one, the same recording), in the item's own order. +export function listArchiveOrgMediaFiles(item: ArchiveOrgItemMetadata): ArchiveOrgItemFile[] { + return item.files.filter((f) => f.source === "original" && isMediaFileName(f.name)); +} + +// ─── Picking files for a bulk import ─── + +export type ArchiveOrgFileSelection = + | { files: string[] } + | { match: string }; + +export type ArchiveOrgFilePick = { + // The chosen media files, in the item's order. + picked: string[]; + // Named in `files` but not a media original of the item. + unknown: string[]; +}; + +// `files` names exact paths; `match` is a case-insensitive regex over each +// media file's path. Either way only media originals can be picked. +export function pickArchiveOrgFiles( + item: ArchiveOrgItemMetadata, + sel: ArchiveOrgFileSelection, +): ArchiveOrgFilePick { + const media = listArchiveOrgMediaFiles(item).map((f) => f.name); + if ("files" in sel) { + const want = new Set(sel.files); + const known = new Set(media); + return { + picked: media.filter((n) => want.has(n)), + unknown: sel.files.filter((n) => !known.has(n)), + }; + } + const re = new RegExp(sel.match, "i"); + return { picked: media.filter((n) => re.test(n)), unknown: [] }; +} + +// ─── The mirror's original ─── + +export type ArchiveOrgMirror = { + platform: "youtube"; + id: string; + url: string; + title?: string; + // YYYYMMDD + uploadDate?: string; + uploader?: string; + channelUrl?: string; + description?: string; + // Where the original's identity was read from: the uploaded info.json (and + // then every field above is the original's own), or only a name. + from: "info-json" | "identifier" | "file-name"; +}; + +// The info.json a mirroring tool uploaded beside a media file: the file's own +// stem + `.info.json`. For a single-media item, the item's only info.json. +export function findArchiveOrgInfoJson( + item: ArchiveOrgItemMetadata, + file: string | undefined, +): string | null { + const names = item.files.map((f) => f.name); + const infos = names.filter((n) => n.endsWith(".info.json")); + if (file) { + const stem = file.replace(/\.[A-Za-z0-9]{1,8}$/, ""); + if (infos.includes(`${stem}.info.json`)) return `${stem}.info.json`; + return null; + } + return infos.length === 1 ? infos[0] : null; +} + +function youtubeWatchUrl(id: string): string { + return `https://www.youtube.com/watch?v=${id}`; +} + +// The original, from an uploaded info.json when there is one (authoritative: +// the record yt-dlp wrote when it downloaded the upload), else from the names. +export function archiveOrgMirrorOf(opts: { + identifier: string; + file?: string; + infoJson?: unknown; +}): ArchiveOrgMirror | null { + const info = opts.infoJson as Record<string, unknown> | null | undefined; + if (info && typeof info === "object") { + const key = String(info.extractor_key ?? info.extractor ?? ""); + const id = typeof info.id === "string" ? info.id : ""; + if (/^youtube$/i.test(key) && /^[A-Za-z0-9_-]{11}$/.test(id)) { + const str = (k: string) => (typeof info[k] === "string" && (info[k] as string).trim() ? (info[k] as string) : undefined); + const date = str("upload_date"); + return { + platform: "youtube", + id, + url: youtubeWatchUrl(id), + title: str("title"), + uploadDate: date && /^\d{8}$/.test(date) ? date : undefined, + uploader: str("uploader") ?? str("channel"), + channelUrl: str("channel_url") ?? str("uploader_url"), + description: str("description"), + from: "info-json", + }; + } + } + const fromIdent = youtubeIdFromIdentifier(opts.identifier); + if (fromIdent) return { platform: "youtube", id: fromIdent, url: youtubeWatchUrl(fromIdent), from: "identifier" }; + const fromName = opts.file ? youtubeIdFromFileName(opts.file) : null; + if (fromName) return { platform: "youtube", id: fromName, url: youtubeWatchUrl(fromName), from: "file-name" }; + return null; +} + +// ─── The provenance sidecar ─── + +export type ArchiveOrgProvenance = { + version: number; + identifier: string; + // The file's path inside the item; absent for a whole item. + file?: string; + // The item's page, and the file's own page inside it. + itemUrl: string; + fileUrl?: string; + // The media file's bytes, when the record is one file. + downloadUrl?: string; + // The item's BitTorrent file, when the item lists one. + torrentUrl?: string; + item: { + title?: string; + // archive.org's `date` (the content date the uploader gave), and + // `publicdate` (when the item went up). + date?: string; + publicDate?: string; + creator?: string; + collections: string[]; + mediatype?: string; + }; + // The file's own title in the item, when it has one. + fileTitle?: string; + mirror?: ArchiveOrgMirror; + fetchedAt: string; +}; + +export function buildArchiveOrgProvenance(opts: { + ref: ArchiveOrgRef; + item: ArchiveOrgItemMetadata; + infoJson?: unknown; + fetchedAt: string; +}): ArchiveOrgProvenance { + const { ref, item } = opts; + const m = item.metadata; + const identifier = m.identifier || ref.identifier; + const fileEntry = ref.file ? item.files.find((f) => f.name === ref.file) : undefined; + const torrentName = `${identifier}_archive.torrent`; + const hasTorrent = item.files.some((f) => f.name === torrentName); + const mirror = archiveOrgMirrorOf({ identifier, file: ref.file, infoJson: opts.infoJson }); + // The uploader field is the uploading account's e-mail address; it is never + // kept. `creator` is the public credit. + const prov: ArchiveOrgProvenance = { + version: ARCHIVE_ORG_PROVENANCE_VERSION, + identifier, + ...(ref.file ? { file: ref.file } : {}), + itemUrl: archiveOrgDetailsUrl({ identifier }), + ...(ref.file + ? { + fileUrl: archiveOrgDetailsUrl({ identifier, file: ref.file }), + downloadUrl: archiveOrgDownloadUrl(identifier, ref.file), + } + : {}), + ...(hasTorrent ? { torrentUrl: archiveOrgTorrentUrl(identifier) } : {}), + item: { + title: firstString(m.title), + date: firstString(m.date), + publicDate: firstString(m.publicdate), + creator: firstString(m.creator), + collections: stringList(m.collection), + mediatype: firstString(m.mediatype), + }, + ...(fileEntry?.title?.trim() ? { fileTitle: fileEntry.title.trim() } : {}), + ...(mirror ? { mirror } : {}), + fetchedAt: opts.fetchedAt, + }; + return stripUndefinedDeep(prov); +} + +function stripUndefinedDeep<T>(v: T): T { + if (Array.isArray(v)) return v.map(stripUndefinedDeep) as T; + if (v && typeof v === "object") { + const out: Record<string, unknown> = {}; + for (const [k, x] of Object.entries(v)) if (x !== undefined) out[k] = stripUndefinedDeep(x); + return out as T; + } + return v; +} + +// The sidecar's shape check: a record or null. +export function coerceArchiveOrgProvenance(value: unknown): ArchiveOrgProvenance | null { + const v = value as Partial<ArchiveOrgProvenance> | null; + if ( + !v || + typeof v !== "object" || + typeof v.identifier !== "string" || + typeof v.itemUrl !== "string" || + typeof v.fetchedAt !== "string" || + !v.item || + typeof v.item !== "object" + ) { + return null; + } + const item = v.item as ArchiveOrgProvenance["item"]; + const mirror = v.mirror; + return { + ...(v as ArchiveOrgProvenance), + item: { ...item, collections: Array.isArray(item.collections) ? item.collections : [] }, + ...(mirror && (typeof mirror.id !== "string" || typeof mirror.url !== "string") + ? { mirror: undefined } + : {}), + }; +} + +// ─── What the record's metadata.info.json is corrected to ─── + +// yt-dlp describes an entry of a multi-file item with the ITEM's title and +// page, and a mirror with archive.org's upload date. The record says: +// +// webpage_url the page of what was imported (the file's page inside the +// item). NOT cosmetic: the snapshot renames every video dir to +// `extractVideoId(webpage_url)` (reconcileVideoDirs.ts), so an +// entry left with the item's page would be merged into the item. +// title the original's title (a mirror), else the file's own title, +// else its file name — never the item's for one file of many. +// upload_date the original's date (a mirror), with its timestamp dropped +// description the original's (a mirror), when it had one +// uploader the original's uploader, else the item's public credit — never +// the uploading account's e-mail address +// +// Returns only the keys that differ from `info`; empty when nothing does. +export function archiveOrgMetadataPatch( + prov: ArchiveOrgProvenance, + info: Record<string, unknown>, +): Record<string, unknown> { + const want: Record<string, unknown> = {}; + want.webpage_url = prov.fileUrl ?? prov.itemUrl; + const mirror = prov.mirror; + const fileName = prov.file ? (prov.file.split("/").pop() ?? prov.file).replace(/\.[A-Za-z0-9]{1,8}$/, "") : undefined; + const title = mirror?.title ?? (prov.file ? (prov.fileTitle ?? fileName) : undefined); + if (title) want.title = title; + if (mirror?.uploadDate) want.upload_date = mirror.uploadDate; + if (mirror?.description) want.description = mirror.description; + const uploader = mirror?.uploader ?? prov.item.creator; + const current = info.uploader; + if (uploader) want.uploader = uploader; + else if (typeof current === "string" && current.includes("@")) want.uploader = null; + const patch: Record<string, unknown> = {}; + for (const [k, v] of Object.entries(want)) { + if (JSON.stringify(info[k]) !== JSON.stringify(v)) patch[k] = v; + } + // A mirror's date replaces archive.org's; the timestamp yt-dlp derived from + // the item's publicdate would contradict it. + if (patch.upload_date !== undefined && typeof info.timestamp === "number") patch.timestamp = null; + return patch; +} + +// ─── The links a record derives ─── + +export type ExternalLink = { label: string; url: string }; + +// What a citation of an archive.org record links, beside its moment page: +// +// original where the recording was first published — the YouTube upload +// at the cited second for a mirror, else the archive.org page +// downloads where a reader can fetch the file to check it: the archive.org +// page (for a mirror, since `original` is YouTube there) and the +// item's torrent +// +// Derived from the record's `webpage_url` alone when there is no provenance +// (the torrent name is archive.org's fixed `<identifier>_archive.torrent`); +// the provenance adds the mirror. +export function archiveOrgCitationLinks(opts: { + webpageUrl: string | null | undefined; + provenance?: ArchiveOrgProvenance | null; + seconds?: number; +}): { original?: ExternalLink; downloads: ExternalLink[] } { + const prov = opts.provenance ?? null; + const ref = opts.webpageUrl ? parseArchiveOrgUrl(opts.webpageUrl) : null; + const identifier = prov?.identifier ?? ref?.identifier; + if (!identifier) return { downloads: [] }; + const page = prov?.fileUrl ?? prov?.itemUrl ?? archiveOrgDetailsUrl(ref!); + const torrent: ExternalLink = { + label: "torrent", + url: prov ? (prov.torrentUrl ?? archiveOrgTorrentUrl(identifier)) : archiveOrgTorrentUrl(identifier), + }; + const archive: ExternalLink = { label: "archive.org", url: page }; + const mirror = prov?.mirror; + if (mirror) { + const secs = Math.max(0, Math.floor(opts.seconds ?? 0)); + const u = new URL(mirror.url); + if (secs > 0) u.searchParams.set("t", `${secs}s`); + return { original: { label: "YouTube", url: u.toString() }, downloads: [archive, torrent] }; + } + return { original: archive, downloads: [torrent] }; +} + +// ─── Playing it ─── + +// Browser-playable containers, best first. archive.org transcodes most video +// originals to an h.264 mp4 derivative, which every browser plays — an .avi or +// .mpeg original does not. +const PLAYABLE_EXTS = ["mp4", "m4v", "webm", "m4a", "mp3", "ogg", "oga", "opus"]; + +type FormatLike = { url?: unknown; ext?: unknown; format_note?: unknown }; + +// The one file of a record a <video> element can play from archive.org, or +// undefined. Read from the record's yt-dlp `formats` (each an archive.org +// download URL; `format_note` is "original" or "derivative"): the playable +// extension ranked first wins, an original before a derivative of the same +// extension. A file record with no formats falls back to its own file. +export function archiveOrgPlayableUrl(meta: { + id?: string; + formats?: unknown; +}): string | undefined { + const formats = Array.isArray(meta.formats) ? (meta.formats as FormatLike[]) : []; + let best: { rank: number; url: string } | null = null; + for (const f of formats) { + if (typeof f?.url !== "string" || !/^https:\/\/archive\.org\/download\//.test(f.url)) continue; + const ext = typeof f.ext === "string" ? f.ext.toLowerCase() : extOf(f.url); + const i = PLAYABLE_EXTS.indexOf(ext); + if (i < 0) continue; + const rank = i * 2 + (f.format_note === "original" ? 0 : 1); + if (!best || rank < best.rank) best = { rank, url: f.url }; + } + if (best) return best.url; + const id = meta.id ?? ""; + const slash = id.indexOf("/"); + if (slash > 0) { + const file = id.slice(slash + 1); + if (PLAYABLE_EXTS.includes(extOf(file))) return archiveOrgDownloadUrl(id.slice(0, slash), file); + } + return undefined; +} diff --git a/common/lib/archiveOrgClient.test.ts b/common/lib/archiveOrgClient.test.ts @@ -0,0 +1,131 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + ARCHIVE_ORG_USER_AGENT, + ArchiveOrgClient, + ArchiveOrgRequestError, + parseRetryAfterMs, +} from "./archiveOrgClient"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrgClient.test.ts +// +// THE POLITENESS RULES, ON A FAKE CLOCK. No request leaves the process: the +// fetch is a script of canned responses, the clock only moves when the client +// sleeps (or when a "request" takes time), and every sleep is recorded. + +type Scripted = { status: number; body?: unknown; headers?: Record<string, string>; takesMs?: number }; + +function harness(script: Scripted[], opts: ConstructorParameters<typeof ArchiveOrgClient>[1] = {}) { + let now = 1_000_000; + const sleeps: number[] = []; + const calls: { url: string; at: number; ua: string | null }[] = []; + const client = new ArchiveOrgClient( + { + now: () => now, + sleep: async (ms) => { + sleeps.push(ms); + now += ms; + }, + random: () => 0.5, // no jitter + fetch: async (url, init) => { + const headers = new Headers(init.headers); + calls.push({ url, at: now, ua: headers.get("user-agent") }); + const next = script.shift(); + if (!next) throw new Error("script exhausted"); + now += next.takesMs ?? 100; + return new Response(next.body === undefined ? "{}" : JSON.stringify(next.body), { + status: next.status, + headers: next.headers, + }); + }, + }, + opts, + ); + return { client, sleeps, calls, advance: (ms: number) => (now += ms) }; +} + +const META = { metadata: { identifier: "example-item", title: "Example" }, files: [{ name: "a.mp4", source: "original" }] }; + +test("an item's metadata is asked for once, identified, and cached", async () => { + const h = harness([{ status: 200, body: META }]); + const a = await h.client.itemMetadata("example-item"); + const b = await h.client.itemMetadata("example-item"); + assert.equal(a, b); + assert.equal(h.calls.length, 1); + assert.equal(h.calls[0].url, "https://archive.org/metadata/example-item"); + assert.equal(h.calls[0].ua, ARCHIVE_ORG_USER_AGENT); + assert.match(ARCHIVE_ORG_USER_AGENT, /\(\+https?:\/\//); +}); + +test("the cache expires", async () => { + const h = harness([{ status: 200, body: META }, { status: 200, body: META }], { cacheTtlMs: 1000 }); + await h.client.itemMetadata("example-item"); + h.advance(500); + await h.client.itemMetadata("example-item"); + assert.equal(h.calls.length, 1); + h.advance(1000); + await h.client.itemMetadata("example-item"); + assert.equal(h.calls.length, 2); +}); + +test("requests are one at a time with a gap between them", async () => { + const h = harness( + [ + { status: 200, body: { n: 1 }, takesMs: 500 }, + { status: 200, body: { n: 2 }, takesMs: 500 }, + { status: 200, body: { n: 3 }, takesMs: 500 }, + ], + { minGapMs: 2000 }, + ); + // Started together; served in order, each after the previous ended + the gap. + const results = await Promise.all([ + h.client.getJson("https://archive.org/download/i/1.json"), + h.client.getJson("https://archive.org/download/i/2.json"), + h.client.getJson("https://archive.org/download/i/3.json"), + ]); + assert.deepEqual(results, [{ n: 1 }, { n: 2 }, { n: 3 }]); + assert.equal(h.calls[1].at - h.calls[0].at, 2500); + assert.equal(h.calls[2].at - h.calls[1].at, 2500); +}); + +test("a 429 waits for Retry-After, then succeeds", async () => { + const h = harness([ + { status: 429, headers: { "retry-after": "30" } }, + { status: 200, body: META }, + ]); + await h.client.itemMetadata("example-item"); + assert.equal(h.calls.length, 2); + assert.ok(h.sleeps.includes(30_000), `slept ${h.sleeps}`); +}); + +test("a 503 without Retry-After backs off exponentially, and stops after maxAttempts", async () => { + const h = harness( + [{ status: 503 }, { status: 503 }, { status: 503 }, { status: 503 }, { status: 200, body: META }], + { maxAttempts: 4, baseBackoffMs: 5000, minGapMs: 0 }, + ); + await assert.rejects(h.client.itemMetadata("example-item"), (err: unknown) => { + assert.ok(err instanceof ArchiveOrgRequestError); + assert.equal(err.rateLimited, true); + assert.match(err.message, /after 4 attempts/); + return true; + }); + assert.equal(h.calls.length, 4); + // 5 s, 10 s, 20 s between the four attempts; none after the last. + assert.deepEqual(h.sleeps, [5000, 10000, 20000]); +}); + +test("a 404 is not retried; an empty answer is no item", async () => { + const h = harness([{ status: 404 }]); + await assert.rejects(h.client.getJson("https://archive.org/download/i/x.json"), /HTTP 404/); + assert.equal(h.calls.length, 1); + const h2 = harness([{ status: 200, body: {} }]); + await assert.rejects(h2.client.itemMetadata("no-such-item"), /has no item "no-such-item"/); +}); + +test("Retry-After as seconds or as an HTTP date", () => { + assert.equal(parseRetryAfterMs("12", 0), 12_000); + assert.equal(parseRetryAfterMs(new Date(60_000).toUTCString(), 0), 60_000); + assert.equal(parseRetryAfterMs(null, 0), null); + assert.equal(parseRetryAfterMs("soon", 0), null); +}); diff --git a/common/lib/archiveOrgClient.ts b/common/lib/archiveOrgClient.ts @@ -0,0 +1,196 @@ +// THE POLITE archive.org CLIENT — every request this app makes to archive.org +// that is not a yt-dlp spawn (the item metadata API, a mirror's info.json). +// +// archive.org is a non-profit serving files off its own disks; the rules here +// are the operator's "be polite to archive.org", made mechanical: +// +// ONE AT A TIME requests are chained: a second caller waits for the first +// to finish, and then for the gap. +// A GAP at least `minGapMs` (2 s) between the end of one request +// and the start of the next, from this process. +// IDENTIFIED a User-Agent naming the project and its URL. +// BACKS OFF a 429 or 503 (and 502/504, and a dropped connection) is +// retried after the server's Retry-After when it sends one, +// else after an exponential wait (5 s, 10 s, 20 s… capped at +// 2 min, ±20 % jitter), at most `maxAttempts` times in all; +// then it stops with ArchiveOrgRequestError, `rateLimited`. +// ASKS ONCE an item's metadata is cached in memory for `cacheTtlMs` +// (6 h): a bulk import of 161 files of one item asks for it +// once, and so does every per-file provenance step after it. +// +// The downloads themselves are yt-dlp's (one plain HTTP stream per file, paced +// by PLATFORM_ARGS.archiveorg) and run on the `platform:archiveorg` job queue, +// one at a time. + +import { PROJECT_NAME, PROJECT_URL } from "./project"; +import { + parseArchiveOrgItemMetadata, + type ArchiveOrgItemMetadata, +} from "./archiveOrg"; +import { archiveOrgDownloadUrl, archiveOrgMetadataUrl } from "./archiveOrgId"; + +export const ARCHIVE_ORG_USER_AGENT = `${PROJECT_NAME} archive.org import (+${PROJECT_URL})`; + +export class ArchiveOrgRequestError extends Error { + readonly status: number | null; + readonly rateLimited: boolean; + constructor(message: string, status: number | null, rateLimited: boolean) { + super(message); + this.name = "ArchiveOrgRequestError"; + this.status = status; + this.rateLimited = rateLimited; + } +} + +export type ArchiveOrgClientDeps = { + fetch?: (url: string, init: RequestInit) => Promise<Response>; + now?: () => number; + sleep?: (ms: number, signal?: AbortSignal) => Promise<void>; + random?: () => number; +}; + +export type ArchiveOrgClientOpts = { + minGapMs?: number; + maxAttempts?: number; + baseBackoffMs?: number; + maxBackoffMs?: number; + cacheTtlMs?: number; + timeoutMs?: number; +}; + +const RETRYABLE = new Set([429, 502, 503, 504]); + +function defaultSleep(ms: number, signal?: AbortSignal): Promise<void> { + return new Promise((resolve, reject) => { + if (signal?.aborted) return reject(signal.reason ?? new Error("aborted")); + const t = setTimeout(() => { + signal?.removeEventListener("abort", onAbort); + resolve(); + }, ms); + const onAbort = () => { + clearTimeout(t); + reject(signal?.reason ?? new Error("aborted")); + }; + signal?.addEventListener("abort", onAbort, { once: true }); + }); +} + +// Retry-After: delta-seconds or an HTTP date. Null when absent or unreadable. +export function parseRetryAfterMs(value: string | null, nowMs: number): number | null { + if (!value) return null; + const v = value.trim(); + if (/^\d+$/.test(v)) return Number(v) * 1000; + const at = Date.parse(v); + if (Number.isFinite(at)) return Math.max(0, at - nowMs); + return null; +} + +export class ArchiveOrgClient { + private readonly deps: Required<ArchiveOrgClientDeps>; + private readonly opts: Required<ArchiveOrgClientOpts>; + private chain: Promise<unknown> = Promise.resolve(); + private lastDoneAt = -Infinity; + private readonly cache = new Map<string, { at: number; value: ArchiveOrgItemMetadata }>(); + // Requests actually sent (each attempt), for tests and logs. + requests = 0; + + constructor(deps: ArchiveOrgClientDeps = {}, opts: ArchiveOrgClientOpts = {}) { + this.deps = { + fetch: deps.fetch ?? ((url, init) => fetch(url, init)), + now: deps.now ?? (() => Date.now()), + sleep: deps.sleep ?? defaultSleep, + random: deps.random ?? Math.random, + }; + this.opts = { + minGapMs: opts.minGapMs ?? 2_000, + maxAttempts: opts.maxAttempts ?? 4, + baseBackoffMs: opts.baseBackoffMs ?? 5_000, + maxBackoffMs: opts.maxBackoffMs ?? 120_000, + cacheTtlMs: opts.cacheTtlMs ?? 6 * 60 * 60 * 1000, + timeoutMs: opts.timeoutMs ?? 60_000, + }; + } + + // Run `fn` after every earlier request (and its gap) has finished. + private serial<T>(fn: () => Promise<T>): Promise<T> { + const run = this.chain.then(fn, fn); + this.chain = run.catch(() => {}); + return run; + } + + private backoffMs(attempt: number): number { + const exp = Math.min(this.opts.maxBackoffMs, this.opts.baseBackoffMs * 2 ** (attempt - 1)); + const jitter = 1 + (this.deps.random() * 0.4 - 0.2); + return Math.round(exp * jitter); + } + + // One GET, with the gap before it and the retry policy around it. + private async getOnce(url: string, accept: string, signal?: AbortSignal): Promise<Response> { + let lastError = ""; + for (let attempt = 1; attempt <= this.opts.maxAttempts; attempt++) { + const wait = this.lastDoneAt + this.opts.minGapMs - this.deps.now(); + if (wait > 0) await this.deps.sleep(wait, signal); + this.requests++; + let res: Response | null = null; + try { + const timeout = AbortSignal.timeout(this.opts.timeoutMs); + res = await this.deps.fetch(url, { + signal: signal ? AbortSignal.any([signal, timeout]) : timeout, + redirect: "follow", + headers: { accept, "user-agent": ARCHIVE_ORG_USER_AGENT }, + }); + } catch (err) { + if (signal?.aborted) throw err; + lastError = (err as Error).message; + } finally { + this.lastDoneAt = this.deps.now(); + } + if (res && res.ok) return res; + if (res && !RETRYABLE.has(res.status)) { + throw new ArchiveOrgRequestError( + `archive.org answered HTTP ${res.status} for ${url}`, + res.status, + false, + ); + } + if (res) lastError = `HTTP ${res.status}`; + if (attempt === this.opts.maxAttempts) break; + const retryAfter = res ? parseRetryAfterMs(res.headers.get("retry-after"), this.deps.now()) : null; + const delay = Math.min(this.opts.maxBackoffMs, retryAfter ?? this.backoffMs(attempt)); + await this.deps.sleep(delay, signal); + } + throw new ArchiveOrgRequestError( + `archive.org did not answer ${url} after ${this.opts.maxAttempts} attempts (${lastError}); stopping`, + null, + true, + ); + } + + getJson(url: string, signal?: AbortSignal): Promise<unknown> { + return this.serial(async () => { + const res = await this.getOnce(url, "application/json", signal); + return res.json(); + }); + } + + // The item's metadata, once per `cacheTtlMs`. + async itemMetadata(identifier: string, signal?: AbortSignal): Promise<ArchiveOrgItemMetadata> { + const hit = this.cache.get(identifier); + if (hit && this.deps.now() - hit.at < this.opts.cacheTtlMs) return hit.value; + const raw = await this.getJson(archiveOrgMetadataUrl(identifier), signal); + const item = parseArchiveOrgItemMetadata(raw); + if (!item) { + throw new ArchiveOrgRequestError(`archive.org has no item "${identifier}"`, 404, false); + } + this.cache.set(identifier, { at: this.deps.now(), value: item }); + return item; + } + + // A small JSON file inside an item (a mirror's `.info.json`). + itemJsonFile(identifier: string, file: string, signal?: AbortSignal): Promise<unknown> { + return this.getJson(archiveOrgDownloadUrl(identifier, file), signal); + } +} + +// THE process's client: one chain, one gap, one cache for every caller. +export const archiveOrgClient = new ArchiveOrgClient(); diff --git a/common/lib/archiveOrgId.test.ts b/common/lib/archiveOrgId.test.ts @@ -0,0 +1,95 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + archiveOrgDetailsUrl, + archiveOrgDownloadUrl, + archiveOrgTorrentUrl, + archiveOrgVideoId, + archiveOrgVideoIdFromNativeId, + parseArchiveOrgUrl, + youtubeIdFromFileName, + youtubeIdFromIdentifier, +} from "./archiveOrgId"; +import { extractVideoId } from "./videoId"; +import { defaultWebpageUrl, detectPlatform, queueKeyForUrl } from "./platform"; +import { dataDirIdForUrl } from "../ytdlp/runYtdlp"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrgId.test.ts +// +// Every identifier, file and YouTube id here is invented. + +const ITEM = "example-item"; +const FILE = "Example Talk (Part 1)-AbC123xyz_9.mp4"; + +test("archive.org item hosts are detected; the Wayback Machine is not", () => { + assert.equal(detectPlatform(`https://archive.org/details/${ITEM}`), "archiveorg"); + assert.equal(detectPlatform(`https://www.archive.org/details/${ITEM}`), "archiveorg"); + assert.equal(detectPlatform("https://web.archive.org/web/2020/https://example.com/"), null); + assert.equal(queueKeyForUrl(`https://archive.org/details/${ITEM}`), "platform:archiveorg"); +}); + +test("a whole item's id is its identifier, from details, embed and download URLs", () => { + for (const url of [ + `https://archive.org/details/${ITEM}`, + `https://archive.org/details/${ITEM}/`, + `https://archive.org/details/${ITEM}?autoplay=1`, + `https://archive.org/embed/${ITEM}`, + `https://archive.org/download/${ITEM}`, + ]) { + assert.equal(extractVideoId(url), ITEM, url); + } +}); + +test("a file inside an item gets a stable, filesystem-safe, unique id", () => { + const url = archiveOrgDetailsUrl({ identifier: ITEM, file: FILE }); + assert.equal(url, `https://archive.org/details/${ITEM}/Example%20Talk%20(Part%201)-AbC123xyz_9.mp4`); + const id = extractVideoId(url)!; + assert.match(id, /^example-item__Example-Talk-Part-1-AbC123xyz_9-[0-9a-f]{8}$/); + // The same file by every URL form, encoded or not (yt-dlp unquotes `+`). + assert.equal(extractVideoId(`https://archive.org/embed/${ITEM}/${encodeURIComponent(FILE)}`), id); + assert.equal(extractVideoId(archiveOrgDownloadUrl(ITEM, FILE)), id); + assert.equal( + extractVideoId(`https://archive.org/details/${ITEM}/Example+Talk+(Part+1)-AbC123xyz_9.mp4`), + id, + ); + // yt-dlp's native id for the entry maps to the same id. + assert.equal(archiveOrgVideoIdFromNativeId(`${ITEM}/${FILE}`), id); + assert.equal(archiveOrgVideoIdFromNativeId(ITEM), ITEM); + // The output path pins to it (a safe directory name). + assert.equal(dataDirIdForUrl(url), id); + // Paths that slug alike still differ. + const a = archiveOrgVideoId({ identifier: ITEM, file: "a b.mp4" }); + const b = archiveOrgVideoId({ identifier: ITEM, file: "a_b.mp4" }); + const c = archiveOrgVideoId({ identifier: ITEM, file: "a b.mkv" }); + assert.equal(new Set([a, b, c]).size, 3); + // Sub-directories are part of the path. + assert.equal( + parseArchiveOrgUrl(`https://archive.org/details/${ITEM}/disc1/01%20Intro.mp3`)?.file, + "disc1/01 Intro.mp3", + ); +}); + +test("URLs that name no item have no id", () => { + assert.equal(parseArchiveOrgUrl("https://archive.org/search?query=x"), null); + assert.equal(parseArchiveOrgUrl("https://archive.org/details/"), null); + assert.equal(extractVideoId("https://archive.org/search?query=x"), null); +}); + +test("the torrent and page URLs", () => { + assert.equal(archiveOrgTorrentUrl(ITEM), `https://archive.org/download/${ITEM}/${ITEM}_archive.torrent`); + assert.equal(defaultWebpageUrl("archiveorg", ITEM), `https://archive.org/details/${ITEM}`); + const fileId = archiveOrgVideoId({ identifier: ITEM, file: FILE }); + assert.equal(defaultWebpageUrl("archiveorg", fileId), `https://archive.org/details/${ITEM}`); +}); + +test("YouTube ids read from mirror names", () => { + assert.equal(youtubeIdFromIdentifier("youtube-AbC123xyz_9"), "AbC123xyz_9"); + assert.equal(youtubeIdFromIdentifier(ITEM), null); + assert.equal(youtubeIdFromFileName(FILE), "AbC123xyz_9"); + assert.equal(youtubeIdFromFileName("Some Title [Zz9-Qq8_Ww7].webm"), "Zz9-Qq8_Ww7"); + assert.equal(youtubeIdFromFileName("dir/Some Title [Zz9-Qq8_Ww7].info.json"), "Zz9-Qq8_Ww7"); + // An eleven-letter lowercase word is not an id. + assert.equal(youtubeIdFromFileName("interview-performance.mp4"), null); + assert.equal(youtubeIdFromFileName("plain.mp4"), null); +}); diff --git a/common/lib/archiveOrgId.ts b/common/lib/archiveOrgId.ts @@ -0,0 +1,171 @@ +// archive.org ITEM AND FILE IDENTITY — what an archive.org URL names, and the +// canonical video id (the `data/<id>/` dir name) it maps to. +// +// A leaf with no imports, like lib/videoId.ts (which calls into it): the roster +// store, the browser and the export build all canonicalize URLs through here. +// +// AN ITEM is archive.org's unit of upload: `https://archive.org/details/<identifier>`. +// An identifier is ASCII letters, digits, `.`, `-` and `_` — already a safe +// directory name, so a WHOLE ITEM (one that holds a single media file) has the +// identifier itself as its video id. +// +// A FILE INSIDE AN ITEM (a channel-archive item can hold a hundred and sixty +// videos) is `https://archive.org/details/<identifier>/<file path>` — the form +// yt-dlp's ArchiveOrg extractor resolves to that one entry. A file path is free +// text (spaces, brackets, unicode, sub-directories), so its id is +// +// <identifier>__<slug of the path without its extension>-<8 hex> +// +// where the slug keeps `[A-Za-z0-9_-]` and folds every other run into one `-` +// (cut to 48 characters), and the hex is a 32-bit FNV-1a hash of the EXACT file +// path. The slug keeps the id readable; the hash keeps it unique — two paths +// that slug alike ("a b.mp4", "a_b.mp4", "a b.mkv") still differ. Stable: the +// same path always gives the same id, so a re-import lands in the same dir. +// +// `/embed/` and `/download/` URLs name the same things and canonicalize to the +// same ids. `web.archive.org` (the Wayback Machine) is NOT an item host and is +// not handled here. + +export type ArchiveOrgRef = { + identifier: string; + // The file's path inside the item, decoded; absent for the whole item. + file?: string; +}; + +const IDENTIFIER_RE = /^[A-Za-z0-9][A-Za-z0-9._-]*$/; + +// Hosts that serve items. Not `web.archive.org` (Wayback) and not the +// `iaNNNNNN.us.archive.org` storage nodes a download redirects to. +export function isArchiveOrgItemHost(host: string): boolean { + const h = host.toLowerCase(); + return h === "archive.org" || h === "www.archive.org"; +} + +// yt-dlp reads the path with `unquote_plus`; so do we, so the file we name is +// the file it resolves. +function unquotePlus(s: string): string { + try { + return decodeURIComponent(s.replace(/\+/g, " ")); + } catch { + return s; + } +} + +// The item (and file) an archive.org URL names, or null for anything else. +export function parseArchiveOrgUrl(url: string): ArchiveOrgRef | null { + let u: URL; + try { + u = new URL(url); + } catch { + return null; + } + if (!isArchiveOrgItemHost(u.hostname)) return null; + const segs = u.pathname.split("/").filter(Boolean); + if (segs.length < 2) return null; + const kind = segs[0]; + if (kind !== "details" && kind !== "embed" && kind !== "download") return null; + const identifier = unquotePlus(segs[1]); + if (!IDENTIFIER_RE.test(identifier)) return null; + const rest = segs.slice(2).map(unquotePlus); + const file = rest.length > 0 ? rest.join("/") : undefined; + // A download URL with no file is the item's file listing, i.e. the item. + return file ? { identifier, file } : { identifier }; +} + +// FNV-1a, 32 bits, over the UTF-16 code units — dependency-free and the same in +// every runtime this module is imported into. +function fnv1a32(s: string): string { + let h = 0x811c9dc5; + for (let i = 0; i < s.length; i++) { + h ^= s.charCodeAt(i); + h = Math.imul(h, 0x01000193) >>> 0; + } + return h.toString(16).padStart(8, "0"); +} + +const SLUG_MAX = 48; + +function fileSlug(file: string): string { + const base = file.replace(/\.[A-Za-z0-9]{1,8}$/, ""); + const slug = base + .replace(/[^A-Za-z0-9_-]+/g, "-") + .replace(/-{2,}/g, "-") + .replace(/^-+|-+$/g, "") + .slice(0, SLUG_MAX) + .replace(/-+$/, ""); + return slug || "file"; +} + +// The canonical video id of an item, or of one file inside it. +export function archiveOrgVideoId(ref: ArchiveOrgRef): string { + if (!ref.file) return ref.identifier; + return `${ref.identifier}__${fileSlug(ref.file)}-${fnv1a32(ref.file)}`; +} + +// yt-dlp's own id for an archive.org record: the identifier, or +// `<identifier>/<file path>` for an entry of a multi-file item. The canonical id +// of that record, or null when it is not one. +export function archiveOrgVideoIdFromNativeId( + nativeId: string | null | undefined, +): string | null { + if (!nativeId) return null; + const slash = nativeId.indexOf("/"); + const identifier = slash < 0 ? nativeId : nativeId.slice(0, slash); + if (!IDENTIFIER_RE.test(identifier)) return null; + const file = slash < 0 ? "" : nativeId.slice(slash + 1); + return archiveOrgVideoId(file ? { identifier, file } : { identifier }); +} + +function encodePath(file: string): string { + return file.split("/").map(encodeURIComponent).join("/"); +} + +// The item page, or the file's own page inside it — the URL a record keeps as +// its `webpage_url`, and the one every link to it uses. +export function archiveOrgDetailsUrl(ref: ArchiveOrgRef): string { + const base = `https://archive.org/details/${ref.identifier}`; + return ref.file ? `${base}/${encodePath(ref.file)}` : base; +} + +// The file's bytes. +export function archiveOrgDownloadUrl(identifier: string, file: string): string { + return `https://archive.org/download/${identifier}/${encodePath(file)}`; +} + +// The item's BitTorrent file. archive.org derives one for every item, named +// `<identifier>_archive.torrent`; it covers every file in the item. +export function archiveOrgTorrentUrl(identifier: string): string { + return `https://archive.org/download/${identifier}/${identifier}_archive.torrent`; +} + +// The item's metadata API (one JSON document: the item's fields and its file +// list). +export function archiveOrgMetadataUrl(identifier: string): string { + return `https://archive.org/metadata/${identifier}`; +} + +// THE YOUTUBE ID A MIRROR CARRIES, when it says so in its name. +// +// `youtube-<id>` is the identifier tubeup (the usual YouTube → archive.org +// mirroring tool) gives an item; a file yt-dlp named carries the id as +// `<title>-<id>.<ext>` or `<title> [<id>].<ext>`. The bracketed form is +// unambiguous. The dashed one is a guess at the last eleven characters before +// the extension, so it is only taken when the token is not plain lowercase +// letters — a real id is random base64 and almost never is, while an English +// word of eleven letters ("performance") always is. +const YT_ID = "[A-Za-z0-9_-]{11}"; + +export function youtubeIdFromIdentifier(identifier: string): string | null { + const m = new RegExp(`^youtube-(${YT_ID})$`).exec(identifier); + return m ? m[1] : null; +} + +export function youtubeIdFromFileName(file: string): string | null { + const name = file.split("/").pop() ?? file; + const stem = name.replace(/(\.[A-Za-z0-9]{1,8})+$/, ""); + const bracket = new RegExp(`\\[(${YT_ID})\\]$`).exec(stem); + if (bracket) return bracket[1]; + const dashed = new RegExp(`(?:^|[-_ ])(${YT_ID})$`).exec(stem); + if (dashed && /[A-Z0-9_-]/.test(dashed[1])) return dashed[1]; + return null; +} diff --git a/common/lib/detectPlatform.mjs b/common/lib/detectPlatform.mjs @@ -24,6 +24,9 @@ export function detectPlatform(url) { if (host === "x.com" || host.endsWith(".x.com")) return "twitter"; if (host.endsWith("twitter.com")) return "twitter"; if (host === "bsky.app" || host.endsWith(".bsky.app")) return "bluesky"; + // archive.org ITEMS only. Not web.archive.org (the Wayback Machine's page + // captures are a different kind of record) — see lib/archiveOrgId.ts. + if (host === "archive.org" || host === "www.archive.org") return "archiveorg"; } catch { /* fall through */ } diff --git a/common/lib/metadataHistory.ts b/common/lib/metadataHistory.ts @@ -60,6 +60,10 @@ export const METADATA_HISTORY_WRITERS = [ // date, description and duration from the channel's RSS feed, for a record // imported by its enclosure URL, which carries none of them. "feed-backfill", + // An archive.org record corrected from its provenance (lib/archiveOrg.ts + // archiveOrgMetadataPatch): the file's page and title instead of the item's, + // and a mirror's original title, date and uploader. + "archiveorg-provenance", ] as const; export type MetadataHistoryWriter = (typeof METADATA_HISTORY_WRITERS)[number]; diff --git a/common/lib/momentUrl.ts b/common/lib/momentUrl.ts @@ -71,8 +71,10 @@ export function platformMomentUrl( case "twitch": u.searchParams.set("t", twitchTime(secs)); return u.toString(); - // Rumble / Kick / unknown: the watch page has no dependable start param — - // return the plain webpage URL rather than an invalid seek. + // Rumble / Kick / archive.org / unknown: the watch page has no dependable + // start param — return the plain webpage URL rather than an invalid seek. + // (archive.org's own player takes none we can rely on; the archive's + // viewer plays the file itself and seeks it — PlayerProvider.) default: return webpageUrl; } @@ -163,8 +165,9 @@ export function viewerMomentBaseUrl( // The platform base. Unlike platformMomentUrl (which falls back to the bare // webpage URL), a base MUST be appendable — so only platforms whose time param // takes raw seconds qualify. Twitch is excluded (its `t` takes an `XhYmZs` -// token, so appending an integer would be an invalid seek); Rumble/Kick/unknown -// have no dependable start param at all. Null in every non-appendable case. +// token, so appending an integer would be an invalid seek); Rumble/Kick/ +// archive.org/unknown have no dependable start param at all. Null in every +// non-appendable case. export function platformMomentBaseUrl( webpageUrl: string | null | undefined, platform: Platform | null | undefined, diff --git a/common/lib/platform.ts b/common/lib/platform.ts @@ -8,6 +8,9 @@ export type Platform = | "odysee" | "twitch" | "kick" + // archive.org items (lib/archiveOrgId.ts): a whole item, or one file inside + // a multi-file item. + | "archiveorg" | "twitter" | "bluesky"; @@ -17,6 +20,7 @@ export const PLATFORM_VALUES: ReadonlyArray<Platform> = [ "odysee", "twitch", "kick", + "archiveorg", "twitter", "bluesky", ]; @@ -49,6 +53,13 @@ export function defaultWebpageUrl(platform: Platform, id: string): string { if (platform === "twitch") return `https://www.twitch.tv/videos/${id}`; // Kick's canonical id is the VOD UUID; /video/<uuid> resolves to the VOD. if (platform === "kick") return `https://kick.com/video/${id}`; + // A whole item's id IS its identifier. A file's id is a slug + hash the file + // path cannot be recovered from, so this is the right page only for a whole + // item; every archived record carries its own `webpage_url`. + if (platform === "archiveorg") { + const file = /^(.+?)__.*-[0-9a-f]{8}$/.exec(id); + return `https://archive.org/details/${file ? file[1] : id}`; + } // Social posts: /i/status/<id> resolves without knowing the handle. Bluesky // has no handle-free permalink, so this is only a last-resort fallback — // every archived post carries its own canonical `url` (see postPermalink). diff --git a/common/lib/sidecar-server.test.ts b/common/lib/sidecar-server.test.ts @@ -6,6 +6,7 @@ import path from "node:path"; import { SIDECAR_FILENAMES, sidecar, sidecarField } from "./sidecar-server"; import { SUB_FILE_RE } from "./videoStatus"; // Every sidecar module, so the enumeration below sees every declaration. +import "./archiveOrg-server"; import "./attribution-server"; import "./diarization-server"; import "./digest-server"; @@ -41,10 +42,11 @@ async function scratch(): Promise<string> { return mkdtemp(path.join(os.tmpdir(), "sidecar-")); } -test("every declared sidecar filename escapes SUB_FILE_RE, and all ten are declared", () => { +test("every declared sidecar filename escapes SUB_FILE_RE, and all eleven are declared", () => { assert.deepEqual([...SIDECAR_FILENAMES].sort(), [ "ai-digest.json", "ai-digest.overrides.json", + "archiveorg.json", "attribution.json", "availability.json", "diarization.json", diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts @@ -1,7 +1,9 @@ import { readFile } from "node:fs/promises"; import path from "node:path"; import { formatDate, formatDuration } from "./format"; -import { defaultWebpageUrl } from "./platform"; +import { defaultWebpageUrl, detectPlatform } from "./platform"; +import { archiveOrgPlayableUrl } from "./archiveOrg"; +import { archiveOrgVideoIdFromNativeId } from "./archiveOrgId"; import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts"; import type { MediaType, VideoStat, VideoStatus } from "./stats"; import type { VideoState } from "./availability"; @@ -39,6 +41,10 @@ export type RawMetadata = { // yt-dlp's coarse kind: "video" | "livestream" | "short" (YouTube). Absent on // platforms that don't distinguish, where we fall back to is/was_live. media_type?: string; + // Every format the extractor offered. Read only for archive.org, where each + // is a plain download URL and one of them is what the player plays + // (lib/archiveOrg.ts archiveOrgPlayableUrl). + formats?: unknown; }; // Read and parse a video's metadata.info.json into the typed RawMetadata @@ -64,12 +70,25 @@ export function loadRawMetadataFromDir( return loadRawMetadata(path.join(videoDir, "metadata.info.json")); } +// The platform a record is from, by yt-dlp's extractor. archive.org's +// extractor is `ArchiveOrg` (key) / `archive.org` (name) — NOT web.archive.org's +// `YoutubeWebArchive`, which is a YouTube video. +// +// AN UNKNOWN EXTRACTOR: the record's own page decides when its host is one the +// app knows (lib/detectPlatform.mjs); otherwise "youtube", as it always has +// been — the generic extractor (a podcast episode imported by its enclosure +// URL) still lands there, and changing that would relabel records already +// published. export function platformFromMetadata(meta: RawMetadata): Platform { const key = meta.extractor_key ?? meta.extractor ?? ""; if (/^rumble/i.test(key)) return "rumble"; if (/^lbry/i.test(key)) return "odysee"; if (/^twitch/i.test(key)) return "twitch"; if (/^kick/i.test(key)) return "kick"; + if (/^archive\.?org$/i.test(key)) return "archiveorg"; + if (/^youtube/i.test(key)) return "youtube"; + const fromPage = detectPlatform(meta.webpage_url); + if (fromPage === "archiveorg") return fromPage; return "youtube"; } @@ -97,10 +116,14 @@ export function summarize( ): TranscriptSummary { const isLivestream = isLivestreamMetadata(meta); const platform = platformFromMetadata(meta); + // archive.org: yt-dlp's id for one file of an item is `<identifier>/<path>`, + // which is not a slug; the canonical id (lib/archiveOrgId.ts) is. const id = platform === "odysee" ? (meta.webpage_url_basename ?? meta.id ?? videoDir) - : (meta.id ?? videoDir); + : platform === "archiveorg" + ? (archiveOrgVideoIdFromNativeId(meta.id) ?? videoDir) + : (meta.id ?? videoDir); const dateFromDir = videoDir.match(/^(\d{8})(?:_|$)/)?.[1]; return { slug: `${channelSlug}/${id}`, @@ -118,6 +141,10 @@ export function summarize( webpageUrl: meta.webpage_url ?? defaultWebpageUrl(platform, id), // Kick VODs play from a persisted HLS manifest (no iframe embed exists). hlsUrl: platform === "kick" ? meta.manifest_url : undefined, + // archive.org plays the file itself in a native <video>, which seeks. + ...(platform === "archiveorg" + ? { mediaUrl: archiveOrgPlayableUrl(meta) } + : {}), }; } diff --git a/common/lib/transcripts.ts b/common/lib/transcripts.ts @@ -26,6 +26,11 @@ export type TranscriptSummary = { // HLS master-playlist URL for platforms with no iframe embed (Kick VODs). // Undefined for every other platform. Played via react-player/file + hls.js. hlsUrl?: string; + // A plain media file URL a native <video> plays and seeks (archive.org: the + // record's browser-playable file, lib/archiveOrg.ts). Undefined everywhere + // else. New with the archiveorg platform, so no cached or normalized record + // predates it and no cache version moved. + mediaUrl?: string; // Curated tag ids (common/lib/curatedTags.ts) — the operator's cross-channel // vocabulary, NOT the yt-dlp keywords in `tags` above. OMITTED when empty, so // an untagged corpus's pages stay byte-identical to the ones already on disk diff --git a/common/lib/videoId.ts b/common/lib/videoId.ts @@ -1,17 +1,27 @@ // The canonical video id derived from a video URL — the name every video's -// data/<id>/ dir carries, on every platform. Deliberately a leaf module with no -// imports at all: it lives here rather than in ytdlp/runYtdlp.ts (its original -// home, which still re-exports it) so that low-level stores like +// data/<id>/ dir carries, on every platform. Deliberately a leaf module whose +// one import (archiveOrgId.ts) is itself a leaf with no imports: it lives here +// rather than in ytdlp/runYtdlp.ts (its original home, which still re-exports it) so that low-level stores like // controller/rosterStore.ts can canonicalize a URL without pulling in execa, // the settings loader and the whole download pipeline. // // Canonical is NOT the same as yt-dlp's native extractor id: the two coincide // on YouTube and diverge everywhere else. See archiveIdForUrl in runYtdlp.ts // for the native-id resolution that reads metadata.info.json. + +import { isArchiveOrgItemHost, parseArchiveOrgUrl, archiveOrgVideoId } from "./archiveOrgId"; + export function extractVideoId(url: string): string | null { try { const u = new URL(url); const host = u.hostname.toLowerCase(); + if (isArchiveOrgItemHost(host)) { + // A whole item → its identifier; one file inside an item → a stable + // `<identifier>__<slug>-<hash>` (lib/archiveOrgId.ts). A URL that names + // no item (a search page, a collection listing) has no id. + const ref = parseArchiveOrgUrl(url); + return ref ? archiveOrgVideoId(ref) : null; + } if (host.endsWith("youtube.com") || host === "youtu.be") { const v = u.searchParams.get("v"); if (v) return v; diff --git a/common/ytdlp/channelArgs.test.ts b/common/ytdlp/channelArgs.test.ts @@ -147,3 +147,20 @@ test("the live pace (the shared state) reaches every channelExtraArgs call", () globalThis.__yttAutoQueueState__ = undefined; } }); + +test("archive.org is paced and backs off, and keeps the fetched page as webpage_url", () => { + const args = platformArgs("archiveorg"); + assert.equal(staticSleepRequestsSeconds("archiveorg"), 2); + assert.deepEqual(args.slice(0, 2), ["--sleep-requests", "2"]); + assert.ok(args.includes("http:exp=2:120")); + const i = args.indexOf("--parse-metadata"); + const [field, re] = [args[i + 1].slice(0, args[i + 1].indexOf(":")), args[i + 1].slice(args[i + 1].indexOf(":") + 1)]; + assert.equal(field, "original_url"); + // The regex (Python's syntax, which this subset shares) takes an archive.org + // page and nothing else. + const js = new RegExp(`^${re.replace("(?P<webpage_url>", "(?<webpage_url>")}$`); + assert.equal(js.exec("https://archive.org/details/example-item/a%20b.mp4")?.groups?.webpage_url, "https://archive.org/details/example-item/a%20b.mp4"); + assert.equal(js.exec("https://www.youtube.com/watch?v=AbC123xyz_9"), null); + // No parallel transfer is ever asked for. + assert.equal(args.some((a) => /concurrent|downloader|^-N$/.test(a)), false); +}); diff --git a/common/ytdlp/downloadFormat.test.ts b/common/ytdlp/downloadFormat.test.ts @@ -1,6 +1,7 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import { + ARCHIVE_ORG_AUTO_FORMAT_SELECTOR, DOWNLOAD_FORMAT_LABELS, DOWNLOAD_FORMAT_PRESETS, ORIGINAL_SOURCE_FORMAT_SELECTOR, @@ -102,3 +103,12 @@ test("sourceVideoQualityForMaxHeight: a cap at or under 720 is video_720, above assert.equal(sourceVideoQualityForMaxHeight(1080), "original"); assert.equal(sourceVideoQualityForMaxHeight(2160), "original"); }); + +test("auto on archive.org takes the uploader's original, an audio item's mp3", () => { + const sel = resolveDownloadFormatSelector("auto", "archiveorg"); + assert.equal(sel, ARCHIVE_ORG_AUTO_FORMAT_SELECTOR); + assert.match(sel, /^b\[format_note=original\]\[ext=mp4\]\//); + assert.ok(sel.split("/").includes("mp3")); + // An explicit preset still applies literally. + assert.equal(resolveDownloadFormatSelector("bestaudio", "archiveorg"), "bestaudio/worst"); +}); diff --git a/common/ytdlp/downloadFormat.ts b/common/ytdlp/downloadFormat.ts @@ -4,7 +4,9 @@ import type { Platform } from "../lib/platform"; // an enum rather than a free-form `-f` string so it validates like audioFormat). // "auto" is platform-aware: Odysee/LBRY only serves the full-length audio in its // `original` format (every HLS rung is CDN-truncated to a few minutes), so auto -// prefers `original` there and the historical `bestaudio/worst` everywhere else. +// prefers `original` there, archive.org gets the uploader's original file +// (ARCHIVE_ORG_AUTO_FORMAT_SELECTOR), and the historical `bestaudio/worst` +// everywhere else. export type DownloadFormatPreset = | "auto" | "original" @@ -126,12 +128,32 @@ export function resolveDownloadFormatSelector( ].join("/"); case "auto": default: + if (platform === "archiveorg") return ARCHIVE_ORG_AUTO_FORMAT_SELECTOR; return platform === "odysee" ? "original/bestaudio/worst" : "bestaudio/worst"; } } +// archive.org's "auto": the ORIGINAL file the uploader put up, in a common +// container, and for an audio item its MP3 (or Ogg) — not "bestaudio/worst". +// yt-dlp knows no codecs for an archive.org format (only its extension and +// `format_note`, "original" or "derivative"), so `bestaudio` never matches a +// video file and `worst` would take whichever transcode sorts last. A video +// original in another container (.avi, .mpeg) is the next rung; a FLAC/WAV +// original loses to its MP3 derivative, a fraction of the bytes for the same +// transcript. Anything at all is the last rung. +export const ARCHIVE_ORG_AUTO_FORMAT_SELECTOR = [ + "b[format_note=original][ext=mp4]", + "b[format_note=original][ext=mkv]", + "b[format_note=original][ext=webm]", + "b[format_note=original][ext=mp3]", + "mp3", + "ogg", + "b[format_note=original]", + "b", +].join("/"); + // The override chain mirrors audioFormat: per-run override beats the per-channel // default beats the global default; "auto" is the baseline when nothing is set. export function resolveDownloadFormatPreset(opts: { diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts @@ -64,6 +64,7 @@ import { type MetadataScanEntry, } from "../controller/metadataScanStore"; import { detectPlatform, type Platform } from "../lib/platform"; +import { ensureArchiveOrgProvenance } from "../lib/archiveOrg-server"; import { probeMediaDurationSec } from "./ffprobeDuration"; import { isShortAudio, @@ -184,6 +185,9 @@ export type ManagedDownloadOpts = { // "video_720" = the ≤720p H.264 selector (downloadFormat.ts). Never touches // the audio-only selector above. persistFormatPreset?: SourceVideoQuality; + // Test seam: the archive.org provenance step (lib/archiveOrg-server.ts). + // Every production caller passes none. + archiveOrgProvenance?: typeof ensureArchiveOrgProvenance; }; // When `reuseInfoJson` is true, the real download reuses the metadata the @@ -926,6 +930,22 @@ async function runManagedDownload( } const metaPath = path.join(videoDir, "metadata.info.json"); + // AN archive.org RECORD IS CORRECTED BEFORE ANYTHING READS IT: its + // provenance sidecar written (one cached metadata request per item), and + // metadata.info.json given the file's own page and title — yt-dlp writes + // the ITEM's for one file of a multi-file item (lib/archiveOrg-server.ts). + // Only after a prefetch that succeeded: a refused one is not a record. + if ( + detectPlatform(opts.videoUrl) === "archiveorg" && + attemptSucceeded(attempts.at(-1)?.ytdlpExitCode ?? null) + ) { + await (opts.archiveOrgProvenance ?? ensureArchiveOrgProvenance)({ + videoDir, + videoUrl: opts.videoUrl, + onLog: opts.onLog, + signal: opts.signal, + }); + } const metadata = await loadRawMetadata(metaPath); // Only wire --load-info-json into the real attempts when we actually have // the metadata file; a failed prefetch falls through to the legacy path so @@ -1240,6 +1260,16 @@ async function runManagedDownload( runAudioCheck, ) : await runAudioCheck(); + // The audio-checked pass re-extracted, so the archive.org correction above + // is re-applied (from the sidecar; no request). + if (canonicalId && detectPlatform(opts.videoUrl) === "archiveorg") { + await (opts.archiveOrgProvenance ?? ensureArchiveOrgProvenance)({ + videoDir: path.join(channelDir, "data", canonicalId), + videoUrl: opts.videoUrl, + onLog: opts.onLog, + signal: opts.signal, + }); + } primaryRes = { exitCode: audioOutcome.ytdlpExitCode, stderrTail: audioOutcome.stderrTail, diff --git a/common/ytdlp/platformArgs.mjs b/common/ytdlp/platformArgs.mjs @@ -34,10 +34,34 @@ import { detectPlatform } from "../lib/detectPlatform.mjs"; // (prefetch, subtitles, retries) back to back, and the two channels that 429'd // were the two most-downloaded. Mirrors rumble's pace; a channel's own // `ytdlpExtraArgs` still wins because it comes after. +// +// archiveorg: be polite to archive.org (a non-profit serving files from its +// own disks). `--sleep-requests 2` spaces the extractor's requests (the embed +// page, then the metadata API); `--retry-sleep` turns yt-dlp's immediate +// retries of a refused request or a dropped transfer into an exponential wait +// (2 s doubling to 120 s), and the stock retry counts bound how many. yt-dlp +// downloads an archive.org file as ONE plain HTTP stream (no fragments, no +// parallel ranges), and nothing here passes `-N` or an external downloader. +// `--parse-metadata` makes the record's `webpage_url` the URL it was fetched +// by: for one file of a multi-file item yt-dlp writes the ITEM's page there, +// and the snapshot renames every video dir to `extractVideoId(webpage_url)` +// (reconcileVideoDirs.ts) — the file's record would be merged into the item's. +// The import always fetches by the canonical file page, so the two agree; the +// regex only takes an archive.org URL, and leaves any other untouched. /** @type {Readonly<Partial<Record<Platform, readonly string[]>>>} */ export const PLATFORM_ARGS = Object.freeze({ rumble: Object.freeze(["--impersonate", "chrome", "--sleep-requests", "1"]), youtube: Object.freeze(["--sleep-requests", "1"]), + archiveorg: Object.freeze([ + "--sleep-requests", + "2", + "--retry-sleep", + "http:exp=2:120", + "--retry-sleep", + "extractor:exp=2:120", + "--parse-metadata", + "original_url:(?P<webpage_url>https://archive\\.org/(?:details|embed|download)/.+)", + ]), }); /**