commit d75d4da88abf351e00396bc3b8a0fd5034856715
parent c407383f03aa83336de341c60ccc399bc5a4ed57
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 15:32:33 -0400
Merge sources/archive-org (archive.org items and files as a video/audio source: polite import, provenance, file player, archive.org + torrent links on citations)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
# Conflicts:
# CHANNEL.md
# common/lib/channelConfig.ts
# common/lib/detectPlatform.mjs
# editor/CHANGELOG.md
# export/CHANGELOG.md
Diffstat:
46 files changed, 2936 insertions(+), 26 deletions(-)
diff --git a/CHANNEL.md b/CHANNEL.md
@@ -19,7 +19,7 @@ Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/f
| `postFetcher` | config | Social channels only: which social fetcher drives ingest (e.g. `"bluesky-atproto"`, `"x-gallery-dl"`). Absent = resolve by URL detection. Trimmed. |
| `socialHandle` | config | Social channels only: the bare account handle (a leading "@" is stripped). Derived from `url` at creation but stored, so a later URL-format change upstream cannot silently re-point ingest at a different account. |
| `postPagePauseSeconds` | config | Social channels that are read page by page (a forum thread) only: the pause between two page loads, in seconds; each pause is jittered to 0.85–1.65× of it. Absent = the fetcher's own (12 s, so 10–20 s); floored at 5, capped at 600. |
-| `platform` | config | The source platform (youtube, rumble, …). An unknown value is dropped. |
+| `platform` | config | The source platform: youtube, rumble, odysee, twitch, kick, archiveorg (archive.org items — imported, never listed; see README.md, "archive.org items"), twitter, bluesky or xenforo (a forum thread). An unknown value is dropped. |
| `name` | config | Display name. |
| `url` | config | The channel / playlist / account URL syncs enumerate. Absent = the channel is never auto-synced. |
| `audioFormat` | config | `"m4a"`, `"mp3"` or `"opus"`: the audio a transcribe-handling download keeps. |
diff --git a/README.md b/README.md
@@ -130,6 +130,42 @@ pnpm start:export # serve it at http://localhost:3000
**[SETUP.md](SETUP.md)** has the full per-OS instructions. Below is the short version.
+### archive.org items
+
+A channel can hold recordings from archive.org — a whole item
+(`https://archive.org/details/<identifier>`, when it holds one media file) or single
+files of a multi-file item (`…/details/<identifier>/<file>`, e.g. one video of a
+channel archive). They are imported, never listed, and transcribed like any
+transcribe channel:
+
+1. **Create the channel**: Platform *archive.org*, handling *Transcribe*, and no URL
+ (an archive.org channel is never auto-synced).
+2. **Import one recording**: the channel's *Import video* with the item or file URL,
+ or `pnpm ops import-video --json '{"slug":"<channel>","url":"https://archive.org/details/<identifier>"}'`.
+ An item with several media files is refused and its files are named.
+3. **Import chosen files of one item**:
+ `pnpm ops import-archive-org --json '{"slug":"<channel>","item":"<identifier>","files":["<file>",…]}'`
+ — or `"match": "<regex>"` over the item's file names; add `"dryRun": true` to see the
+ list first. One job, one file at a time.
+
+A whole item's video id is its identifier; a file's is `<identifier>__<slug>-<hash>`.
+Each record keeps `archiveorg.json` beside its metadata: the item's title, date,
+creator and collections, the item's torrent, and — for a mirror of a YouTube upload —
+the original's id, URL, title and upload date, read from the `.info.json` uploaded
+with it. The video page shows both; a citation of the record links **archive.org**
+and the **torrent** (and, for a mirror, the original **YouTube** upload at the cited
+second), so a reader can fetch the file and check it.
+
+**It is polite to archive.org**: its own job queue, one download at a time with a
+jittered pause of at least 8 s between files, the item's metadata asked once (cached),
+an identifying User-Agent, `Retry-After` and exponential backoff honoured, a stop
+after repeated failures, and a file on disk never fetched again
+(`common/lib/archiveOrgClient.ts`, `common/controller/archiveOrgImport.ts`).
+
+Wayback Machine captures (web.archive.org pages, WARC records) are a different kind of
+record and are not handled by this; they would be a source of their own beside
+`common/lib/archiveOrg.ts`.
+
### Requirements
Always needed, to install and run the apps:
diff --git a/common/components/FilePlayer.tsx b/common/components/FilePlayer.tsx
@@ -0,0 +1,94 @@
+"use client";
+
+import {
+ forwardRef,
+ useEffect,
+ useImperativeHandle,
+ useRef,
+} from "react";
+
+export type FilePlayerHandle = {
+ seekTo: (seconds: number, unit?: "seconds") => void;
+};
+
+type Props = {
+ url: string;
+ playing?: boolean;
+ onReady?: () => void;
+ onPlay?: () => void;
+ onPause?: () => void;
+ onProgress?: (state: { playedSeconds: number }) => void;
+ onError?: () => void;
+};
+
+// A plain media file in a native <video> — archive.org records, whose own
+// embed takes no start time we can rely on but whose files are served over
+// ranged HTTP, so the browser seeks them itself. The same imperative `seekTo`
+// handle and ready/play/pause/progress/error events as KickPlayer, so
+// PlayerProvider drives it the same way (full seeking + cue highlighting). An
+// audio file plays in the same element.
+const FilePlayer = forwardRef<FilePlayerHandle, Props>(function FilePlayer(
+ { url, playing, onReady, onPlay, onPause, onProgress, onError },
+ ref,
+) {
+ const videoRef = useRef<HTMLVideoElement | null>(null);
+ const cbRef = useRef({ onReady, onPlay, onPause, onProgress, onError });
+ cbRef.current = { onReady, onPlay, onPause, onProgress, onError };
+
+ useImperativeHandle(
+ ref,
+ () => ({
+ seekTo(seconds: number) {
+ const video = videoRef.current;
+ if (!video) return;
+ video.currentTime = Math.max(0, seconds);
+ void video.play().catch(() => {});
+ },
+ }),
+ [],
+ );
+
+ useEffect(() => {
+ const video = videoRef.current;
+ if (!video) return;
+ const emitReady = () => cbRef.current.onReady?.();
+ const emitPlay = () => cbRef.current.onPlay?.();
+ const emitPause = () => cbRef.current.onPause?.();
+ const emitError = () => cbRef.current.onError?.();
+ const emitProgress = () =>
+ cbRef.current.onProgress?.({ playedSeconds: video.currentTime });
+ video.addEventListener("loadedmetadata", emitReady, { once: true });
+ video.addEventListener("play", emitPlay);
+ video.addEventListener("pause", emitPause);
+ video.addEventListener("error", emitError);
+ video.addEventListener("timeupdate", emitProgress);
+ video.src = url;
+ return () => {
+ video.removeEventListener("loadedmetadata", emitReady);
+ video.removeEventListener("play", emitPlay);
+ video.removeEventListener("pause", emitPause);
+ video.removeEventListener("error", emitError);
+ video.removeEventListener("timeupdate", emitProgress);
+ };
+ }, [url]);
+
+ // Autoplay may be refused without a gesture; the native controls start it.
+ useEffect(() => {
+ const video = videoRef.current;
+ if (!video) return;
+ if (playing) void video.play().catch(() => {});
+ else video.pause();
+ }, [playing]);
+
+ return (
+ <video
+ ref={videoRef}
+ controls
+ playsInline
+ preload="metadata"
+ style={{ width: "100%", height: "100%", background: "black" }}
+ />
+ );
+});
+
+export default FilePlayer;
diff --git a/common/components/PlayerProvider.tsx b/common/components/PlayerProvider.tsx
@@ -30,6 +30,7 @@ import type { RumblePlayerHandle } from "./RumblePlayer";
import type { OdyseePlayerHandle } from "./OdyseePlayer";
import type { TwitchPlayerHandle } from "./TwitchPlayer";
import type { KickPlayerHandle } from "./KickPlayer";
+import type { FilePlayerHandle } from "./FilePlayer";
const ReactPlayer = dynamic(() => import("react-player/youtube"), {
ssr: false,
@@ -54,12 +55,19 @@ const KickPlayer = dynamic(() => import("./KickPlayer"), {
ssr: false,
});
+// archive.org records play their file in a native <video> (FilePlayer): the
+// embed takes no start time we can rely on, the file seeks.
+const FilePlayer = dynamic(() => import("./FilePlayer"), {
+ ssr: false,
+});
+
type PlayerHandle =
| Pick<ReactPlayerType, "seekTo">
| RumblePlayerHandle
| OdyseePlayerHandle
| TwitchPlayerHandle
- | KickPlayerHandle;
+ | KickPlayerHandle
+ | FilePlayerHandle;
export type Cue = { start: number; end: number; text: string };
@@ -77,6 +85,7 @@ export type TranscriptData = {
platform: Platform;
webpageUrl: string;
hlsUrl?: string;
+ mediaUrl?: string;
cues?: Cue[];
};
@@ -100,6 +109,7 @@ type Detail = {
platform: Platform;
webpageUrl: string;
hlsUrl?: string;
+ mediaUrl?: string;
cues?: Cue[];
};
@@ -387,6 +397,8 @@ export function PlayerProvider({
// Set when the Kick HLS manifest fails to load (usually an expired VOD), so
// the player swaps to the expiry/link fallback. Reset per active video below.
const [kickError, setKickError] = useState(false);
+ // The same for an archive.org file that will not load: swap to the link.
+ const [fileError, setFileError] = useState(false);
const playerRef = useRef<PlayerHandle | null>(null);
const pendingSeekRef = useRef<number | null>(null);
const readyForSlugRef = useRef<string | null>(null);
@@ -431,6 +443,7 @@ export function PlayerProvider({
platform: detail.platform,
webpageUrl: detail.webpageUrl,
hlsUrl: detail.hlsUrl,
+ mediaUrl: detail.mediaUrl,
cues: detail.cues,
};
}, [detail, detailMatches]);
@@ -511,7 +524,9 @@ export function PlayerProvider({
const copyShareUrl = useCallback(async (): Promise<boolean> => {
if (!data) return false;
const t =
- data.platform === "youtube" || data.platform === "kick"
+ data.platform === "youtube" ||
+ data.platform === "kick" ||
+ (data.platform === "archiveorg" && data.mediaUrl)
? currentTime
: (urlTime ?? 0);
const secs = Math.max(0, Math.floor(t));
@@ -636,6 +651,7 @@ export function PlayerProvider({
dispatch({ type: "DIGEST_RESET" });
digestInFlightForRef.current = null;
setKickError(false);
+ setFileError(false);
if (!urlSlug) return;
// Posts are not transcripts — never run the video fetch for one.
// A post slug always arrives together with a slug change, so this closure's
@@ -664,6 +680,7 @@ export function PlayerProvider({
platform: full.platform,
webpageUrl: full.webpageUrl,
hlsUrl: full.hlsUrl,
+ mediaUrl: full.mediaUrl,
cues: full.cues,
},
});
@@ -1020,6 +1037,22 @@ export function PlayerProvider({
}
onError={() => setKickError(true)}
/>
+ ) : data.platform === "archiveorg" && data.mediaUrl && !fileError ? (
+ <FilePlayer
+ key={data.id}
+ ref={(p: FilePlayerHandle | null) => {
+ playerRef.current = p;
+ }}
+ url={data.mediaUrl}
+ playing={playing}
+ onReady={handleReady}
+ onPlay={() => setPlaying(true)}
+ onPause={() => setPlaying(false)}
+ onProgress={(s: { playedSeconds: number }) =>
+ setCurrentTime(s.playedSeconds)
+ }
+ onError={() => setFileError(true)}
+ />
) : data.platform === "kick" ? (
// Kick VOD with no playable manifest (expired or pre-feature
// archive): show the expiry notice + a link to the source.
diff --git a/common/components/citations/CitationCard.tsx b/common/components/citations/CitationCard.tsx
@@ -139,7 +139,7 @@ function SpanLinks({ c }: { c: SpanCitationView }) {
<Play className="size-3 shrink-0" aria-hidden />
Play {spanLabel(c.start, c.end)}
</a>
- {c.record.originalUrl && <ExternalA href={c.record.originalUrl}>Original</ExternalA>}
+ <RecordLinks r={c.record} />
{c.record.corpusUrl && (
<a href={c.record.corpusUrl} className={linkClass}>
Transcript
@@ -149,6 +149,21 @@ function SpanLinks({ c }: { c: SpanCitationView }) {
);
}
+// The record's original, then where it can be downloaded to check it
+// ("YouTube · archive.org · torrent" for an archive.org mirror).
+function RecordLinks({ r }: { r: SpanCitationView["record"] }) {
+ return (
+ <>
+ {r.originalUrl && <ExternalA href={r.originalUrl}>{r.originalLabel ?? "Original"}</ExternalA>}
+ {r.downloads?.map((d) => (
+ <ExternalA key={d.url} href={d.url}>
+ {d.label}
+ </ExternalA>
+ ))}
+ </>
+ );
+}
+
function PostLinks({ c }: { c: PostCitationView }) {
return (
<>
diff --git a/common/controller/archiveOrgImport.test.ts b/common/controller/archiveOrgImport.test.ts
@@ -0,0 +1,223 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import type { DownloadOutcomeRecord } from "../lib/downloadOutcome";
+import type { ChannelConfig } from "../lib/channelConfig";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/archiveOrgImport.test.ts
+//
+// The bulk import with everything injected: a scripted archive.org, a fake
+// per-file download, a sleeper that only records. getPaths()/getSettings()
+// memoize, so the env is set before anything imports them. Every name here is
+// invented.
+
+const ROOT = mkdtempSync(path.join(os.tmpdir(), "archiveorg-import-"));
+process.env.TRANSCRIPTS_DIR = ROOT;
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+writeFileSync(
+ process.env.SETTINGS_FILE,
+ JSON.stringify({ minFreeDiskGB: 0, sleepBetweenDownloadsSeconds: 3 }) + "\n",
+);
+mkdirSync(path.join(ROOT, "channels", "c", "data"), { recursive: true });
+
+const {
+ ARCHIVE_ORG_MIN_GAP_SECONDS,
+ archiveOrgGapMs,
+ resolveArchiveOrgImportUrl,
+ runArchiveOrgImport,
+} = await import("./archiveOrgImport");
+const { ArchiveOrgClient } = await import("../lib/archiveOrgClient");
+const { getPaths } = await import("../lib/paths");
+const { archiveOrgVideoId } = await import("../lib/archiveOrgId");
+
+const ITEM = "example-item";
+const FILES = ["Alpha-AbC123xyz_9.mp4", "Beta-Def456uvw_8.mp4", "Gamma-Ghi789rst_7.mp4"];
+
+function client(meta: unknown) {
+ return new ArchiveOrgClient(
+ {
+ sleep: async () => {},
+ fetch: async () => new Response(JSON.stringify(meta), { status: 200 }),
+ },
+ { minGapMs: 0 },
+ );
+}
+
+const MULTI = {
+ metadata: { identifier: ITEM, title: "Example Archive" },
+ files: [
+ ...FILES.map((name) => ({ name, source: "original" })),
+ { name: "Alpha-AbC123xyz_9.info.json", source: "original" },
+ { name: "Alpha-AbC123xyz_9.ogv", source: "derivative", original: FILES[0] },
+ ],
+};
+const SINGLE = {
+ metadata: { identifier: "example-film", title: "Example Film" },
+ files: [{ name: "film.mp4", source: "original" }, { name: "film.ogv", source: "derivative" }],
+};
+
+const CONFIG: ChannelConfig = { handling: "transcribe", platform: "archiveorg" } as ChannelConfig;
+
+function outcome(status: DownloadOutcomeRecord["status"], failureClass?: DownloadOutcomeRecord["failureClass"]): DownloadOutcomeRecord {
+ return {
+ videoId: "x",
+ status,
+ startedAt: "2026-01-01T00:00:00.000Z",
+ finishedAt: "2026-01-01T00:00:01.000Z",
+ attempts: [],
+ ...(failureClass ? { failureClass } : {}),
+ } as DownloadOutcomeRecord;
+}
+
+test("one URL: an item with several media files is refused, naming the way to choose", async () => {
+ const r = await resolveArchiveOrgImportUrl(`https://archive.org/details/${ITEM}`, { client: client(MULTI) });
+ assert.equal(r.ok, false);
+ assert.match(!r.ok ? r.error : "", /holds 3 media files.*import-archive-org/s);
+});
+
+test("one URL: a file of many is that file; the only file of an item is the item", async () => {
+ const r = await resolveArchiveOrgImportUrl(
+ `https://archive.org/details/${ITEM}/${encodeURIComponent(FILES[1])}`,
+ { client: client(MULTI) },
+ );
+ assert.ok(r.ok);
+ assert.equal(r.ok && r.id, archiveOrgVideoId({ identifier: ITEM, file: FILES[1] }));
+ assert.equal(r.ok && r.url, `https://archive.org/details/${ITEM}/Beta-Def456uvw_8.mp4`);
+
+ const whole = await resolveArchiveOrgImportUrl("https://archive.org/embed/example-film", { client: client(SINGLE) });
+ assert.deepEqual(whole, { ok: true, url: "https://archive.org/details/example-film", id: "example-film", identifier: "example-film" });
+ const byFile = await resolveArchiveOrgImportUrl("https://archive.org/details/example-film/film.mp4", { client: client(SINGLE) });
+ assert.equal(byFile.ok && byFile.id, "example-film");
+
+ const missing = await resolveArchiveOrgImportUrl(`https://archive.org/details/${ITEM}/nope.mp4`, { client: client(MULTI) });
+ assert.equal(missing.ok, false);
+});
+
+test("the gap is the configured pause, floored, plus up to half again", () => {
+ assert.equal(archiveOrgGapMs(0, 0), ARCHIVE_ORG_MIN_GAP_SECONDS * 1000);
+ assert.equal(archiveOrgGapMs(3, 0), ARCHIVE_ORG_MIN_GAP_SECONDS * 1000);
+ assert.equal(archiveOrgGapMs(30, 0), 30_000);
+ assert.equal(archiveOrgGapMs(30, 1), 45_000);
+});
+
+test("bulk: chosen files one at a time, paced, skipping what is on disk", async () => {
+ const paths = getPaths();
+ // Beta is already downloaded (a transcript on disk).
+ const betaId = archiveOrgVideoId({ identifier: ITEM, file: FILES[1] });
+ mkdirSync(path.join(ROOT, "channels", "c", "data", betaId), { recursive: true });
+ writeFileSync(path.join(ROOT, "channels", "c", "data", betaId, "transcript.json"), "{}");
+
+ const urls: string[] = [];
+ const sleeps: number[] = [];
+ const imported: string[] = [];
+ const result = await runArchiveOrgImport({
+ paths,
+ slug: "c",
+ channelConfig: CONFIG,
+ identifier: ITEM,
+ selection: { match: "\\.mp4$" },
+ onLog: () => {},
+ signal: new AbortController().signal,
+ deps: {
+ client: client(MULTI),
+ downloadOne: async (o) => {
+ urls.push(o.videoUrl);
+ return outcome("ok");
+ },
+ sleep: async (ms) => {
+ sleeps.push(ms);
+ },
+ random: () => 0,
+ onImported: (id) => imported.push(id),
+ },
+ });
+ assert.deepEqual(urls, [
+ `https://archive.org/details/${ITEM}/Alpha-AbC123xyz_9.mp4`,
+ `https://archive.org/details/${ITEM}/Gamma-Ghi789rst_7.mp4`,
+ ]);
+ // One gap, between the two fetches — none before the first, none for the skip.
+ assert.deepEqual(sleeps, [ARCHIVE_ORG_MIN_GAP_SECONDS * 1000]);
+ assert.deepEqual(result.imported, [FILES[0], FILES[2]]);
+ assert.deepEqual(result.skipped, [FILES[1]]);
+ assert.equal(imported.length, 2);
+ assert.equal(result.stopped, undefined);
+});
+
+test("bulk: a rate-limited file stops the batch at once", async () => {
+ let n = 0;
+ const result = await runArchiveOrgImport({
+ paths: getPaths(),
+ slug: "c",
+ channelConfig: CONFIG,
+ identifier: ITEM,
+ selection: { files: [FILES[0], FILES[2]] },
+ onLog: () => {},
+ signal: new AbortController().signal,
+ deps: {
+ client: client(MULTI),
+ downloadOne: async () => {
+ n++;
+ return outcome("failed", "rate_limit");
+ },
+ sleep: async () => {},
+ },
+ });
+ assert.equal(n, 1);
+ assert.equal(result.rateLimited, true);
+ assert.match(result.stopped ?? "", /rate-limited/);
+});
+
+test("bulk: three failures in a row stop it; unknown names are reported", async () => {
+ const many = {
+ metadata: { identifier: ITEM },
+ files: ["a", "b", "c", "d", "e"].map((x) => ({ name: `${x}.mp3`, source: "original" })),
+ };
+ let n = 0;
+ const result = await runArchiveOrgImport({
+ paths: getPaths(),
+ slug: "c",
+ channelConfig: CONFIG,
+ identifier: ITEM,
+ selection: { files: ["a.mp3", "b.mp3", "c.mp3", "d.mp3", "zz.mp3"] },
+ onLog: () => {},
+ signal: new AbortController().signal,
+ deps: {
+ client: client(many),
+ downloadOne: async () => {
+ n++;
+ throw new Error("boom");
+ },
+ sleep: async () => {},
+ },
+ });
+ assert.equal(n, 3);
+ assert.equal(result.failed.length, 3);
+ assert.deepEqual(result.unknown, ["zz.mp3"]);
+ assert.match(result.stopped ?? "", /3 failures in a row/);
+});
+
+test("bulk: a dry run fetches nothing", async () => {
+ let n = 0;
+ const result = await runArchiveOrgImport({
+ paths: getPaths(),
+ slug: "c",
+ channelConfig: CONFIG,
+ identifier: ITEM,
+ selection: { match: "." },
+ onLog: () => {},
+ signal: new AbortController().signal,
+ dryRun: true,
+ deps: {
+ client: client(MULTI),
+ downloadOne: async () => {
+ n++;
+ return outcome("ok");
+ },
+ },
+ });
+ assert.equal(n, 0);
+ assert.equal(result.planned, 3);
+});
diff --git a/common/controller/archiveOrgImport.ts b/common/controller/archiveOrgImport.ts
@@ -0,0 +1,358 @@
+// IMPORTING FROM archive.org — one item, one file of an item, or a chosen set
+// of files of one item, into an existing channel.
+//
+// Every download is `downloadOneManaged` (the same managed path every other
+// import takes), fetched by the CANONICAL page of what is imported
+// (lib/archiveOrgId.ts): `https://archive.org/details/<identifier>` for an item
+// holding one media file, `…/details/<identifier>/<file>` for one file of
+// many. yt-dlp's ArchiveOrg extractor resolves either to exactly one record,
+// and the provenance step (lib/archiveOrg-server.ts) runs inside it.
+//
+// AN ITEM WITH SEVERAL MEDIA FILES IS NEVER IMPORTED WHOLE. yt-dlp would treat
+// it as a playlist and write every file into the one pinned `data/<id>/`; the
+// import refuses it and names the way to choose files instead.
+//
+// POLITE (the operator: "be polite to archive.org"):
+// - the item's metadata is asked for once (lib/archiveOrgClient.ts caches it)
+// and before any download, so a typo'd identifier costs one request;
+// - files go one at a time, on the `platform:archiveorg` queue, with a
+// jittered gap between them — the channel's (else the global)
+// `sleepBetweenDownloadsSeconds`, never under ARCHIVE_ORG_MIN_GAP_SECONDS,
+// plus up to half again at random;
+// - a file already downloaded is never fetched again;
+// - a rate-limited download stops the batch at once (the platform's own
+// backoff takes over), and ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES failures in
+// a row stop it too. Re-running the same command resumes: what landed is
+// skipped.
+
+import path from "node:path";
+import type { ChannelConfig } from "../lib/channelConfig";
+import type { Paths } from "../lib/paths";
+import type { DownloadOutcomeRecord } from "../lib/downloadOutcome";
+import {
+ listArchiveOrgMediaFiles,
+ pickArchiveOrgFiles,
+ type ArchiveOrgFileSelection,
+ type ArchiveOrgItemMetadata,
+} from "../lib/archiveOrg";
+import {
+ archiveOrgDetailsUrl,
+ archiveOrgVideoId,
+ parseArchiveOrgUrl,
+} from "../lib/archiveOrgId";
+import { archiveOrgClient, type ArchiveOrgClient } from "../lib/archiveOrgClient";
+import { getSettings } from "../lib/settings";
+import { diskGate } from "../lib/diskSpace";
+import { resolveCookiePolicy } from "../lib/cookiePolicy";
+import { downloadOneManaged, type ManagedDownloadOpts } from "../ytdlp/downloadOneManaged";
+import { destinationExists } from "../ytdlp/runYtdlp";
+import { mergeRosterFile } from "./rosterStore";
+
+export const ARCHIVE_ORG_MIN_GAP_SECONDS = 8;
+export const ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES = 3;
+
+// ─── One URL ───
+
+export type ResolvedArchiveOrgImport =
+ | { ok: true; url: string; id: string; identifier: string; file?: string }
+ | { ok: false; error: string };
+
+// What an archive.org URL imports as. An item URL for an item with ONE media
+// file is the item (id = identifier); a file URL is that file — unless the
+// item holds only that one file, when it is the item too, so the same
+// recording never gets two ids. An item URL for an item with several media
+// files is refused, naming them.
+export async function resolveArchiveOrgImportUrl(
+ url: string,
+ opts: { client?: ArchiveOrgClient; signal?: AbortSignal } = {},
+): Promise<ResolvedArchiveOrgImport> {
+ const ref = parseArchiveOrgUrl(url);
+ if (!ref) {
+ return {
+ ok: false,
+ error: `Not an archive.org item URL: ${url} (expected https://archive.org/details/<identifier>[/<file>])`,
+ };
+ }
+ const client = opts.client ?? archiveOrgClient;
+ let item: ArchiveOrgItemMetadata;
+ try {
+ item = await client.itemMetadata(ref.identifier, opts.signal);
+ } catch (err) {
+ return { ok: false, error: (err as Error).message };
+ }
+ const identifier = item.metadata.identifier || ref.identifier;
+ const media = listArchiveOrgMediaFiles(item).map((f) => f.name);
+ if (ref.file) {
+ if (!item.files.some((f) => f.name === ref.file)) {
+ return { ok: false, error: `archive.org item "${identifier}" has no file "${ref.file}"` };
+ }
+ if (media.length === 1 && media[0] === ref.file) {
+ return { ok: true, url: archiveOrgDetailsUrl({ identifier }), id: identifier, identifier };
+ }
+ const file = ref.file;
+ return {
+ ok: true,
+ url: archiveOrgDetailsUrl({ identifier, file }),
+ id: archiveOrgVideoId({ identifier, file }),
+ identifier,
+ file,
+ };
+ }
+ if (media.length === 0) {
+ return { ok: false, error: `archive.org item "${identifier}" has no media files to import` };
+ }
+ if (media.length > 1) {
+ const shown = media.slice(0, 5).map((n) => `"${n}"`).join(", ");
+ return {
+ ok: false,
+ error:
+ `archive.org item "${identifier}" holds ${media.length} media files (${shown}${media.length > 5 ? ", …" : ""}). ` +
+ `Import one by its file URL (https://archive.org/details/${identifier}/<file>), or several with ` +
+ `pnpm ops import-archive-org --json '{"slug":"<channel>","item":"${identifier}","files":[…]}' (or "match": "<regex>").`,
+ };
+ }
+ return { ok: true, url: archiveOrgDetailsUrl({ identifier }), id: identifier, identifier };
+}
+
+// ─── Many files of one item ───
+
+export type ArchiveOrgImportPlanEntry = {
+ file: string;
+ url: string;
+ id: string;
+ // Already downloaded: skipped, never fetched again.
+ onDisk: boolean;
+};
+
+export type ArchiveOrgImportPlan = {
+ identifier: string;
+ title?: string;
+ mediaFiles: number;
+ entries: ArchiveOrgImportPlanEntry[];
+ // Named in `files` but not a media original of the item.
+ unknown: string[];
+};
+
+export async function planArchiveOrgImport(opts: {
+ identifier: string;
+ selection: ArchiveOrgFileSelection;
+ dataDir: string;
+ handling: ChannelConfig["handling"];
+ client?: ArchiveOrgClient;
+ signal?: AbortSignal;
+}): Promise<ArchiveOrgImportPlan> {
+ const client = opts.client ?? archiveOrgClient;
+ const item = await client.itemMetadata(opts.identifier, opts.signal);
+ const identifier = item.metadata.identifier || opts.identifier;
+ const media = listArchiveOrgMediaFiles(item);
+ const { picked, unknown } = pickArchiveOrgFiles(item, opts.selection);
+ const entries: ArchiveOrgImportPlanEntry[] = [];
+ for (const file of picked) {
+ // An item of one media file imports as the item (resolveArchiveOrgImportUrl).
+ const whole = media.length === 1;
+ const id = whole ? identifier : archiveOrgVideoId({ identifier, file });
+ const url = whole ? archiveOrgDetailsUrl({ identifier }) : archiveOrgDetailsUrl({ identifier, file });
+ entries.push({
+ file,
+ url,
+ id,
+ onDisk: await destinationExists(opts.dataDir, id, opts.handling),
+ });
+ }
+ const title = item.metadata.title;
+ return {
+ identifier,
+ ...(typeof title === "string" ? { title } : {}),
+ mediaFiles: media.length,
+ entries,
+ unknown,
+ };
+}
+
+// The gap before the next file: the configured pause, floored, plus up to
+// half again at random so a batch never settles into a fixed beat.
+export function archiveOrgGapMs(sleepBetweenDownloadsSeconds: number, random: number): number {
+ const base = Math.max(ARCHIVE_ORG_MIN_GAP_SECONDS, sleepBetweenDownloadsSeconds || 0);
+ return Math.round(base * (1 + 0.5 * Math.min(1, Math.max(0, random))) * 1000);
+}
+
+function isOk(rec: DownloadOutcomeRecord): boolean {
+ return rec.status.startsWith("ok");
+}
+
+export type ArchiveOrgImportResult = {
+ identifier: string;
+ planned: number;
+ imported: string[];
+ skipped: string[];
+ failed: { file: string; error: string }[];
+ unknown: string[];
+ // Why the batch ended before its last file, when it did.
+ stopped?: string;
+ // archive.org refused a download as a rate limit: the caller backs the
+ // platform off (jobs/downloadBackoff.ts).
+ rateLimited?: boolean;
+};
+
+export type ArchiveOrgImportDeps = {
+ client?: ArchiveOrgClient;
+ downloadOne?: (opts: ManagedDownloadOpts) => Promise<DownloadOutcomeRecord>;
+ sleep?: (ms: number, signal: AbortSignal) => Promise<void>;
+ random?: () => number;
+ // After each imported file (the editor revalidates its pages).
+ onImported?: (id: string) => void;
+};
+
+function abortableSleep(ms: number, signal: AbortSignal): Promise<void> {
+ return new Promise((resolve) => {
+ if (signal.aborted) return resolve();
+ const t = setTimeout(done, ms);
+ function done() {
+ clearTimeout(t);
+ signal.removeEventListener("abort", done);
+ resolve();
+ }
+ signal.addEventListener("abort", done, { once: true });
+ });
+}
+
+export async function runArchiveOrgImport(opts: {
+ paths: Paths;
+ slug: string;
+ channelConfig: ChannelConfig;
+ identifier: string;
+ selection: ArchiveOrgFileSelection;
+ onLog: (line: string) => void;
+ signal: AbortSignal;
+ drainSignal?: AbortSignal;
+ dryRun?: boolean;
+ deps?: ArchiveOrgImportDeps;
+}): Promise<ArchiveOrgImportResult> {
+ const deps = opts.deps ?? {};
+ const downloadOne = deps.downloadOne ?? downloadOneManaged;
+ const sleep = deps.sleep ?? abortableSleep;
+ const random = deps.random ?? Math.random;
+ const settings = getSettings();
+ const dataDir = path.join(opts.paths.channelsDir, opts.slug, "data");
+ const log = opts.onLog;
+
+ const plan = await planArchiveOrgImport({
+ identifier: opts.identifier,
+ selection: opts.selection,
+ dataDir,
+ handling: opts.channelConfig.handling,
+ client: deps.client,
+ signal: opts.signal,
+ });
+ const result: ArchiveOrgImportResult = {
+ identifier: plan.identifier,
+ planned: plan.entries.length,
+ imported: [],
+ skipped: [],
+ failed: [],
+ unknown: plan.unknown,
+ };
+ log(
+ `archive.org item ${plan.identifier}${plan.title ? ` ("${plan.title}")` : ""}: ` +
+ `${plan.mediaFiles} media files, ${plan.entries.length} chosen, ` +
+ `${plan.entries.filter((e) => e.onDisk).length} already downloaded.\n`,
+ );
+ if (plan.unknown.length > 0) {
+ log(`Not media files of the item (ignored): ${plan.unknown.map((n) => JSON.stringify(n)).join(", ")}\n`);
+ }
+ if (opts.dryRun) {
+ for (const e of plan.entries) log(` ${e.onDisk ? "on disk " : "would get"} ${e.id} ${e.file}\n`);
+ result.skipped = plan.entries.filter((e) => e.onDisk).map((e) => e.file);
+ return result;
+ }
+
+ const sleepSeconds =
+ opts.channelConfig.sleepBetweenDownloadsSeconds ?? settings.sleepBetweenDownloadsSeconds;
+ let fetched = 0;
+ let consecutiveFailures = 0;
+ for (const entry of plan.entries) {
+ if (opts.signal.aborted) {
+ result.stopped = "cancelled";
+ break;
+ }
+ if (opts.drainSignal?.aborted) {
+ result.stopped = "drained";
+ break;
+ }
+ if (entry.onDisk || (await destinationExists(dataDir, entry.id, opts.channelConfig.handling))) {
+ result.skipped.push(entry.file);
+ continue;
+ }
+ const gate = await diskGate(opts.paths, settings, { dir: dataDir });
+ if (!gate.ok) {
+ result.stopped = gate.message;
+ log(`Stopping: ${gate.message}.\n`);
+ break;
+ }
+ if (fetched > 0) {
+ const gap = archiveOrgGapMs(sleepSeconds, random());
+ log(`Waiting ${(gap / 1000).toFixed(1)}s before the next file (archive.org pacing)...\n`);
+ await sleep(gap, opts.signal);
+ if (opts.signal.aborted) {
+ result.stopped = "cancelled";
+ break;
+ }
+ }
+ fetched++;
+ log(`[${fetched}] ${entry.file} → data/${entry.id}/\n`);
+ let rec: DownloadOutcomeRecord | null = null;
+ let error = "";
+ try {
+ rec = await downloadOne({
+ channelSlug: opts.slug,
+ channelConfig: opts.channelConfig,
+ paths: opts.paths,
+ videoUrl: entry.url,
+ onLog: log,
+ signal: opts.signal,
+ cookiePolicy: resolveCookiePolicy(settings, opts.channelConfig),
+ inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback,
+ globalSkipLiveDownloads: settings.skipLiveDownloads,
+ appendArchive: true,
+ });
+ } catch (err) {
+ error = (err as Error).message;
+ }
+ if (rec && isOk(rec)) {
+ consecutiveFailures = 0;
+ result.imported.push(entry.file);
+ await mergeRosterFile(
+ opts.paths,
+ opts.slug,
+ [{ id: entry.id, url: entry.url }],
+ new Date().toISOString(),
+ "import",
+ ).catch(() => {
+ /* the download succeeded; a roster write failure must not fail it */
+ });
+ deps.onImported?.(entry.id);
+ continue;
+ }
+ consecutiveFailures++;
+ const why = error || rec?.attempts.at(-1)?.error || rec?.status || "failed";
+ result.failed.push({ file: entry.file, error: why });
+ log(` failed: ${why}\n`);
+ if (rec?.failureClass === "rate_limit") {
+ result.rateLimited = true;
+ result.stopped = "archive.org rate-limited the download; stopping (re-run later — files on disk are skipped)";
+ log(`${result.stopped}.\n`);
+ break;
+ }
+ if (consecutiveFailures >= ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES) {
+ result.stopped = `${consecutiveFailures} failures in a row; stopping`;
+ log(`${result.stopped}.\n`);
+ break;
+ }
+ }
+ log(
+ `archive.org import of ${plan.identifier}: ${result.imported.length} imported, ` +
+ `${result.skipped.length} already on disk, ${result.failed.length} failed` +
+ `${result.stopped ? ` — stopped: ${result.stopped}` : ""}.\n`,
+ );
+ return result;
+}
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -281,6 +281,18 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
queueKeyStrategy: "custom",
needsMedia: true,
},
+ // CHOSEN FILES OF ONE archive.org ITEM, imported one at a time on
+ // archive.org's own queue (controller/archiveOrgImport.ts). Drainable: a
+ // drain lets the file in flight finish and starts no more; re-running the
+ // same command resumes, skipping what landed.
+ "import-archive-org": {
+ kind: "import-archive-org",
+ label: "Import from archive.org",
+ drainable: true,
+ replayable: false,
+ queueKeyStrategy: "platform",
+ needsMedia: true,
+ },
// ONE WINDOW of a video's source media, fetched into data/<id>/clips/ for a
// tool that asked for it by name (umtool's clip bench). Platform-queued like
// every other fetch so it takes its turn behind the channel's own downloads;
diff --git a/common/lib/archiveOrg-server.test.ts b/common/lib/archiveOrg-server.test.ts
@@ -0,0 +1,113 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, readFile, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { ArchiveOrgClient } from "./archiveOrgClient";
+import {
+ ensureArchiveOrgProvenance,
+ loadArchiveOrgProvenance,
+} from "./archiveOrg-server";
+import { loadMetadataHistory } from "./metadataHistory-server";
+import { archiveOrgDetailsUrl } from "./archiveOrgId";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrg-server.test.ts
+//
+// The provenance step on a temp video dir, against a scripted archive.org.
+// Every name here is invented.
+
+const ITEM = "example-item";
+const FILE = "First Upload-AbC123xyz_9.mp4";
+const URL_FILE = archiveOrgDetailsUrl({ identifier: ITEM, file: FILE });
+
+const ITEM_META = {
+ metadata: { identifier: ITEM, title: "Example Channel Archive", creator: "Example Creator", collection: "community" },
+ files: [
+ { name: FILE, source: "original" },
+ { name: "First Upload-AbC123xyz_9.info.json", source: "original" },
+ { name: "Second-Def456uvw_8.mp4", source: "original" },
+ { name: `${ITEM}_archive.torrent`, source: "metadata" },
+ ],
+};
+
+const INFO = {
+ id: "AbC123xyz_9",
+ extractor_key: "Youtube",
+ title: "First Upload (original)",
+ upload_date: "20230405",
+ uploader: "Example Creator",
+};
+
+function fakeClient(answers: Record<string, unknown>, seen: string[]) {
+ return new ArchiveOrgClient(
+ {
+ sleep: async () => {},
+ fetch: async (url) => {
+ seen.push(url);
+ if (!(url in answers)) return new Response("{}", { status: 404 });
+ return new Response(JSON.stringify(answers[url]), { status: 200 });
+ },
+ },
+ { minGapMs: 0 },
+ );
+}
+
+async function videoDir(info: Record<string, unknown>): Promise<string> {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "archiveorg-prov-"));
+ await writeFile(path.join(dir, "metadata.info.json"), JSON.stringify(info));
+ return dir;
+}
+
+test("writes the sidecar once and corrects the record, through the history", async () => {
+ const seen: string[] = [];
+ const client = fakeClient(
+ {
+ [`https://archive.org/metadata/${ITEM}`]: ITEM_META,
+ [`https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.info.json`]: INFO,
+ },
+ seen,
+ );
+ const dir = await videoDir({
+ id: `${ITEM}/${FILE}`,
+ extractor_key: "ArchiveOrg",
+ title: "Example Channel Archive",
+ webpage_url: `https://archive.org/details/${ITEM}`,
+ upload_date: "20240304",
+ });
+ const lines: string[] = [];
+ const prov = await ensureArchiveOrgProvenance({ videoDir: dir, videoUrl: URL_FILE, client, onLog: (l) => lines.push(l) });
+ assert.equal(prov?.mirror?.id, "AbC123xyz_9");
+ assert.deepEqual(await loadArchiveOrgProvenance(dir), prov);
+ const info = JSON.parse(await readFile(path.join(dir, "metadata.info.json"), "utf8"));
+ assert.equal(info.webpage_url, URL_FILE);
+ assert.equal(info.title, "First Upload (original)");
+ assert.equal(info.upload_date, "20230405");
+ assert.equal(info.id, `${ITEM}/${FILE}`);
+ const history = await loadMetadataHistory(dir);
+ assert.equal(history?.entries.at(-1)?.by, "archiveorg-provenance");
+ assert.equal(seen.length, 2);
+
+ // A second run (the next download of the record) asks nothing.
+ await ensureArchiveOrgProvenance({ videoDir: dir, videoUrl: URL_FILE, client });
+ assert.equal(seen.length, 2);
+});
+
+test("archive.org unreachable: no sidecar, but the page is still the file's", async () => {
+ const seen: string[] = [];
+ const client = fakeClient({}, seen);
+ const dir = await videoDir({ id: `${ITEM}/${FILE}`, webpage_url: `https://archive.org/details/${ITEM}`, title: "Item" });
+ const lines: string[] = [];
+ const prov = await ensureArchiveOrgProvenance({ videoDir: dir, videoUrl: URL_FILE, client, onLog: (l) => lines.push(l) });
+ assert.equal(prov, null);
+ assert.equal(await loadArchiveOrgProvenance(dir), null);
+ const info = JSON.parse(await readFile(path.join(dir, "metadata.info.json"), "utf8"));
+ assert.equal(info.webpage_url, URL_FILE);
+ assert.equal(info.title, "Item");
+ assert.ok(lines.some((l) => /provenance not fetched/.test(l)));
+});
+
+test("a URL that is not archive.org is left alone", async () => {
+ const dir = await videoDir({ id: "x", webpage_url: "https://example.com/x" });
+ assert.equal(await ensureArchiveOrgProvenance({ videoDir: dir, videoUrl: "https://example.com/x" }), null);
+});
diff --git a/common/lib/archiveOrg-server.ts b/common/lib/archiveOrg-server.ts
@@ -0,0 +1,136 @@
+// archive.org PROVENANCE ON DISK — the `archiveorg.json` sidecar, and the step
+// every managed download of an archive.org record runs after yt-dlp writes its
+// metadata (ytdlp/downloadOneManaged.ts, right after the prefetch).
+//
+// THE STEP, `ensureArchiveOrgProvenance`:
+//
+// 1. the sidecar: kept when it is already there for this item and file;
+// otherwise built from the item's metadata API (one cached request per
+// item, lib/archiveOrgClient.ts) and, when the item carries one, the
+// `.info.json` a mirroring tool uploaded beside the file (one request).
+// 2. the record: metadata.info.json corrected from it
+// (lib/archiveOrg.ts archiveOrgMetadataPatch) through `patchMetadataInfo`,
+// so the change is in metadata.history.json as `archiveorg-provenance`.
+//
+// Step 2 needs no network and always runs: a sidecar that could not be fetched
+// (archive.org refusing, the operator offline) still leaves the record's
+// `webpage_url` the file's page, which is what keeps its directory its own.
+// A failure is logged and does not fail the download — the next download of
+// the record, or a re-import, fills it in.
+
+import path from "node:path";
+import { readFile } from "node:fs/promises";
+import {
+ ARCHIVE_ORG_PROVENANCE_FILENAME,
+ archiveOrgMetadataPatch,
+ buildArchiveOrgProvenance,
+ coerceArchiveOrgProvenance,
+ findArchiveOrgInfoJson,
+ type ArchiveOrgProvenance,
+} from "./archiveOrg";
+import {
+ archiveOrgDetailsUrl,
+ parseArchiveOrgUrl,
+ type ArchiveOrgRef,
+} from "./archiveOrgId";
+import { archiveOrgClient, type ArchiveOrgClient } from "./archiveOrgClient";
+import { patchMetadataInfo } from "./metadataHistory-server";
+import { sidecar, sidecarField } from "./sidecar-server";
+
+export const archiveOrgProvenanceSidecar = sidecar(
+ ARCHIVE_ORG_PROVENANCE_FILENAME,
+ sidecarField(coerceArchiveOrgProvenance),
+);
+
+export const {
+ load: loadArchiveOrgProvenance,
+ write: writeArchiveOrgProvenance,
+} = archiveOrgProvenanceSidecar;
+
+// Fetch what the sidecar records: the item's metadata (cached) and, for a
+// mirror, its uploaded info.json.
+export async function fetchArchiveOrgProvenance(
+ ref: ArchiveOrgRef,
+ opts: { client?: ArchiveOrgClient; signal?: AbortSignal; now?: () => Date } = {},
+): Promise<ArchiveOrgProvenance> {
+ const client = opts.client ?? archiveOrgClient;
+ const item = await client.itemMetadata(ref.identifier, opts.signal);
+ const infoName = findArchiveOrgInfoJson(item, ref.file);
+ let infoJson: unknown;
+ if (infoName) {
+ try {
+ infoJson = await client.itemJsonFile(item.metadata.identifier, infoName, opts.signal);
+ } catch {
+ // The names still say whether it is a mirror; the info.json only adds
+ // the original's title and date.
+ infoJson = undefined;
+ }
+ }
+ return buildArchiveOrgProvenance({
+ ref,
+ item,
+ infoJson,
+ fetchedAt: (opts.now?.() ?? new Date()).toISOString(),
+ });
+}
+
+async function readInfo(videoDir: string): Promise<Record<string, unknown> | null> {
+ try {
+ const v = JSON.parse(await readFile(path.join(videoDir, "metadata.info.json"), "utf8"));
+ return v && typeof v === "object" && !Array.isArray(v) ? (v as Record<string, unknown>) : null;
+ } catch {
+ return null;
+ }
+}
+
+export type EnsureArchiveOrgProvenanceOpts = {
+ videoDir: string;
+ // The URL the record was fetched by (a details/embed/download URL).
+ videoUrl: string;
+ onLog?: (line: string) => void;
+ signal?: AbortSignal;
+ client?: ArchiveOrgClient;
+ now?: () => Date;
+};
+
+export async function ensureArchiveOrgProvenance(
+ opts: EnsureArchiveOrgProvenanceOpts,
+): Promise<ArchiveOrgProvenance | null> {
+ const log = opts.onLog ?? (() => {});
+ const ref = parseArchiveOrgUrl(opts.videoUrl);
+ if (!ref) return null;
+ let prov = await loadArchiveOrgProvenance(opts.videoDir);
+ if (prov && (prov.identifier !== ref.identifier || (prov.file ?? "") !== (ref.file ?? ""))) {
+ prov = null;
+ }
+ if (!prov) {
+ try {
+ prov = await fetchArchiveOrgProvenance(ref, opts);
+ await writeArchiveOrgProvenance(opts.videoDir, prov);
+ log(
+ `archive.org provenance: ${prov.identifier}${prov.file ? ` / ${prov.file}` : ""}` +
+ (prov.mirror ? ` — mirror of YouTube ${prov.mirror.id} (from ${prov.mirror.from})` : "") +
+ `.\n`,
+ );
+ } catch (err) {
+ log(`archive.org provenance not fetched (${(err as Error).message}); the record keeps yt-dlp's fields.\n`);
+ }
+ }
+ const info = await readInfo(opts.videoDir);
+ if (!info) return prov;
+ // Without a sidecar only the page is corrected — it is the one field the
+ // directory's name depends on.
+ const patch = prov
+ ? archiveOrgMetadataPatch(prov, info)
+ : info.webpage_url === archiveOrgDetailsUrl(ref)
+ ? {}
+ : { webpage_url: archiveOrgDetailsUrl(ref) };
+ if (Object.keys(patch).length > 0) {
+ try {
+ await patchMetadataInfo(opts.videoDir, patch, { by: "archiveorg-provenance", onLog: log });
+ } catch (err) {
+ log(`Could not correct metadata.info.json from the archive.org provenance: ${(err as Error).message}\n`);
+ }
+ }
+ return prov;
+}
diff --git a/common/lib/archiveOrg.test.ts b/common/lib/archiveOrg.test.ts
@@ -0,0 +1,269 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ archiveOrgCitationLinks,
+ archiveOrgMetadataPatch,
+ archiveOrgPlayableUrl,
+ buildArchiveOrgProvenance,
+ coerceArchiveOrgProvenance,
+ findArchiveOrgInfoJson,
+ listArchiveOrgMediaFiles,
+ parseArchiveOrgItemMetadata,
+ pickArchiveOrgFiles,
+ type ArchiveOrgItemMetadata,
+} from "./archiveOrg";
+import { archiveOrgVideoId } from "./archiveOrgId";
+import { platformFromMetadata, summarize } from "./transcripts-server";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrg.test.ts
+//
+// A synthetic channel-archive item: three videos uploaded by a mirroring
+// tool, each with its yt-dlp info.json, plus archive.org's derivatives. Every
+// name, id and date is invented.
+
+const ITEM = "example-item";
+const V1 = "First Upload-AbC123xyz_9.mp4";
+const V2 = "Second Upload-Def456uvw_8.mp4";
+const V3 = "Third Upload-Ghi789rst_7.mkv";
+
+function item(over: Partial<ArchiveOrgItemMetadata["metadata"]> = {}): ArchiveOrgItemMetadata {
+ return parseArchiveOrgItemMetadata({
+ metadata: {
+ identifier: ITEM,
+ title: "Example Channel Archive",
+ date: "2024-01-02",
+ publicdate: "2024-03-04 05:06:07",
+ creator: "Example Creator",
+ uploader: "someone@example.org",
+ collection: ["opensource_movies", "community"],
+ mediatype: "movies",
+ ...over,
+ },
+ files: [
+ { name: V1, source: "original", format: "MPEG4" },
+ { name: "First Upload-AbC123xyz_9.info.json", source: "original", format: "JSON" },
+ { name: V2, source: "original", format: "MPEG4", title: "Second, as titled on archive.org" },
+ { name: "Second Upload-Def456uvw_8.info.json", source: "original", format: "JSON" },
+ { name: V3, source: "original", format: "Matroska" },
+ { name: "Third Upload-Ghi789rst_7.mp4", source: "derivative", format: "h.264", original: V3 },
+ { name: "Third Upload-Ghi789rst_7.ogv", source: "derivative", format: "Ogg Video", original: V3 },
+ { name: `${ITEM}_archive.torrent`, source: "metadata", format: "Archive BitTorrent" },
+ { name: `${ITEM}_meta.xml`, source: "original", format: "Metadata" },
+ ],
+ })!;
+}
+
+const MIRROR_INFO = {
+ id: "AbC123xyz_9",
+ extractor_key: "Youtube",
+ title: "First Upload (original title)",
+ upload_date: "20230405",
+ uploader: "Example Creator",
+ channel_url: "https://www.youtube.com/channel/UCexample",
+ description: "The original description.",
+};
+
+test("the metadata API's answer: an item, or null for an unknown identifier", () => {
+ assert.equal(parseArchiveOrgItemMetadata({}), null);
+ assert.equal(parseArchiveOrgItemMetadata(null), null);
+ assert.equal(item().files.length, 9);
+});
+
+test("media files are the originals with a media extension", () => {
+ assert.deepEqual(listArchiveOrgMediaFiles(item()).map((f) => f.name), [V1, V2, V3]);
+});
+
+test("a bulk import picks by exact names or by a case-insensitive regex", () => {
+ assert.deepEqual(pickArchiveOrgFiles(item(), { files: [V2, "nope.mp4", `${ITEM}_meta.xml`] }), {
+ picked: [V2],
+ unknown: ["nope.mp4", `${ITEM}_meta.xml`],
+ });
+ assert.deepEqual(pickArchiveOrgFiles(item(), { match: "^(first|third)" }).picked, [V1, V3]);
+ assert.deepEqual(pickArchiveOrgFiles(item(), { match: "\\.mkv$" }).picked, [V3]);
+});
+
+test("a file's info.json is its stem's; a single-media item's is its only one", () => {
+ assert.equal(findArchiveOrgInfoJson(item(), V1), "First Upload-AbC123xyz_9.info.json");
+ assert.equal(findArchiveOrgInfoJson(item(), V3), null);
+ assert.equal(findArchiveOrgInfoJson(item(), undefined), null);
+});
+
+test("provenance of a mirror reads the original from its info.json", () => {
+ const prov = buildArchiveOrgProvenance({
+ ref: { identifier: ITEM, file: V1 },
+ item: item(),
+ infoJson: MIRROR_INFO,
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ assert.equal(prov.identifier, ITEM);
+ assert.equal(prov.file, V1);
+ assert.equal(prov.itemUrl, `https://archive.org/details/${ITEM}`);
+ assert.equal(prov.fileUrl, `https://archive.org/details/${ITEM}/First%20Upload-AbC123xyz_9.mp4`);
+ assert.equal(prov.downloadUrl, `https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.mp4`);
+ assert.equal(prov.torrentUrl, `https://archive.org/download/${ITEM}/${ITEM}_archive.torrent`);
+ assert.deepEqual(prov.item, {
+ title: "Example Channel Archive",
+ date: "2024-01-02",
+ publicDate: "2024-03-04 05:06:07",
+ creator: "Example Creator",
+ collections: ["opensource_movies", "community"],
+ mediatype: "movies",
+ });
+ assert.deepEqual(prov.mirror, {
+ platform: "youtube",
+ id: "AbC123xyz_9",
+ url: "https://www.youtube.com/watch?v=AbC123xyz_9",
+ title: "First Upload (original title)",
+ uploadDate: "20230405",
+ uploader: "Example Creator",
+ channelUrl: "https://www.youtube.com/channel/UCexample",
+ description: "The original description.",
+ from: "info-json",
+ });
+ // The uploading account's e-mail is never kept.
+ assert.equal(JSON.stringify(prov).includes("@"), false);
+ assert.deepEqual(coerceArchiveOrgProvenance(JSON.parse(JSON.stringify(prov))), prov);
+ assert.equal(coerceArchiveOrgProvenance({ identifier: ITEM }), null);
+});
+
+test("without an info.json the mirror is known by name only; a plain item is no mirror", () => {
+ const byName = buildArchiveOrgProvenance({
+ ref: { identifier: ITEM, file: V3 },
+ item: item(),
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ assert.deepEqual(byName.mirror, {
+ platform: "youtube",
+ id: "Ghi789rst_7",
+ url: "https://www.youtube.com/watch?v=Ghi789rst_7",
+ from: "file-name",
+ });
+ const byIdent = buildArchiveOrgProvenance({
+ ref: { identifier: "youtube-Jkl012mno_6" },
+ item: item({ identifier: "youtube-Jkl012mno_6" }),
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ assert.equal(byIdent.mirror?.from, "identifier");
+ const plain = buildArchiveOrgProvenance({
+ ref: { identifier: "example-film" },
+ item: item({ identifier: "example-film" }),
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ assert.equal(plain.mirror, undefined);
+ assert.equal(plain.file, undefined);
+ // A non-YouTube info.json is not a YouTube mirror.
+ const other = buildArchiveOrgProvenance({
+ ref: { identifier: "example-film", file: "film.mp4" },
+ item: item({ identifier: "example-film" }),
+ infoJson: { id: "abc", extractor_key: "Generic" },
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ assert.equal(other.mirror, undefined);
+});
+
+test("the record is corrected: the file's page and title, the original's date", () => {
+ const prov = buildArchiveOrgProvenance({
+ ref: { identifier: ITEM, file: V1 },
+ item: item(),
+ infoJson: MIRROR_INFO,
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ const info = {
+ id: `${ITEM}/${V1}`,
+ title: "Example Channel Archive",
+ webpage_url: `https://archive.org/details/${ITEM}`,
+ upload_date: "20240304",
+ timestamp: 1709528767,
+ uploader: "someone@example.org",
+ };
+ const patch = archiveOrgMetadataPatch(prov, info);
+ assert.deepEqual(patch, {
+ webpage_url: prov.fileUrl,
+ title: "First Upload (original title)",
+ upload_date: "20230405",
+ description: "The original description.",
+ uploader: "Example Creator",
+ timestamp: null,
+ });
+ // Applied, nothing is left to change.
+ assert.deepEqual(archiveOrgMetadataPatch(prov, { ...info, ...patch }), {});
+
+ // One file of many, no mirror: the file's own archive.org title.
+ const v2 = buildArchiveOrgProvenance({
+ ref: { identifier: ITEM, file: V2 },
+ item: item({ creator: undefined }),
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ const p2 = archiveOrgMetadataPatch({ ...v2, mirror: undefined }, info);
+ assert.equal(p2.title, "Second, as titled on archive.org");
+ assert.equal(p2.uploader, null);
+ assert.equal("upload_date" in p2, false);
+});
+
+test("citation links: a mirror cites YouTube at the second, archive.org and the torrent as downloads", () => {
+ const prov = buildArchiveOrgProvenance({
+ ref: { identifier: ITEM, file: V1 },
+ item: item(),
+ infoJson: MIRROR_INFO,
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ assert.deepEqual(archiveOrgCitationLinks({ webpageUrl: prov.fileUrl, provenance: prov, seconds: 75.6 }), {
+ original: { label: "YouTube", url: "https://www.youtube.com/watch?v=AbC123xyz_9&t=75s" },
+ downloads: [
+ { label: "archive.org", url: prov.fileUrl },
+ { label: "torrent", url: `https://archive.org/download/${ITEM}/${ITEM}_archive.torrent` },
+ ],
+ });
+});
+
+test("citation links: a plain record is derived from its page alone", () => {
+ assert.deepEqual(archiveOrgCitationLinks({ webpageUrl: "https://archive.org/details/example-film", seconds: 30 }), {
+ original: { label: "archive.org", url: "https://archive.org/details/example-film" },
+ downloads: [{ label: "torrent", url: "https://archive.org/download/example-film/example-film_archive.torrent" }],
+ });
+ assert.deepEqual(archiveOrgCitationLinks({ webpageUrl: "https://example.com/x" }), { downloads: [] });
+});
+
+test("the player plays a browser-playable file: an mp4 before an mkv original", () => {
+ const formats = [
+ { url: `https://archive.org/download/${ITEM}/Third%20Upload-Ghi789rst_7.mkv`, ext: "mkv", format_note: "original" },
+ { url: `https://archive.org/download/${ITEM}/Third%20Upload-Ghi789rst_7.ogv`, ext: "ogv", format_note: "derivative" },
+ { url: `https://archive.org/download/${ITEM}/Third%20Upload-Ghi789rst_7.mp4`, ext: "mp4", format_note: "derivative" },
+ ];
+ assert.equal(archiveOrgPlayableUrl({ formats }), formats[2].url);
+ assert.equal(
+ archiveOrgPlayableUrl({ id: `${ITEM}/a b.mp3` }),
+ `https://archive.org/download/${ITEM}/a%20b.mp3`,
+ );
+ assert.equal(archiveOrgPlayableUrl({ id: ITEM }), undefined);
+});
+
+test("platformFromMetadata and summarize know an archive.org record", () => {
+ assert.equal(platformFromMetadata({ extractor_key: "ArchiveOrg" }), "archiveorg");
+ assert.equal(platformFromMetadata({ extractor: "archive.org" }), "archiveorg");
+ assert.equal(platformFromMetadata({ extractor_key: "YoutubeWebArchive" }), "youtube");
+ assert.equal(platformFromMetadata({ extractor_key: "Youtube" }), "youtube");
+ assert.equal(
+ platformFromMetadata({ extractor_key: "Generic", webpage_url: `https://archive.org/details/${ITEM}` }),
+ "archiveorg",
+ );
+ assert.equal(platformFromMetadata({ extractor_key: "Generic", webpage_url: "https://example.com/a.mp3" }), "youtube");
+
+ const id = archiveOrgVideoId({ identifier: ITEM, file: V1 });
+ const s = summarize("example-channel", id, {
+ id: `${ITEM}/${V1}`,
+ extractor_key: "ArchiveOrg",
+ title: "First Upload (original title)",
+ upload_date: "20230405",
+ webpage_url: `https://archive.org/details/${ITEM}/First%20Upload-AbC123xyz_9.mp4`,
+ formats: [{ url: `https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.mp4`, ext: "mp4", format_note: "original" }],
+ });
+ assert.equal(s.platform, "archiveorg");
+ assert.equal(s.id, id);
+ assert.equal(s.slug, `example-channel/${id}`);
+ assert.equal(s.mediaUrl, `https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.mp4`);
+ // Other platforms gain no key.
+ const yt = summarize("c", "AbC123xyz_9", { id: "AbC123xyz_9", extractor_key: "Youtube" });
+ assert.equal("mediaUrl" in yt, false);
+});
diff --git a/common/lib/archiveOrg.ts b/common/lib/archiveOrg.ts
@@ -0,0 +1,446 @@
+// archive.org AS A SOURCE — the pure half: what an item's metadata says, the
+// provenance a record keeps, the links a citation derives from it.
+//
+// Pure (no node:, no fetch) so the browser, the export build and the report
+// compose can all use it. The network half — the polite metadata client — is
+// lib/archiveOrgClient.ts; the per-video sidecar and the metadata patch are
+// lib/archiveOrg-server.ts.
+//
+// THE RECORD. A video imported from archive.org keeps yt-dlp's
+// metadata.info.json like every other (extractor_key "ArchiveOrg"), plus one
+// sidecar, `archiveorg.json` (ArchiveOrgProvenance below): the identifier and
+// file, the item's own title/date/creator/collections, the torrent, and — when
+// the item is a mirror of a YouTube upload — the ORIGINAL's id, URL, title,
+// date and uploader, read from the info.json the mirroring tool uploaded beside
+// the media.
+//
+// WEB-ARCHIVE (WARC / Wayback) RECORDS ARE NOT THIS. A captured web page is a
+// different kind of record with its own reader; it would get its own platform
+// and module beside this one, never a branch in it.
+
+import {
+ archiveOrgDetailsUrl,
+ archiveOrgDownloadUrl,
+ archiveOrgTorrentUrl,
+ parseArchiveOrgUrl,
+ youtubeIdFromFileName,
+ youtubeIdFromIdentifier,
+ type ArchiveOrgRef,
+} from "./archiveOrgId";
+
+export const ARCHIVE_ORG_PROVENANCE_FILENAME = "archiveorg.json";
+export const ARCHIVE_ORG_PROVENANCE_VERSION = 1;
+
+// ─── The item, as the metadata API returns it ───
+
+// https://archive.org/metadata/<identifier> — only the fields read here.
+export type ArchiveOrgItemFile = {
+ name: string;
+ // "original" | "derivative" | "metadata"
+ source?: string;
+ // archive.org's format name ("h.264", "MPEG4", "VBR MP3", "JSON", …).
+ format?: string;
+ size?: string;
+ // Seconds, as a string ("123.45") or a clock ("02:03").
+ length?: string;
+ title?: string;
+ // On a derivative: the original it was made from.
+ original?: string;
+};
+
+export type ArchiveOrgItemMetadata = {
+ metadata: {
+ identifier: string;
+ title?: string | string[];
+ date?: string;
+ publicdate?: string;
+ creator?: string | string[];
+ uploader?: string;
+ collection?: string | string[];
+ mediatype?: string;
+ description?: string | string[];
+ };
+ files: ArchiveOrgItemFile[];
+};
+
+function firstString(v: unknown): string | undefined {
+ if (typeof v === "string") return v.trim() || undefined;
+ if (Array.isArray(v)) {
+ for (const x of v) if (typeof x === "string" && x.trim()) return x.trim();
+ }
+ return undefined;
+}
+
+function stringList(v: unknown): string[] {
+ if (typeof v === "string") return v.trim() ? [v.trim()] : [];
+ if (Array.isArray(v)) return v.filter((x): x is string => typeof x === "string" && !!x.trim());
+ return [];
+}
+
+// The metadata API answers `{}` (HTTP 200) for an identifier with no item, and
+// a "dark" item has metadata but no files. Null for anything without both.
+export function parseArchiveOrgItemMetadata(raw: unknown): ArchiveOrgItemMetadata | null {
+ if (!raw || typeof raw !== "object") return null;
+ const r = raw as { metadata?: unknown; files?: unknown };
+ const m = r.metadata as Record<string, unknown> | undefined;
+ if (!m || typeof m !== "object" || typeof m.identifier !== "string") return null;
+ const files = Array.isArray(r.files)
+ ? r.files.filter(
+ (f): f is ArchiveOrgItemFile =>
+ !!f && typeof f === "object" && typeof (f as { name?: unknown }).name === "string",
+ )
+ : [];
+ return { metadata: m as ArchiveOrgItemMetadata["metadata"], files };
+}
+
+// The extensions archive.org serves media under that yt-dlp can download.
+const MEDIA_EXTS = new Set([
+ "mp4", "m4v", "mkv", "webm", "mov", "avi", "mpeg", "mpg", "ogv", "flv", "wmv", "3gp", "ts",
+ "mp3", "m4a", "ogg", "oga", "opus", "flac", "wav", "aac", "wma",
+]);
+
+function extOf(name: string): string {
+ const m = /\.([A-Za-z0-9]{1,8})$/.exec(name);
+ return m ? m[1].toLowerCase() : "";
+}
+
+export function isMediaFileName(name: string): boolean {
+ return MEDIA_EXTS.has(extOf(name));
+}
+
+// The item's media files: ORIGINALS only (a derivative is archive.org's
+// transcode of one, the same recording), in the item's own order.
+export function listArchiveOrgMediaFiles(item: ArchiveOrgItemMetadata): ArchiveOrgItemFile[] {
+ return item.files.filter((f) => f.source === "original" && isMediaFileName(f.name));
+}
+
+// ─── Picking files for a bulk import ───
+
+export type ArchiveOrgFileSelection =
+ | { files: string[] }
+ | { match: string };
+
+export type ArchiveOrgFilePick = {
+ // The chosen media files, in the item's order.
+ picked: string[];
+ // Named in `files` but not a media original of the item.
+ unknown: string[];
+};
+
+// `files` names exact paths; `match` is a case-insensitive regex over each
+// media file's path. Either way only media originals can be picked.
+export function pickArchiveOrgFiles(
+ item: ArchiveOrgItemMetadata,
+ sel: ArchiveOrgFileSelection,
+): ArchiveOrgFilePick {
+ const media = listArchiveOrgMediaFiles(item).map((f) => f.name);
+ if ("files" in sel) {
+ const want = new Set(sel.files);
+ const known = new Set(media);
+ return {
+ picked: media.filter((n) => want.has(n)),
+ unknown: sel.files.filter((n) => !known.has(n)),
+ };
+ }
+ const re = new RegExp(sel.match, "i");
+ return { picked: media.filter((n) => re.test(n)), unknown: [] };
+}
+
+// ─── The mirror's original ───
+
+export type ArchiveOrgMirror = {
+ platform: "youtube";
+ id: string;
+ url: string;
+ title?: string;
+ // YYYYMMDD
+ uploadDate?: string;
+ uploader?: string;
+ channelUrl?: string;
+ description?: string;
+ // Where the original's identity was read from: the uploaded info.json (and
+ // then every field above is the original's own), or only a name.
+ from: "info-json" | "identifier" | "file-name";
+};
+
+// The info.json a mirroring tool uploaded beside a media file: the file's own
+// stem + `.info.json`. For a single-media item, the item's only info.json.
+export function findArchiveOrgInfoJson(
+ item: ArchiveOrgItemMetadata,
+ file: string | undefined,
+): string | null {
+ const names = item.files.map((f) => f.name);
+ const infos = names.filter((n) => n.endsWith(".info.json"));
+ if (file) {
+ const stem = file.replace(/\.[A-Za-z0-9]{1,8}$/, "");
+ if (infos.includes(`${stem}.info.json`)) return `${stem}.info.json`;
+ return null;
+ }
+ return infos.length === 1 ? infos[0] : null;
+}
+
+function youtubeWatchUrl(id: string): string {
+ return `https://www.youtube.com/watch?v=${id}`;
+}
+
+// The original, from an uploaded info.json when there is one (authoritative:
+// the record yt-dlp wrote when it downloaded the upload), else from the names.
+export function archiveOrgMirrorOf(opts: {
+ identifier: string;
+ file?: string;
+ infoJson?: unknown;
+}): ArchiveOrgMirror | null {
+ const info = opts.infoJson as Record<string, unknown> | null | undefined;
+ if (info && typeof info === "object") {
+ const key = String(info.extractor_key ?? info.extractor ?? "");
+ const id = typeof info.id === "string" ? info.id : "";
+ if (/^youtube$/i.test(key) && /^[A-Za-z0-9_-]{11}$/.test(id)) {
+ const str = (k: string) => (typeof info[k] === "string" && (info[k] as string).trim() ? (info[k] as string) : undefined);
+ const date = str("upload_date");
+ return {
+ platform: "youtube",
+ id,
+ url: youtubeWatchUrl(id),
+ title: str("title"),
+ uploadDate: date && /^\d{8}$/.test(date) ? date : undefined,
+ uploader: str("uploader") ?? str("channel"),
+ channelUrl: str("channel_url") ?? str("uploader_url"),
+ description: str("description"),
+ from: "info-json",
+ };
+ }
+ }
+ const fromIdent = youtubeIdFromIdentifier(opts.identifier);
+ if (fromIdent) return { platform: "youtube", id: fromIdent, url: youtubeWatchUrl(fromIdent), from: "identifier" };
+ const fromName = opts.file ? youtubeIdFromFileName(opts.file) : null;
+ if (fromName) return { platform: "youtube", id: fromName, url: youtubeWatchUrl(fromName), from: "file-name" };
+ return null;
+}
+
+// ─── The provenance sidecar ───
+
+export type ArchiveOrgProvenance = {
+ version: number;
+ identifier: string;
+ // The file's path inside the item; absent for a whole item.
+ file?: string;
+ // The item's page, and the file's own page inside it.
+ itemUrl: string;
+ fileUrl?: string;
+ // The media file's bytes, when the record is one file.
+ downloadUrl?: string;
+ // The item's BitTorrent file, when the item lists one.
+ torrentUrl?: string;
+ item: {
+ title?: string;
+ // archive.org's `date` (the content date the uploader gave), and
+ // `publicdate` (when the item went up).
+ date?: string;
+ publicDate?: string;
+ creator?: string;
+ collections: string[];
+ mediatype?: string;
+ };
+ // The file's own title in the item, when it has one.
+ fileTitle?: string;
+ mirror?: ArchiveOrgMirror;
+ fetchedAt: string;
+};
+
+export function buildArchiveOrgProvenance(opts: {
+ ref: ArchiveOrgRef;
+ item: ArchiveOrgItemMetadata;
+ infoJson?: unknown;
+ fetchedAt: string;
+}): ArchiveOrgProvenance {
+ const { ref, item } = opts;
+ const m = item.metadata;
+ const identifier = m.identifier || ref.identifier;
+ const fileEntry = ref.file ? item.files.find((f) => f.name === ref.file) : undefined;
+ const torrentName = `${identifier}_archive.torrent`;
+ const hasTorrent = item.files.some((f) => f.name === torrentName);
+ const mirror = archiveOrgMirrorOf({ identifier, file: ref.file, infoJson: opts.infoJson });
+ // The uploader field is the uploading account's e-mail address; it is never
+ // kept. `creator` is the public credit.
+ const prov: ArchiveOrgProvenance = {
+ version: ARCHIVE_ORG_PROVENANCE_VERSION,
+ identifier,
+ ...(ref.file ? { file: ref.file } : {}),
+ itemUrl: archiveOrgDetailsUrl({ identifier }),
+ ...(ref.file
+ ? {
+ fileUrl: archiveOrgDetailsUrl({ identifier, file: ref.file }),
+ downloadUrl: archiveOrgDownloadUrl(identifier, ref.file),
+ }
+ : {}),
+ ...(hasTorrent ? { torrentUrl: archiveOrgTorrentUrl(identifier) } : {}),
+ item: {
+ title: firstString(m.title),
+ date: firstString(m.date),
+ publicDate: firstString(m.publicdate),
+ creator: firstString(m.creator),
+ collections: stringList(m.collection),
+ mediatype: firstString(m.mediatype),
+ },
+ ...(fileEntry?.title?.trim() ? { fileTitle: fileEntry.title.trim() } : {}),
+ ...(mirror ? { mirror } : {}),
+ fetchedAt: opts.fetchedAt,
+ };
+ return stripUndefinedDeep(prov);
+}
+
+function stripUndefinedDeep<T>(v: T): T {
+ if (Array.isArray(v)) return v.map(stripUndefinedDeep) as T;
+ if (v && typeof v === "object") {
+ const out: Record<string, unknown> = {};
+ for (const [k, x] of Object.entries(v)) if (x !== undefined) out[k] = stripUndefinedDeep(x);
+ return out as T;
+ }
+ return v;
+}
+
+// The sidecar's shape check: a record or null.
+export function coerceArchiveOrgProvenance(value: unknown): ArchiveOrgProvenance | null {
+ const v = value as Partial<ArchiveOrgProvenance> | null;
+ if (
+ !v ||
+ typeof v !== "object" ||
+ typeof v.identifier !== "string" ||
+ typeof v.itemUrl !== "string" ||
+ typeof v.fetchedAt !== "string" ||
+ !v.item ||
+ typeof v.item !== "object"
+ ) {
+ return null;
+ }
+ const item = v.item as ArchiveOrgProvenance["item"];
+ const mirror = v.mirror;
+ return {
+ ...(v as ArchiveOrgProvenance),
+ item: { ...item, collections: Array.isArray(item.collections) ? item.collections : [] },
+ ...(mirror && (typeof mirror.id !== "string" || typeof mirror.url !== "string")
+ ? { mirror: undefined }
+ : {}),
+ };
+}
+
+// ─── What the record's metadata.info.json is corrected to ───
+
+// yt-dlp describes an entry of a multi-file item with the ITEM's title and
+// page, and a mirror with archive.org's upload date. The record says:
+//
+// webpage_url the page of what was imported (the file's page inside the
+// item). NOT cosmetic: the snapshot renames every video dir to
+// `extractVideoId(webpage_url)` (reconcileVideoDirs.ts), so an
+// entry left with the item's page would be merged into the item.
+// title the original's title (a mirror), else the file's own title,
+// else its file name — never the item's for one file of many.
+// upload_date the original's date (a mirror), with its timestamp dropped
+// description the original's (a mirror), when it had one
+// uploader the original's uploader, else the item's public credit — never
+// the uploading account's e-mail address
+//
+// Returns only the keys that differ from `info`; empty when nothing does.
+export function archiveOrgMetadataPatch(
+ prov: ArchiveOrgProvenance,
+ info: Record<string, unknown>,
+): Record<string, unknown> {
+ const want: Record<string, unknown> = {};
+ want.webpage_url = prov.fileUrl ?? prov.itemUrl;
+ const mirror = prov.mirror;
+ const fileName = prov.file ? (prov.file.split("/").pop() ?? prov.file).replace(/\.[A-Za-z0-9]{1,8}$/, "") : undefined;
+ const title = mirror?.title ?? (prov.file ? (prov.fileTitle ?? fileName) : undefined);
+ if (title) want.title = title;
+ if (mirror?.uploadDate) want.upload_date = mirror.uploadDate;
+ if (mirror?.description) want.description = mirror.description;
+ const uploader = mirror?.uploader ?? prov.item.creator;
+ const current = info.uploader;
+ if (uploader) want.uploader = uploader;
+ else if (typeof current === "string" && current.includes("@")) want.uploader = null;
+ const patch: Record<string, unknown> = {};
+ for (const [k, v] of Object.entries(want)) {
+ if (JSON.stringify(info[k]) !== JSON.stringify(v)) patch[k] = v;
+ }
+ // A mirror's date replaces archive.org's; the timestamp yt-dlp derived from
+ // the item's publicdate would contradict it.
+ if (patch.upload_date !== undefined && typeof info.timestamp === "number") patch.timestamp = null;
+ return patch;
+}
+
+// ─── The links a record derives ───
+
+export type ExternalLink = { label: string; url: string };
+
+// What a citation of an archive.org record links, beside its moment page:
+//
+// original where the recording was first published — the YouTube upload
+// at the cited second for a mirror, else the archive.org page
+// downloads where a reader can fetch the file to check it: the archive.org
+// page (for a mirror, since `original` is YouTube there) and the
+// item's torrent
+//
+// Derived from the record's `webpage_url` alone when there is no provenance
+// (the torrent name is archive.org's fixed `<identifier>_archive.torrent`);
+// the provenance adds the mirror.
+export function archiveOrgCitationLinks(opts: {
+ webpageUrl: string | null | undefined;
+ provenance?: ArchiveOrgProvenance | null;
+ seconds?: number;
+}): { original?: ExternalLink; downloads: ExternalLink[] } {
+ const prov = opts.provenance ?? null;
+ const ref = opts.webpageUrl ? parseArchiveOrgUrl(opts.webpageUrl) : null;
+ const identifier = prov?.identifier ?? ref?.identifier;
+ if (!identifier) return { downloads: [] };
+ const page = prov?.fileUrl ?? prov?.itemUrl ?? archiveOrgDetailsUrl(ref!);
+ const torrent: ExternalLink = {
+ label: "torrent",
+ url: prov ? (prov.torrentUrl ?? archiveOrgTorrentUrl(identifier)) : archiveOrgTorrentUrl(identifier),
+ };
+ const archive: ExternalLink = { label: "archive.org", url: page };
+ const mirror = prov?.mirror;
+ if (mirror) {
+ const secs = Math.max(0, Math.floor(opts.seconds ?? 0));
+ const u = new URL(mirror.url);
+ if (secs > 0) u.searchParams.set("t", `${secs}s`);
+ return { original: { label: "YouTube", url: u.toString() }, downloads: [archive, torrent] };
+ }
+ return { original: archive, downloads: [torrent] };
+}
+
+// ─── Playing it ───
+
+// Browser-playable containers, best first. archive.org transcodes most video
+// originals to an h.264 mp4 derivative, which every browser plays — an .avi or
+// .mpeg original does not.
+const PLAYABLE_EXTS = ["mp4", "m4v", "webm", "m4a", "mp3", "ogg", "oga", "opus"];
+
+type FormatLike = { url?: unknown; ext?: unknown; format_note?: unknown };
+
+// The one file of a record a <video> element can play from archive.org, or
+// undefined. Read from the record's yt-dlp `formats` (each an archive.org
+// download URL; `format_note` is "original" or "derivative"): the playable
+// extension ranked first wins, an original before a derivative of the same
+// extension. A file record with no formats falls back to its own file.
+export function archiveOrgPlayableUrl(meta: {
+ id?: string;
+ formats?: unknown;
+}): string | undefined {
+ const formats = Array.isArray(meta.formats) ? (meta.formats as FormatLike[]) : [];
+ let best: { rank: number; url: string } | null = null;
+ for (const f of formats) {
+ if (typeof f?.url !== "string" || !/^https:\/\/archive\.org\/download\//.test(f.url)) continue;
+ const ext = typeof f.ext === "string" ? f.ext.toLowerCase() : extOf(f.url);
+ const i = PLAYABLE_EXTS.indexOf(ext);
+ if (i < 0) continue;
+ const rank = i * 2 + (f.format_note === "original" ? 0 : 1);
+ if (!best || rank < best.rank) best = { rank, url: f.url };
+ }
+ if (best) return best.url;
+ const id = meta.id ?? "";
+ const slash = id.indexOf("/");
+ if (slash > 0) {
+ const file = id.slice(slash + 1);
+ if (PLAYABLE_EXTS.includes(extOf(file))) return archiveOrgDownloadUrl(id.slice(0, slash), file);
+ }
+ return undefined;
+}
diff --git a/common/lib/archiveOrgClient.test.ts b/common/lib/archiveOrgClient.test.ts
@@ -0,0 +1,131 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ ARCHIVE_ORG_USER_AGENT,
+ ArchiveOrgClient,
+ ArchiveOrgRequestError,
+ parseRetryAfterMs,
+} from "./archiveOrgClient";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrgClient.test.ts
+//
+// THE POLITENESS RULES, ON A FAKE CLOCK. No request leaves the process: the
+// fetch is a script of canned responses, the clock only moves when the client
+// sleeps (or when a "request" takes time), and every sleep is recorded.
+
+type Scripted = { status: number; body?: unknown; headers?: Record<string, string>; takesMs?: number };
+
+function harness(script: Scripted[], opts: ConstructorParameters<typeof ArchiveOrgClient>[1] = {}) {
+ let now = 1_000_000;
+ const sleeps: number[] = [];
+ const calls: { url: string; at: number; ua: string | null }[] = [];
+ const client = new ArchiveOrgClient(
+ {
+ now: () => now,
+ sleep: async (ms) => {
+ sleeps.push(ms);
+ now += ms;
+ },
+ random: () => 0.5, // no jitter
+ fetch: async (url, init) => {
+ const headers = new Headers(init.headers);
+ calls.push({ url, at: now, ua: headers.get("user-agent") });
+ const next = script.shift();
+ if (!next) throw new Error("script exhausted");
+ now += next.takesMs ?? 100;
+ return new Response(next.body === undefined ? "{}" : JSON.stringify(next.body), {
+ status: next.status,
+ headers: next.headers,
+ });
+ },
+ },
+ opts,
+ );
+ return { client, sleeps, calls, advance: (ms: number) => (now += ms) };
+}
+
+const META = { metadata: { identifier: "example-item", title: "Example" }, files: [{ name: "a.mp4", source: "original" }] };
+
+test("an item's metadata is asked for once, identified, and cached", async () => {
+ const h = harness([{ status: 200, body: META }]);
+ const a = await h.client.itemMetadata("example-item");
+ const b = await h.client.itemMetadata("example-item");
+ assert.equal(a, b);
+ assert.equal(h.calls.length, 1);
+ assert.equal(h.calls[0].url, "https://archive.org/metadata/example-item");
+ assert.equal(h.calls[0].ua, ARCHIVE_ORG_USER_AGENT);
+ assert.match(ARCHIVE_ORG_USER_AGENT, /\(\+https?:\/\//);
+});
+
+test("the cache expires", async () => {
+ const h = harness([{ status: 200, body: META }, { status: 200, body: META }], { cacheTtlMs: 1000 });
+ await h.client.itemMetadata("example-item");
+ h.advance(500);
+ await h.client.itemMetadata("example-item");
+ assert.equal(h.calls.length, 1);
+ h.advance(1000);
+ await h.client.itemMetadata("example-item");
+ assert.equal(h.calls.length, 2);
+});
+
+test("requests are one at a time with a gap between them", async () => {
+ const h = harness(
+ [
+ { status: 200, body: { n: 1 }, takesMs: 500 },
+ { status: 200, body: { n: 2 }, takesMs: 500 },
+ { status: 200, body: { n: 3 }, takesMs: 500 },
+ ],
+ { minGapMs: 2000 },
+ );
+ // Started together; served in order, each after the previous ended + the gap.
+ const results = await Promise.all([
+ h.client.getJson("https://archive.org/download/i/1.json"),
+ h.client.getJson("https://archive.org/download/i/2.json"),
+ h.client.getJson("https://archive.org/download/i/3.json"),
+ ]);
+ assert.deepEqual(results, [{ n: 1 }, { n: 2 }, { n: 3 }]);
+ assert.equal(h.calls[1].at - h.calls[0].at, 2500);
+ assert.equal(h.calls[2].at - h.calls[1].at, 2500);
+});
+
+test("a 429 waits for Retry-After, then succeeds", async () => {
+ const h = harness([
+ { status: 429, headers: { "retry-after": "30" } },
+ { status: 200, body: META },
+ ]);
+ await h.client.itemMetadata("example-item");
+ assert.equal(h.calls.length, 2);
+ assert.ok(h.sleeps.includes(30_000), `slept ${h.sleeps}`);
+});
+
+test("a 503 without Retry-After backs off exponentially, and stops after maxAttempts", async () => {
+ const h = harness(
+ [{ status: 503 }, { status: 503 }, { status: 503 }, { status: 503 }, { status: 200, body: META }],
+ { maxAttempts: 4, baseBackoffMs: 5000, minGapMs: 0 },
+ );
+ await assert.rejects(h.client.itemMetadata("example-item"), (err: unknown) => {
+ assert.ok(err instanceof ArchiveOrgRequestError);
+ assert.equal(err.rateLimited, true);
+ assert.match(err.message, /after 4 attempts/);
+ return true;
+ });
+ assert.equal(h.calls.length, 4);
+ // 5 s, 10 s, 20 s between the four attempts; none after the last.
+ assert.deepEqual(h.sleeps, [5000, 10000, 20000]);
+});
+
+test("a 404 is not retried; an empty answer is no item", async () => {
+ const h = harness([{ status: 404 }]);
+ await assert.rejects(h.client.getJson("https://archive.org/download/i/x.json"), /HTTP 404/);
+ assert.equal(h.calls.length, 1);
+ const h2 = harness([{ status: 200, body: {} }]);
+ await assert.rejects(h2.client.itemMetadata("no-such-item"), /has no item "no-such-item"/);
+});
+
+test("Retry-After as seconds or as an HTTP date", () => {
+ assert.equal(parseRetryAfterMs("12", 0), 12_000);
+ assert.equal(parseRetryAfterMs(new Date(60_000).toUTCString(), 0), 60_000);
+ assert.equal(parseRetryAfterMs(null, 0), null);
+ assert.equal(parseRetryAfterMs("soon", 0), null);
+});
diff --git a/common/lib/archiveOrgClient.ts b/common/lib/archiveOrgClient.ts
@@ -0,0 +1,196 @@
+// THE POLITE archive.org CLIENT — every request this app makes to archive.org
+// that is not a yt-dlp spawn (the item metadata API, a mirror's info.json).
+//
+// archive.org is a non-profit serving files off its own disks; the rules here
+// are the operator's "be polite to archive.org", made mechanical:
+//
+// ONE AT A TIME requests are chained: a second caller waits for the first
+// to finish, and then for the gap.
+// A GAP at least `minGapMs` (2 s) between the end of one request
+// and the start of the next, from this process.
+// IDENTIFIED a User-Agent naming the project and its URL.
+// BACKS OFF a 429 or 503 (and 502/504, and a dropped connection) is
+// retried after the server's Retry-After when it sends one,
+// else after an exponential wait (5 s, 10 s, 20 s… capped at
+// 2 min, ±20 % jitter), at most `maxAttempts` times in all;
+// then it stops with ArchiveOrgRequestError, `rateLimited`.
+// ASKS ONCE an item's metadata is cached in memory for `cacheTtlMs`
+// (6 h): a bulk import of 161 files of one item asks for it
+// once, and so does every per-file provenance step after it.
+//
+// The downloads themselves are yt-dlp's (one plain HTTP stream per file, paced
+// by PLATFORM_ARGS.archiveorg) and run on the `platform:archiveorg` job queue,
+// one at a time.
+
+import { PROJECT_NAME, PROJECT_URL } from "./project";
+import {
+ parseArchiveOrgItemMetadata,
+ type ArchiveOrgItemMetadata,
+} from "./archiveOrg";
+import { archiveOrgDownloadUrl, archiveOrgMetadataUrl } from "./archiveOrgId";
+
+export const ARCHIVE_ORG_USER_AGENT = `${PROJECT_NAME} archive.org import (+${PROJECT_URL})`;
+
+export class ArchiveOrgRequestError extends Error {
+ readonly status: number | null;
+ readonly rateLimited: boolean;
+ constructor(message: string, status: number | null, rateLimited: boolean) {
+ super(message);
+ this.name = "ArchiveOrgRequestError";
+ this.status = status;
+ this.rateLimited = rateLimited;
+ }
+}
+
+export type ArchiveOrgClientDeps = {
+ fetch?: (url: string, init: RequestInit) => Promise<Response>;
+ now?: () => number;
+ sleep?: (ms: number, signal?: AbortSignal) => Promise<void>;
+ random?: () => number;
+};
+
+export type ArchiveOrgClientOpts = {
+ minGapMs?: number;
+ maxAttempts?: number;
+ baseBackoffMs?: number;
+ maxBackoffMs?: number;
+ cacheTtlMs?: number;
+ timeoutMs?: number;
+};
+
+const RETRYABLE = new Set([429, 502, 503, 504]);
+
+function defaultSleep(ms: number, signal?: AbortSignal): Promise<void> {
+ return new Promise((resolve, reject) => {
+ if (signal?.aborted) return reject(signal.reason ?? new Error("aborted"));
+ const t = setTimeout(() => {
+ signal?.removeEventListener("abort", onAbort);
+ resolve();
+ }, ms);
+ const onAbort = () => {
+ clearTimeout(t);
+ reject(signal?.reason ?? new Error("aborted"));
+ };
+ signal?.addEventListener("abort", onAbort, { once: true });
+ });
+}
+
+// Retry-After: delta-seconds or an HTTP date. Null when absent or unreadable.
+export function parseRetryAfterMs(value: string | null, nowMs: number): number | null {
+ if (!value) return null;
+ const v = value.trim();
+ if (/^\d+$/.test(v)) return Number(v) * 1000;
+ const at = Date.parse(v);
+ if (Number.isFinite(at)) return Math.max(0, at - nowMs);
+ return null;
+}
+
+export class ArchiveOrgClient {
+ private readonly deps: Required<ArchiveOrgClientDeps>;
+ private readonly opts: Required<ArchiveOrgClientOpts>;
+ private chain: Promise<unknown> = Promise.resolve();
+ private lastDoneAt = -Infinity;
+ private readonly cache = new Map<string, { at: number; value: ArchiveOrgItemMetadata }>();
+ // Requests actually sent (each attempt), for tests and logs.
+ requests = 0;
+
+ constructor(deps: ArchiveOrgClientDeps = {}, opts: ArchiveOrgClientOpts = {}) {
+ this.deps = {
+ fetch: deps.fetch ?? ((url, init) => fetch(url, init)),
+ now: deps.now ?? (() => Date.now()),
+ sleep: deps.sleep ?? defaultSleep,
+ random: deps.random ?? Math.random,
+ };
+ this.opts = {
+ minGapMs: opts.minGapMs ?? 2_000,
+ maxAttempts: opts.maxAttempts ?? 4,
+ baseBackoffMs: opts.baseBackoffMs ?? 5_000,
+ maxBackoffMs: opts.maxBackoffMs ?? 120_000,
+ cacheTtlMs: opts.cacheTtlMs ?? 6 * 60 * 60 * 1000,
+ timeoutMs: opts.timeoutMs ?? 60_000,
+ };
+ }
+
+ // Run `fn` after every earlier request (and its gap) has finished.
+ private serial<T>(fn: () => Promise<T>): Promise<T> {
+ const run = this.chain.then(fn, fn);
+ this.chain = run.catch(() => {});
+ return run;
+ }
+
+ private backoffMs(attempt: number): number {
+ const exp = Math.min(this.opts.maxBackoffMs, this.opts.baseBackoffMs * 2 ** (attempt - 1));
+ const jitter = 1 + (this.deps.random() * 0.4 - 0.2);
+ return Math.round(exp * jitter);
+ }
+
+ // One GET, with the gap before it and the retry policy around it.
+ private async getOnce(url: string, accept: string, signal?: AbortSignal): Promise<Response> {
+ let lastError = "";
+ for (let attempt = 1; attempt <= this.opts.maxAttempts; attempt++) {
+ const wait = this.lastDoneAt + this.opts.minGapMs - this.deps.now();
+ if (wait > 0) await this.deps.sleep(wait, signal);
+ this.requests++;
+ let res: Response | null = null;
+ try {
+ const timeout = AbortSignal.timeout(this.opts.timeoutMs);
+ res = await this.deps.fetch(url, {
+ signal: signal ? AbortSignal.any([signal, timeout]) : timeout,
+ redirect: "follow",
+ headers: { accept, "user-agent": ARCHIVE_ORG_USER_AGENT },
+ });
+ } catch (err) {
+ if (signal?.aborted) throw err;
+ lastError = (err as Error).message;
+ } finally {
+ this.lastDoneAt = this.deps.now();
+ }
+ if (res && res.ok) return res;
+ if (res && !RETRYABLE.has(res.status)) {
+ throw new ArchiveOrgRequestError(
+ `archive.org answered HTTP ${res.status} for ${url}`,
+ res.status,
+ false,
+ );
+ }
+ if (res) lastError = `HTTP ${res.status}`;
+ if (attempt === this.opts.maxAttempts) break;
+ const retryAfter = res ? parseRetryAfterMs(res.headers.get("retry-after"), this.deps.now()) : null;
+ const delay = Math.min(this.opts.maxBackoffMs, retryAfter ?? this.backoffMs(attempt));
+ await this.deps.sleep(delay, signal);
+ }
+ throw new ArchiveOrgRequestError(
+ `archive.org did not answer ${url} after ${this.opts.maxAttempts} attempts (${lastError}); stopping`,
+ null,
+ true,
+ );
+ }
+
+ getJson(url: string, signal?: AbortSignal): Promise<unknown> {
+ return this.serial(async () => {
+ const res = await this.getOnce(url, "application/json", signal);
+ return res.json();
+ });
+ }
+
+ // The item's metadata, once per `cacheTtlMs`.
+ async itemMetadata(identifier: string, signal?: AbortSignal): Promise<ArchiveOrgItemMetadata> {
+ const hit = this.cache.get(identifier);
+ if (hit && this.deps.now() - hit.at < this.opts.cacheTtlMs) return hit.value;
+ const raw = await this.getJson(archiveOrgMetadataUrl(identifier), signal);
+ const item = parseArchiveOrgItemMetadata(raw);
+ if (!item) {
+ throw new ArchiveOrgRequestError(`archive.org has no item "${identifier}"`, 404, false);
+ }
+ this.cache.set(identifier, { at: this.deps.now(), value: item });
+ return item;
+ }
+
+ // A small JSON file inside an item (a mirror's `.info.json`).
+ itemJsonFile(identifier: string, file: string, signal?: AbortSignal): Promise<unknown> {
+ return this.getJson(archiveOrgDownloadUrl(identifier, file), signal);
+ }
+}
+
+// THE process's client: one chain, one gap, one cache for every caller.
+export const archiveOrgClient = new ArchiveOrgClient();
diff --git a/common/lib/archiveOrgId.test.ts b/common/lib/archiveOrgId.test.ts
@@ -0,0 +1,95 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ archiveOrgDetailsUrl,
+ archiveOrgDownloadUrl,
+ archiveOrgTorrentUrl,
+ archiveOrgVideoId,
+ archiveOrgVideoIdFromNativeId,
+ parseArchiveOrgUrl,
+ youtubeIdFromFileName,
+ youtubeIdFromIdentifier,
+} from "./archiveOrgId";
+import { extractVideoId } from "./videoId";
+import { defaultWebpageUrl, detectPlatform, queueKeyForUrl } from "./platform";
+import { dataDirIdForUrl } from "../ytdlp/runYtdlp";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrgId.test.ts
+//
+// Every identifier, file and YouTube id here is invented.
+
+const ITEM = "example-item";
+const FILE = "Example Talk (Part 1)-AbC123xyz_9.mp4";
+
+test("archive.org item hosts are detected; the Wayback Machine is not", () => {
+ assert.equal(detectPlatform(`https://archive.org/details/${ITEM}`), "archiveorg");
+ assert.equal(detectPlatform(`https://www.archive.org/details/${ITEM}`), "archiveorg");
+ assert.equal(detectPlatform("https://web.archive.org/web/2020/https://example.com/"), null);
+ assert.equal(queueKeyForUrl(`https://archive.org/details/${ITEM}`), "platform:archiveorg");
+});
+
+test("a whole item's id is its identifier, from details, embed and download URLs", () => {
+ for (const url of [
+ `https://archive.org/details/${ITEM}`,
+ `https://archive.org/details/${ITEM}/`,
+ `https://archive.org/details/${ITEM}?autoplay=1`,
+ `https://archive.org/embed/${ITEM}`,
+ `https://archive.org/download/${ITEM}`,
+ ]) {
+ assert.equal(extractVideoId(url), ITEM, url);
+ }
+});
+
+test("a file inside an item gets a stable, filesystem-safe, unique id", () => {
+ const url = archiveOrgDetailsUrl({ identifier: ITEM, file: FILE });
+ assert.equal(url, `https://archive.org/details/${ITEM}/Example%20Talk%20(Part%201)-AbC123xyz_9.mp4`);
+ const id = extractVideoId(url)!;
+ assert.match(id, /^example-item__Example-Talk-Part-1-AbC123xyz_9-[0-9a-f]{8}$/);
+ // The same file by every URL form, encoded or not (yt-dlp unquotes `+`).
+ assert.equal(extractVideoId(`https://archive.org/embed/${ITEM}/${encodeURIComponent(FILE)}`), id);
+ assert.equal(extractVideoId(archiveOrgDownloadUrl(ITEM, FILE)), id);
+ assert.equal(
+ extractVideoId(`https://archive.org/details/${ITEM}/Example+Talk+(Part+1)-AbC123xyz_9.mp4`),
+ id,
+ );
+ // yt-dlp's native id for the entry maps to the same id.
+ assert.equal(archiveOrgVideoIdFromNativeId(`${ITEM}/${FILE}`), id);
+ assert.equal(archiveOrgVideoIdFromNativeId(ITEM), ITEM);
+ // The output path pins to it (a safe directory name).
+ assert.equal(dataDirIdForUrl(url), id);
+ // Paths that slug alike still differ.
+ const a = archiveOrgVideoId({ identifier: ITEM, file: "a b.mp4" });
+ const b = archiveOrgVideoId({ identifier: ITEM, file: "a_b.mp4" });
+ const c = archiveOrgVideoId({ identifier: ITEM, file: "a b.mkv" });
+ assert.equal(new Set([a, b, c]).size, 3);
+ // Sub-directories are part of the path.
+ assert.equal(
+ parseArchiveOrgUrl(`https://archive.org/details/${ITEM}/disc1/01%20Intro.mp3`)?.file,
+ "disc1/01 Intro.mp3",
+ );
+});
+
+test("URLs that name no item have no id", () => {
+ assert.equal(parseArchiveOrgUrl("https://archive.org/search?query=x"), null);
+ assert.equal(parseArchiveOrgUrl("https://archive.org/details/"), null);
+ assert.equal(extractVideoId("https://archive.org/search?query=x"), null);
+});
+
+test("the torrent and page URLs", () => {
+ assert.equal(archiveOrgTorrentUrl(ITEM), `https://archive.org/download/${ITEM}/${ITEM}_archive.torrent`);
+ assert.equal(defaultWebpageUrl("archiveorg", ITEM), `https://archive.org/details/${ITEM}`);
+ const fileId = archiveOrgVideoId({ identifier: ITEM, file: FILE });
+ assert.equal(defaultWebpageUrl("archiveorg", fileId), `https://archive.org/details/${ITEM}`);
+});
+
+test("YouTube ids read from mirror names", () => {
+ assert.equal(youtubeIdFromIdentifier("youtube-AbC123xyz_9"), "AbC123xyz_9");
+ assert.equal(youtubeIdFromIdentifier(ITEM), null);
+ assert.equal(youtubeIdFromFileName(FILE), "AbC123xyz_9");
+ assert.equal(youtubeIdFromFileName("Some Title [Zz9-Qq8_Ww7].webm"), "Zz9-Qq8_Ww7");
+ assert.equal(youtubeIdFromFileName("dir/Some Title [Zz9-Qq8_Ww7].info.json"), "Zz9-Qq8_Ww7");
+ // An eleven-letter lowercase word is not an id.
+ assert.equal(youtubeIdFromFileName("interview-performance.mp4"), null);
+ assert.equal(youtubeIdFromFileName("plain.mp4"), null);
+});
diff --git a/common/lib/archiveOrgId.ts b/common/lib/archiveOrgId.ts
@@ -0,0 +1,171 @@
+// archive.org ITEM AND FILE IDENTITY — what an archive.org URL names, and the
+// canonical video id (the `data/<id>/` dir name) it maps to.
+//
+// A leaf with no imports, like lib/videoId.ts (which calls into it): the roster
+// store, the browser and the export build all canonicalize URLs through here.
+//
+// AN ITEM is archive.org's unit of upload: `https://archive.org/details/<identifier>`.
+// An identifier is ASCII letters, digits, `.`, `-` and `_` — already a safe
+// directory name, so a WHOLE ITEM (one that holds a single media file) has the
+// identifier itself as its video id.
+//
+// A FILE INSIDE AN ITEM (a channel-archive item can hold a hundred and sixty
+// videos) is `https://archive.org/details/<identifier>/<file path>` — the form
+// yt-dlp's ArchiveOrg extractor resolves to that one entry. A file path is free
+// text (spaces, brackets, unicode, sub-directories), so its id is
+//
+// <identifier>__<slug of the path without its extension>-<8 hex>
+//
+// where the slug keeps `[A-Za-z0-9_-]` and folds every other run into one `-`
+// (cut to 48 characters), and the hex is a 32-bit FNV-1a hash of the EXACT file
+// path. The slug keeps the id readable; the hash keeps it unique — two paths
+// that slug alike ("a b.mp4", "a_b.mp4", "a b.mkv") still differ. Stable: the
+// same path always gives the same id, so a re-import lands in the same dir.
+//
+// `/embed/` and `/download/` URLs name the same things and canonicalize to the
+// same ids. `web.archive.org` (the Wayback Machine) is NOT an item host and is
+// not handled here.
+
+export type ArchiveOrgRef = {
+ identifier: string;
+ // The file's path inside the item, decoded; absent for the whole item.
+ file?: string;
+};
+
+const IDENTIFIER_RE = /^[A-Za-z0-9][A-Za-z0-9._-]*$/;
+
+// Hosts that serve items. Not `web.archive.org` (Wayback) and not the
+// `iaNNNNNN.us.archive.org` storage nodes a download redirects to.
+export function isArchiveOrgItemHost(host: string): boolean {
+ const h = host.toLowerCase();
+ return h === "archive.org" || h === "www.archive.org";
+}
+
+// yt-dlp reads the path with `unquote_plus`; so do we, so the file we name is
+// the file it resolves.
+function unquotePlus(s: string): string {
+ try {
+ return decodeURIComponent(s.replace(/\+/g, " "));
+ } catch {
+ return s;
+ }
+}
+
+// The item (and file) an archive.org URL names, or null for anything else.
+export function parseArchiveOrgUrl(url: string): ArchiveOrgRef | null {
+ let u: URL;
+ try {
+ u = new URL(url);
+ } catch {
+ return null;
+ }
+ if (!isArchiveOrgItemHost(u.hostname)) return null;
+ const segs = u.pathname.split("/").filter(Boolean);
+ if (segs.length < 2) return null;
+ const kind = segs[0];
+ if (kind !== "details" && kind !== "embed" && kind !== "download") return null;
+ const identifier = unquotePlus(segs[1]);
+ if (!IDENTIFIER_RE.test(identifier)) return null;
+ const rest = segs.slice(2).map(unquotePlus);
+ const file = rest.length > 0 ? rest.join("/") : undefined;
+ // A download URL with no file is the item's file listing, i.e. the item.
+ return file ? { identifier, file } : { identifier };
+}
+
+// FNV-1a, 32 bits, over the UTF-16 code units — dependency-free and the same in
+// every runtime this module is imported into.
+function fnv1a32(s: string): string {
+ let h = 0x811c9dc5;
+ for (let i = 0; i < s.length; i++) {
+ h ^= s.charCodeAt(i);
+ h = Math.imul(h, 0x01000193) >>> 0;
+ }
+ return h.toString(16).padStart(8, "0");
+}
+
+const SLUG_MAX = 48;
+
+function fileSlug(file: string): string {
+ const base = file.replace(/\.[A-Za-z0-9]{1,8}$/, "");
+ const slug = base
+ .replace(/[^A-Za-z0-9_-]+/g, "-")
+ .replace(/-{2,}/g, "-")
+ .replace(/^-+|-+$/g, "")
+ .slice(0, SLUG_MAX)
+ .replace(/-+$/, "");
+ return slug || "file";
+}
+
+// The canonical video id of an item, or of one file inside it.
+export function archiveOrgVideoId(ref: ArchiveOrgRef): string {
+ if (!ref.file) return ref.identifier;
+ return `${ref.identifier}__${fileSlug(ref.file)}-${fnv1a32(ref.file)}`;
+}
+
+// yt-dlp's own id for an archive.org record: the identifier, or
+// `<identifier>/<file path>` for an entry of a multi-file item. The canonical id
+// of that record, or null when it is not one.
+export function archiveOrgVideoIdFromNativeId(
+ nativeId: string | null | undefined,
+): string | null {
+ if (!nativeId) return null;
+ const slash = nativeId.indexOf("/");
+ const identifier = slash < 0 ? nativeId : nativeId.slice(0, slash);
+ if (!IDENTIFIER_RE.test(identifier)) return null;
+ const file = slash < 0 ? "" : nativeId.slice(slash + 1);
+ return archiveOrgVideoId(file ? { identifier, file } : { identifier });
+}
+
+function encodePath(file: string): string {
+ return file.split("/").map(encodeURIComponent).join("/");
+}
+
+// The item page, or the file's own page inside it — the URL a record keeps as
+// its `webpage_url`, and the one every link to it uses.
+export function archiveOrgDetailsUrl(ref: ArchiveOrgRef): string {
+ const base = `https://archive.org/details/${ref.identifier}`;
+ return ref.file ? `${base}/${encodePath(ref.file)}` : base;
+}
+
+// The file's bytes.
+export function archiveOrgDownloadUrl(identifier: string, file: string): string {
+ return `https://archive.org/download/${identifier}/${encodePath(file)}`;
+}
+
+// The item's BitTorrent file. archive.org derives one for every item, named
+// `<identifier>_archive.torrent`; it covers every file in the item.
+export function archiveOrgTorrentUrl(identifier: string): string {
+ return `https://archive.org/download/${identifier}/${identifier}_archive.torrent`;
+}
+
+// The item's metadata API (one JSON document: the item's fields and its file
+// list).
+export function archiveOrgMetadataUrl(identifier: string): string {
+ return `https://archive.org/metadata/${identifier}`;
+}
+
+// THE YOUTUBE ID A MIRROR CARRIES, when it says so in its name.
+//
+// `youtube-<id>` is the identifier tubeup (the usual YouTube → archive.org
+// mirroring tool) gives an item; a file yt-dlp named carries the id as
+// `<title>-<id>.<ext>` or `<title> [<id>].<ext>`. The bracketed form is
+// unambiguous. The dashed one is a guess at the last eleven characters before
+// the extension, so it is only taken when the token is not plain lowercase
+// letters — a real id is random base64 and almost never is, while an English
+// word of eleven letters ("performance") always is.
+const YT_ID = "[A-Za-z0-9_-]{11}";
+
+export function youtubeIdFromIdentifier(identifier: string): string | null {
+ const m = new RegExp(`^youtube-(${YT_ID})$`).exec(identifier);
+ return m ? m[1] : null;
+}
+
+export function youtubeIdFromFileName(file: string): string | null {
+ const name = file.split("/").pop() ?? file;
+ const stem = name.replace(/(\.[A-Za-z0-9]{1,8})+$/, "");
+ const bracket = new RegExp(`\\[(${YT_ID})\\]$`).exec(stem);
+ if (bracket) return bracket[1];
+ const dashed = new RegExp(`(?:^|[-_ ])(${YT_ID})$`).exec(stem);
+ if (dashed && /[A-Z0-9_-]/.test(dashed[1])) return dashed[1];
+ return null;
+}
diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts
@@ -160,7 +160,7 @@ export const CHANNEL_CONFIG_FIELD_DOCS: FieldDocs<ChannelConfig> = {
'Social channels only: the bare account handle (a leading "@" is stripped). Derived from `url` at creation but stored, so a later URL-format change upstream cannot silently re-point ingest at a different account.',
postPagePauseSeconds:
"Social channels that are read page by page (a forum thread) only: the pause between two page loads, in seconds; each pause is jittered to 0.85–1.65× of it. Absent = the fetcher's own (12 s, so 10–20 s); floored at 5, capped at 600.",
- platform: "The source platform (youtube, rumble, …). An unknown value is dropped.",
+ platform: "The source platform: youtube, rumble, odysee, twitch, kick, archiveorg (archive.org items — imported, never listed; see README.md, \"archive.org items\"), twitter, bluesky or xenforo (a forum thread). An unknown value is dropped.",
name: "Display name.",
url: "The channel / playlist / account URL syncs enumerate. Absent = the channel is never auto-synced.",
audioFormat: '`"m4a"`, `"mp3"` or `"opus"`: the audio a transcribe-handling download keeps.',
diff --git a/common/lib/detectPlatform.mjs b/common/lib/detectPlatform.mjs
@@ -34,6 +34,9 @@ export function detectPlatform(url) {
if (host === "x.com" || host.endsWith(".x.com")) return "twitter";
if (host.endsWith("twitter.com")) return "twitter";
if (host === "bsky.app" || host.endsWith(".bsky.app")) return "bluesky";
+ // archive.org ITEMS only. Not web.archive.org (the Wayback Machine's page
+ // captures are a different kind of record) — see lib/archiveOrgId.ts.
+ if (host === "archive.org" || host === "www.archive.org") return "archiveorg";
if (XENFORO_HOSTS.some((h) => host === h || host.endsWith(`.${h}`))) {
return "xenforo";
}
diff --git a/common/lib/metadataHistory.ts b/common/lib/metadataHistory.ts
@@ -60,6 +60,10 @@ export const METADATA_HISTORY_WRITERS = [
// date, description and duration from the channel's RSS feed, for a record
// imported by its enclosure URL, which carries none of them.
"feed-backfill",
+ // An archive.org record corrected from its provenance (lib/archiveOrg.ts
+ // archiveOrgMetadataPatch): the file's page and title instead of the item's,
+ // and a mirror's original title, date and uploader.
+ "archiveorg-provenance",
] as const;
export type MetadataHistoryWriter = (typeof METADATA_HISTORY_WRITERS)[number];
diff --git a/common/lib/momentUrl.ts b/common/lib/momentUrl.ts
@@ -71,8 +71,10 @@ export function platformMomentUrl(
case "twitch":
u.searchParams.set("t", twitchTime(secs));
return u.toString();
- // Rumble / Kick / unknown: the watch page has no dependable start param —
- // return the plain webpage URL rather than an invalid seek.
+ // Rumble / Kick / archive.org / unknown: the watch page has no dependable
+ // start param — return the plain webpage URL rather than an invalid seek.
+ // (archive.org's own player takes none we can rely on; the archive's
+ // viewer plays the file itself and seeks it — PlayerProvider.)
default:
return webpageUrl;
}
@@ -163,8 +165,9 @@ export function viewerMomentBaseUrl(
// The platform base. Unlike platformMomentUrl (which falls back to the bare
// webpage URL), a base MUST be appendable — so only platforms whose time param
// takes raw seconds qualify. Twitch is excluded (its `t` takes an `XhYmZs`
-// token, so appending an integer would be an invalid seek); Rumble/Kick/unknown
-// have no dependable start param at all. Null in every non-appendable case.
+// token, so appending an integer would be an invalid seek); Rumble/Kick/
+// archive.org/unknown have no dependable start param at all. Null in every
+// non-appendable case.
export function platformMomentBaseUrl(
webpageUrl: string | null | undefined,
platform: Platform | null | undefined,
diff --git a/common/lib/platform.ts b/common/lib/platform.ts
@@ -8,6 +8,9 @@ export type Platform =
| "odysee"
| "twitch"
| "kick"
+ // archive.org items (lib/archiveOrgId.ts): a whole item, or one file inside
+ // a multi-file item.
+ | "archiveorg"
| "twitter"
| "bluesky"
| "xenforo";
@@ -18,6 +21,7 @@ export const PLATFORM_VALUES: ReadonlyArray<Platform> = [
"odysee",
"twitch",
"kick",
+ "archiveorg",
"twitter",
"bluesky",
"xenforo",
@@ -52,6 +56,13 @@ export function defaultWebpageUrl(platform: Platform, id: string): string {
if (platform === "twitch") return `https://www.twitch.tv/videos/${id}`;
// Kick's canonical id is the VOD UUID; /video/<uuid> resolves to the VOD.
if (platform === "kick") return `https://kick.com/video/${id}`;
+ // A whole item's id IS its identifier. A file's id is a slug + hash the file
+ // path cannot be recovered from, so this is the right page only for a whole
+ // item; every archived record carries its own `webpage_url`.
+ if (platform === "archiveorg") {
+ const file = /^(.+?)__.*-[0-9a-f]{8}$/.exec(id);
+ return `https://archive.org/details/${file ? file[1] : id}`;
+ }
// Social posts: /i/status/<id> resolves without knowing the handle. Bluesky
// has no handle-free permalink, so this is only a last-resort fallback —
// every archived post carries its own canonical `url` (see postPermalink).
diff --git a/common/lib/report/views.ts b/common/lib/report/views.ts
@@ -135,6 +135,14 @@ export type RecordView = {
// The record on its own platform at the cited time (lib/momentUrl.ts
// platformMomentUrl), or the post's own URL.
originalUrl?: string;
+ // What `originalUrl` is, when "Original" would not say: "archive.org" for an
+ // archive.org record, "YouTube" for an archive.org mirror of a YouTube
+ // upload (whose originalUrl is the upload at the cited second).
+ originalLabel?: string;
+ // Where a reader can fetch the recording itself to check it, derived from
+ // the record's provenance (lib/archiveOrg.ts archiveOrgCitationLinks): the
+ // archive.org page and the item's torrent. Absent for a record with none.
+ downloads?: { label: string; url: string }[];
// The record in this site's corpus (`/?v=<channel>/<id>&t=<s>`): a FULL site
// only — a cited site has no corpus to open.
corpusUrl?: string;
diff --git a/common/lib/sidecar-server.test.ts b/common/lib/sidecar-server.test.ts
@@ -6,6 +6,7 @@ import path from "node:path";
import { SIDECAR_FILENAMES, sidecar, sidecarField } from "./sidecar-server";
import { SUB_FILE_RE } from "./videoStatus";
// Every sidecar module, so the enumeration below sees every declaration.
+import "./archiveOrg-server";
import "./attribution-server";
import "./diarization-server";
import "./digest-server";
@@ -41,10 +42,11 @@ async function scratch(): Promise<string> {
return mkdtemp(path.join(os.tmpdir(), "sidecar-"));
}
-test("every declared sidecar filename escapes SUB_FILE_RE, and all ten are declared", () => {
+test("every declared sidecar filename escapes SUB_FILE_RE, and all eleven are declared", () => {
assert.deepEqual([...SIDECAR_FILENAMES].sort(), [
"ai-digest.json",
"ai-digest.overrides.json",
+ "archiveorg.json",
"attribution.json",
"availability.json",
"diarization.json",
diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts
@@ -1,7 +1,9 @@
import { readFile } from "node:fs/promises";
import path from "node:path";
import { formatDate, formatDuration } from "./format";
-import { defaultWebpageUrl } from "./platform";
+import { defaultWebpageUrl, detectPlatform } from "./platform";
+import { archiveOrgPlayableUrl } from "./archiveOrg";
+import { archiveOrgVideoIdFromNativeId } from "./archiveOrgId";
import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts";
import type { MediaType, VideoStat, VideoStatus } from "./stats";
import type { VideoState } from "./availability";
@@ -39,6 +41,10 @@ export type RawMetadata = {
// yt-dlp's coarse kind: "video" | "livestream" | "short" (YouTube). Absent on
// platforms that don't distinguish, where we fall back to is/was_live.
media_type?: string;
+ // Every format the extractor offered. Read only for archive.org, where each
+ // is a plain download URL and one of them is what the player plays
+ // (lib/archiveOrg.ts archiveOrgPlayableUrl).
+ formats?: unknown;
};
// Read and parse a video's metadata.info.json into the typed RawMetadata
@@ -64,12 +70,25 @@ export function loadRawMetadataFromDir(
return loadRawMetadata(path.join(videoDir, "metadata.info.json"));
}
+// The platform a record is from, by yt-dlp's extractor. archive.org's
+// extractor is `ArchiveOrg` (key) / `archive.org` (name) — NOT web.archive.org's
+// `YoutubeWebArchive`, which is a YouTube video.
+//
+// AN UNKNOWN EXTRACTOR: the record's own page decides when its host is one the
+// app knows (lib/detectPlatform.mjs); otherwise "youtube", as it always has
+// been — the generic extractor (a podcast episode imported by its enclosure
+// URL) still lands there, and changing that would relabel records already
+// published.
export function platformFromMetadata(meta: RawMetadata): Platform {
const key = meta.extractor_key ?? meta.extractor ?? "";
if (/^rumble/i.test(key)) return "rumble";
if (/^lbry/i.test(key)) return "odysee";
if (/^twitch/i.test(key)) return "twitch";
if (/^kick/i.test(key)) return "kick";
+ if (/^archive\.?org$/i.test(key)) return "archiveorg";
+ if (/^youtube/i.test(key)) return "youtube";
+ const fromPage = detectPlatform(meta.webpage_url);
+ if (fromPage === "archiveorg") return fromPage;
return "youtube";
}
@@ -97,10 +116,14 @@ export function summarize(
): TranscriptSummary {
const isLivestream = isLivestreamMetadata(meta);
const platform = platformFromMetadata(meta);
+ // archive.org: yt-dlp's id for one file of an item is `<identifier>/<path>`,
+ // which is not a slug; the canonical id (lib/archiveOrgId.ts) is.
const id =
platform === "odysee"
? (meta.webpage_url_basename ?? meta.id ?? videoDir)
- : (meta.id ?? videoDir);
+ : platform === "archiveorg"
+ ? (archiveOrgVideoIdFromNativeId(meta.id) ?? videoDir)
+ : (meta.id ?? videoDir);
const dateFromDir = videoDir.match(/^(\d{8})(?:_|$)/)?.[1];
return {
slug: `${channelSlug}/${id}`,
@@ -118,6 +141,10 @@ export function summarize(
webpageUrl: meta.webpage_url ?? defaultWebpageUrl(platform, id),
// Kick VODs play from a persisted HLS manifest (no iframe embed exists).
hlsUrl: platform === "kick" ? meta.manifest_url : undefined,
+ // archive.org plays the file itself in a native <video>, which seeks.
+ ...(platform === "archiveorg"
+ ? { mediaUrl: archiveOrgPlayableUrl(meta) }
+ : {}),
};
}
diff --git a/common/lib/transcripts.ts b/common/lib/transcripts.ts
@@ -26,6 +26,11 @@ export type TranscriptSummary = {
// HLS master-playlist URL for platforms with no iframe embed (Kick VODs).
// Undefined for every other platform. Played via react-player/file + hls.js.
hlsUrl?: string;
+ // A plain media file URL a native <video> plays and seeks (archive.org: the
+ // record's browser-playable file, lib/archiveOrg.ts). Undefined everywhere
+ // else. New with the archiveorg platform, so no cached or normalized record
+ // predates it and no cache version moved.
+ mediaUrl?: string;
// Curated tag ids (common/lib/curatedTags.ts) — the operator's cross-channel
// vocabulary, NOT the yt-dlp keywords in `tags` above. OMITTED when empty, so
// an untagged corpus's pages stay byte-identical to the ones already on disk
diff --git a/common/lib/videoId.ts b/common/lib/videoId.ts
@@ -1,17 +1,27 @@
// The canonical video id derived from a video URL — the name every video's
-// data/<id>/ dir carries, on every platform. Deliberately a leaf module with no
-// imports at all: it lives here rather than in ytdlp/runYtdlp.ts (its original
-// home, which still re-exports it) so that low-level stores like
+// data/<id>/ dir carries, on every platform. Deliberately a leaf module whose
+// one import (archiveOrgId.ts) is itself a leaf with no imports: it lives here
+// rather than in ytdlp/runYtdlp.ts (its original home, which still re-exports it) so that low-level stores like
// controller/rosterStore.ts can canonicalize a URL without pulling in execa,
// the settings loader and the whole download pipeline.
//
// Canonical is NOT the same as yt-dlp's native extractor id: the two coincide
// on YouTube and diverge everywhere else. See archiveIdForUrl in runYtdlp.ts
// for the native-id resolution that reads metadata.info.json.
+
+import { isArchiveOrgItemHost, parseArchiveOrgUrl, archiveOrgVideoId } from "./archiveOrgId";
+
export function extractVideoId(url: string): string | null {
try {
const u = new URL(url);
const host = u.hostname.toLowerCase();
+ if (isArchiveOrgItemHost(host)) {
+ // A whole item → its identifier; one file inside an item → a stable
+ // `<identifier>__<slug>-<hash>` (lib/archiveOrgId.ts). A URL that names
+ // no item (a search page, a collection listing) has no id.
+ const ref = parseArchiveOrgUrl(url);
+ return ref ? archiveOrgVideoId(ref) : null;
+ }
if (host.endsWith("youtube.com") || host === "youtu.be") {
const v = u.searchParams.get("v");
if (v) return v;
diff --git a/common/publish/composeReports.ts b/common/publish/composeReports.ts
@@ -68,6 +68,8 @@ import { parseVtt, type Cue } from "../lib/vtt";
import { parseTranscriptJson } from "../lib/whisper";
import { WHISPER_FILENAME, isEnglishVtt, resolvePrimaryVtt } from "../lib/videoStatus";
import { platformMomentUrl } from "../lib/momentUrl";
+import { archiveOrgCitationLinks, type ArchiveOrgProvenance } from "../lib/archiveOrg";
+import { loadArchiveOrgProvenance } from "../lib/archiveOrg-server";
import type { Platform } from "../lib/platform";
import { readAllPosts } from "../lib/posts-server";
import type { Post } from "../lib/posts";
@@ -218,6 +220,9 @@ type CitedRecord = {
// first). A served `en` track can be a rewrite of what was said, so a
// quote is checked against each and the best match is the one shown.
tracks: { name: string; cues: Cue[] }[];
+ // An archive.org record's provenance (its torrent, a mirror's original);
+ // null for every other record.
+ archiveOrg: ArchiveOrgProvenance | null;
};
// The English VTT tracks of a video dir, `en-orig` first, then the order
@@ -296,7 +301,8 @@ export async function readCitedRecord(
if (v.length > 0) tracks.push({ name, cues: v });
}
if (tracks.length === 0 && cues.length > 0) tracks.push({ name: "cues", cues });
- return { summary, cues, tracks };
+ const archiveOrg = summary.platform === "archiveorg" ? await loadArchiveOrgProvenance(dir) : null;
+ return { summary, cues, tracks, archiveOrg };
}
const isoDay = (uploadDate: string | undefined): string | undefined =>
@@ -607,9 +613,26 @@ export async function resolveSiteReports(opts: ResolveSiteReportsOptions): Promi
corpusUrl: cited ? undefined : corpusLink(`${c.channel}/${c.id}`, { vm: "post" }),
});
}
- const { summary } = (await recordOf(c.channel, c.id))!;
+ const { summary, archiveOrg } = (await recordOf(c.channel, c.id))!;
const audioOnly = isAudioOnlyPlatform(config?.platform);
const seconds = Math.max(0, Math.floor(c.start));
+ if (summary.platform === "archiveorg") {
+ // archive.org: the original (YouTube at the second, for a mirror; else
+ // the archive.org page) plus the downloads a reader can check it from.
+ const links = archiveOrgCitationLinks({ webpageUrl: summary.webpageUrl, provenance: archiveOrg, seconds: c.start });
+ return defined({
+ channel: c.channel,
+ channelTitle: config?.name ?? (summary.channel || undefined),
+ id: c.id,
+ title: summary.title,
+ date: isoDay(summary.uploadDate),
+ platform: summary.platform,
+ originalUrl: links.original?.url ?? (summary.webpageUrl || undefined),
+ originalLabel: links.original?.label,
+ downloads: links.downloads.length > 0 ? links.downloads : undefined,
+ corpusUrl: cited ? undefined : corpusLink(summary.slug ?? `${c.channel}/${summary.id}`, seconds > 0 ? { t: String(seconds) } : {}),
+ });
+ }
return defined({
channel: c.channel,
channelTitle: config?.name ?? (summary.channel || undefined),
diff --git a/common/publish/composeReportsArchiveOrg.test.ts b/common/publish/composeReportsArchiveOrg.test.ts
@@ -0,0 +1,60 @@
+// A cited archive.org record carries its provenance into compose, and the
+// links a citation of it shows are derived from that (lib/archiveOrg.ts
+// archiveOrgCitationLinks): YouTube at the second for a mirror, the
+// archive.org page and the torrent as downloads. Every name here is invented.
+//
+// Run with: node_modules/.bin/tsx --test publish/composeReportsArchiveOrg.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { readCitedRecord } from "./composeReports";
+import { archiveOrgCitationLinks, buildArchiveOrgProvenance } from "../lib/archiveOrg";
+import { archiveOrgDetailsUrl, archiveOrgVideoId } from "../lib/archiveOrgId";
+
+const ITEM = "example-item";
+const FILE = "First Upload-AbC123xyz_9.mp4";
+
+test("readCitedRecord loads an archive.org record's provenance; others get null", async () => {
+ const channels = mkdtempSync(path.join(tmpdir(), "compose-archiveorg-"));
+ const id = archiveOrgVideoId({ identifier: ITEM, file: FILE });
+ const dir = path.join(channels, "demo-archive", "data", id);
+ mkdirSync(dir, { recursive: true });
+ writeFileSync(
+ path.join(dir, "metadata.info.json"),
+ JSON.stringify({
+ id: `${ITEM}/${FILE}`,
+ extractor_key: "ArchiveOrg",
+ title: "First Upload (original)",
+ upload_date: "20230405",
+ webpage_url: archiveOrgDetailsUrl({ identifier: ITEM, file: FILE }),
+ }),
+ );
+ const prov = buildArchiveOrgProvenance({
+ ref: { identifier: ITEM, file: FILE },
+ item: {
+ metadata: { identifier: ITEM, title: "Example Archive" },
+ files: [{ name: FILE, source: "original" }, { name: `${ITEM}_archive.torrent`, source: "metadata" }],
+ },
+ infoJson: { id: "AbC123xyz_9", extractor_key: "Youtube", upload_date: "20230405" },
+ fetchedAt: "2026-01-01T00:00:00.000Z",
+ });
+ writeFileSync(path.join(dir, "archiveorg.json"), JSON.stringify(prov));
+
+ const rec = await readCitedRecord(channels, "demo-archive", id);
+ assert.ok(rec);
+ assert.equal(rec.summary.platform, "archiveorg");
+ assert.equal(rec.summary.id, id);
+ assert.deepEqual(rec.archiveOrg, prov);
+ const links = archiveOrgCitationLinks({ webpageUrl: rec.summary.webpageUrl, provenance: rec.archiveOrg, seconds: 12 });
+ assert.equal(links.original?.label, "YouTube");
+ assert.equal(links.original?.url, "https://www.youtube.com/watch?v=AbC123xyz_9&t=12s");
+ assert.deepEqual(links.downloads.map((d) => d.label), ["archive.org", "torrent"]);
+
+ const yt = path.join(channels, "demo-archive", "data", "AbC123xyz_9");
+ mkdirSync(yt, { recursive: true });
+ writeFileSync(path.join(yt, "metadata.info.json"), JSON.stringify({ id: "AbC123xyz_9", extractor_key: "Youtube" }));
+ assert.equal((await readCitedRecord(channels, "demo-archive", "AbC123xyz_9"))?.archiveOrg, null);
+});
diff --git a/common/ytdlp/archiveOrgProvenanceHook.test.ts b/common/ytdlp/archiveOrgProvenanceHook.test.ts
@@ -0,0 +1,80 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { chmod, mkdtemp, readFile, rm, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import type { ChannelConfig } from "../lib/channelConfig";
+import { downloadOneManaged } from "./downloadOneManaged";
+import { archiveOrgDetailsUrl, archiveOrgVideoId } from "../lib/archiveOrgId";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test ytdlp/archiveOrgProvenanceHook.test.ts
+//
+// THE archive.org PROVENANCE STEP RUNS RIGHT AFTER A PREFETCH THAT SUCCEEDED,
+// and never after one that failed. The yt-dlp here is a temp script: the
+// prefetch (`--skip-download`) writes a metadata.info.json shaped like the
+// ArchiveOrg extractor's (the ITEM's page for one file of many) and exits 0 or
+// 1; the download pass fails. The step itself is a spy. Every name is invented.
+
+const ITEM = "example-item";
+const FILE = "Clip One.mp4";
+const URL = archiveOrgDetailsUrl({ identifier: ITEM, file: FILE });
+const ID = archiveOrgVideoId({ identifier: ITEM, file: FILE });
+
+async function runWith(prefetchExit: 0 | 1) {
+ const root = await mkdtemp(path.join(tmpdir(), "archiveorg-hook-"));
+ try {
+ const bin = path.join(root, "fake-ytdlp.mjs");
+ await writeFile(
+ bin,
+ `#!/usr/bin/env node
+import { mkdirSync, writeFileSync } from "node:fs";
+import path from "node:path";
+const args = process.argv.slice(2);
+if (!args.includes("--skip-download")) { console.error("ERROR: no download in this test"); process.exit(1); }
+const o = args.find((a) => a.startsWith("infojson:"));
+const file = o.slice("infojson:".length) + ".info.json";
+mkdirSync(path.dirname(file), { recursive: true });
+writeFileSync(file, JSON.stringify({ id: ${JSON.stringify(`${ITEM}/${FILE}`)}, extractor_key: "ArchiveOrg", title: "Example Archive", webpage_url: "https://archive.org/details/${ITEM}" }));
+process.exit(${prefetchExit});
+`,
+ );
+ await chmod(bin, 0o755);
+ const paths = { channelsDir: path.join(root, "channels"), ytdlpBin: bin } as Paths;
+ const calls: { videoDir: string; videoUrl: string; infoAtCall: string }[] = [];
+ await downloadOneManaged({
+ channelSlug: "c",
+ channelConfig: { handling: "transcribe", platform: "archiveorg" } as ChannelConfig,
+ paths,
+ videoUrl: URL,
+ onLog: () => {},
+ signal: new AbortController().signal,
+ archiveOrgProvenance: async (o) => {
+ calls.push({
+ videoDir: o.videoDir,
+ videoUrl: o.videoUrl,
+ infoAtCall: await readFile(path.join(o.videoDir, "metadata.info.json"), "utf8").catch(() => ""),
+ });
+ return null;
+ },
+ });
+ return { calls, dataDir: path.join(paths.channelsDir, "c", "data") };
+ } finally {
+ await rm(root, { recursive: true, force: true });
+ }
+}
+
+test("after a prefetch that succeeded, the step runs once on the record's own dir", async () => {
+ const { calls, dataDir } = await runWith(0);
+ assert.equal(calls.length, 1);
+ assert.equal(calls[0].videoDir, path.join(dataDir, ID));
+ assert.equal(calls[0].videoUrl, URL);
+ // The prefetch's record is already on disk when it runs.
+ assert.match(calls[0].infoAtCall, /"extractor_key":"ArchiveOrg"/);
+});
+
+test("after a prefetch that failed, it does not run", async () => {
+ const { calls } = await runWith(1);
+ assert.equal(calls.length, 0);
+});
diff --git a/common/ytdlp/channelArgs.test.ts b/common/ytdlp/channelArgs.test.ts
@@ -147,3 +147,20 @@ test("the live pace (the shared state) reaches every channelExtraArgs call", ()
globalThis.__yttAutoQueueState__ = undefined;
}
});
+
+test("archive.org is paced and backs off, and keeps the fetched page as webpage_url", () => {
+ const args = platformArgs("archiveorg");
+ assert.equal(staticSleepRequestsSeconds("archiveorg"), 2);
+ assert.deepEqual(args.slice(0, 2), ["--sleep-requests", "2"]);
+ assert.ok(args.includes("http:exp=2:120"));
+ const i = args.indexOf("--parse-metadata");
+ const [field, re] = [args[i + 1].slice(0, args[i + 1].indexOf(":")), args[i + 1].slice(args[i + 1].indexOf(":") + 1)];
+ assert.equal(field, "original_url");
+ // The regex (Python's syntax, which this subset shares) takes an archive.org
+ // page and nothing else.
+ const js = new RegExp(`^${re.replace("(?P<webpage_url>", "(?<webpage_url>")}$`);
+ assert.equal(js.exec("https://archive.org/details/example-item/a%20b.mp4")?.groups?.webpage_url, "https://archive.org/details/example-item/a%20b.mp4");
+ assert.equal(js.exec("https://www.youtube.com/watch?v=AbC123xyz_9"), null);
+ // No parallel transfer is ever asked for.
+ assert.equal(args.some((a) => /concurrent|downloader|^-N$/.test(a)), false);
+});
diff --git a/common/ytdlp/downloadFormat.test.ts b/common/ytdlp/downloadFormat.test.ts
@@ -1,6 +1,7 @@
import { test } from "node:test";
import assert from "node:assert/strict";
import {
+ ARCHIVE_ORG_AUTO_FORMAT_SELECTOR,
DOWNLOAD_FORMAT_LABELS,
DOWNLOAD_FORMAT_PRESETS,
ORIGINAL_SOURCE_FORMAT_SELECTOR,
@@ -102,3 +103,12 @@ test("sourceVideoQualityForMaxHeight: a cap at or under 720 is video_720, above
assert.equal(sourceVideoQualityForMaxHeight(1080), "original");
assert.equal(sourceVideoQualityForMaxHeight(2160), "original");
});
+
+test("auto on archive.org takes the uploader's original, an audio item's mp3", () => {
+ const sel = resolveDownloadFormatSelector("auto", "archiveorg");
+ assert.equal(sel, ARCHIVE_ORG_AUTO_FORMAT_SELECTOR);
+ assert.match(sel, /^b\[format_note=original\]\[ext=mp4\]\//);
+ assert.ok(sel.split("/").includes("mp3"));
+ // An explicit preset still applies literally.
+ assert.equal(resolveDownloadFormatSelector("bestaudio", "archiveorg"), "bestaudio/worst");
+});
diff --git a/common/ytdlp/downloadFormat.ts b/common/ytdlp/downloadFormat.ts
@@ -4,7 +4,9 @@ import type { Platform } from "../lib/platform";
// an enum rather than a free-form `-f` string so it validates like audioFormat).
// "auto" is platform-aware: Odysee/LBRY only serves the full-length audio in its
// `original` format (every HLS rung is CDN-truncated to a few minutes), so auto
-// prefers `original` there and the historical `bestaudio/worst` everywhere else.
+// prefers `original` there, archive.org gets the uploader's original file
+// (ARCHIVE_ORG_AUTO_FORMAT_SELECTOR), and the historical `bestaudio/worst`
+// everywhere else.
export type DownloadFormatPreset =
| "auto"
| "original"
@@ -126,12 +128,32 @@ export function resolveDownloadFormatSelector(
].join("/");
case "auto":
default:
+ if (platform === "archiveorg") return ARCHIVE_ORG_AUTO_FORMAT_SELECTOR;
return platform === "odysee"
? "original/bestaudio/worst"
: "bestaudio/worst";
}
}
+// archive.org's "auto": the ORIGINAL file the uploader put up, in a common
+// container, and for an audio item its MP3 (or Ogg) — not "bestaudio/worst".
+// yt-dlp knows no codecs for an archive.org format (only its extension and
+// `format_note`, "original" or "derivative"), so `bestaudio` never matches a
+// video file and `worst` would take whichever transcode sorts last. A video
+// original in another container (.avi, .mpeg) is the next rung; a FLAC/WAV
+// original loses to its MP3 derivative, a fraction of the bytes for the same
+// transcript. Anything at all is the last rung.
+export const ARCHIVE_ORG_AUTO_FORMAT_SELECTOR = [
+ "b[format_note=original][ext=mp4]",
+ "b[format_note=original][ext=mkv]",
+ "b[format_note=original][ext=webm]",
+ "b[format_note=original][ext=mp3]",
+ "mp3",
+ "ogg",
+ "b[format_note=original]",
+ "b",
+].join("/");
+
// The override chain mirrors audioFormat: per-run override beats the per-channel
// default beats the global default; "auto" is the baseline when nothing is set.
export function resolveDownloadFormatPreset(opts: {
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -64,6 +64,7 @@ import {
type MetadataScanEntry,
} from "../controller/metadataScanStore";
import { detectPlatform, type Platform } from "../lib/platform";
+import { ensureArchiveOrgProvenance } from "../lib/archiveOrg-server";
import { probeMediaDurationSec } from "./ffprobeDuration";
import {
isShortAudio,
@@ -184,6 +185,9 @@ export type ManagedDownloadOpts = {
// "video_720" = the ≤720p H.264 selector (downloadFormat.ts). Never touches
// the audio-only selector above.
persistFormatPreset?: SourceVideoQuality;
+ // Test seam: the archive.org provenance step (lib/archiveOrg-server.ts).
+ // Every production caller passes none.
+ archiveOrgProvenance?: typeof ensureArchiveOrgProvenance;
};
// When `reuseInfoJson` is true, the real download reuses the metadata the
@@ -496,6 +500,9 @@ const PREFETCH_OWN_FILES: ReadonlySet<string> = new Set([
// one a previous rejection wrote before this rule existed. It records what
// happened, never what is on disk, so it is ours to drop with the rest.
"download-outcome.json",
+ // An archive.org record's provenance, written by this pass right after the
+ // prefetch (lib/archiveOrg-server.ts) and fetchable again.
+ "archiveorg.json",
// NOT metadata.history.json, deliberately: a history means an earlier
// metadata.info.json was here before this pass, and it is the one record of
// what the source used to say (lib/metadataHistory-server.ts).
@@ -926,6 +933,22 @@ async function runManagedDownload(
}
const metaPath = path.join(videoDir, "metadata.info.json");
+ // AN archive.org RECORD IS CORRECTED BEFORE ANYTHING READS IT: its
+ // provenance sidecar written (one cached metadata request per item), and
+ // metadata.info.json given the file's own page and title — yt-dlp writes
+ // the ITEM's for one file of a multi-file item (lib/archiveOrg-server.ts).
+ // Only after a prefetch that succeeded: a refused one is not a record.
+ if (
+ detectPlatform(opts.videoUrl) === "archiveorg" &&
+ attemptSucceeded(attempts.at(-1)?.ytdlpExitCode ?? null)
+ ) {
+ await (opts.archiveOrgProvenance ?? ensureArchiveOrgProvenance)({
+ videoDir,
+ videoUrl: opts.videoUrl,
+ onLog: opts.onLog,
+ signal: opts.signal,
+ });
+ }
const metadata = await loadRawMetadata(metaPath);
// Only wire --load-info-json into the real attempts when we actually have
// the metadata file; a failed prefetch falls through to the legacy path so
@@ -1240,6 +1263,16 @@ async function runManagedDownload(
runAudioCheck,
)
: await runAudioCheck();
+ // The audio-checked pass re-extracted, so the archive.org correction above
+ // is re-applied (from the sidecar; no request).
+ if (canonicalId && detectPlatform(opts.videoUrl) === "archiveorg") {
+ await (opts.archiveOrgProvenance ?? ensureArchiveOrgProvenance)({
+ videoDir: path.join(channelDir, "data", canonicalId),
+ videoUrl: opts.videoUrl,
+ onLog: opts.onLog,
+ signal: opts.signal,
+ });
+ }
primaryRes = {
exitCode: audioOutcome.ytdlpExitCode,
stderrTail: audioOutcome.stderrTail,
diff --git a/common/ytdlp/platformArgs.mjs b/common/ytdlp/platformArgs.mjs
@@ -34,10 +34,34 @@ import { detectPlatform } from "../lib/detectPlatform.mjs";
// (prefetch, subtitles, retries) back to back, and the two channels that 429'd
// were the two most-downloaded. Mirrors rumble's pace; a channel's own
// `ytdlpExtraArgs` still wins because it comes after.
+//
+// archiveorg: be polite to archive.org (a non-profit serving files from its
+// own disks). `--sleep-requests 2` spaces the extractor's requests (the embed
+// page, then the metadata API); `--retry-sleep` turns yt-dlp's immediate
+// retries of a refused request or a dropped transfer into an exponential wait
+// (2 s doubling to 120 s), and the stock retry counts bound how many. yt-dlp
+// downloads an archive.org file as ONE plain HTTP stream (no fragments, no
+// parallel ranges), and nothing here passes `-N` or an external downloader.
+// `--parse-metadata` makes the record's `webpage_url` the URL it was fetched
+// by: for one file of a multi-file item yt-dlp writes the ITEM's page there,
+// and the snapshot renames every video dir to `extractVideoId(webpage_url)`
+// (reconcileVideoDirs.ts) — the file's record would be merged into the item's.
+// The import always fetches by the canonical file page, so the two agree; the
+// regex only takes an archive.org URL, and leaves any other untouched.
/** @type {Readonly<Partial<Record<Platform, readonly string[]>>>} */
export const PLATFORM_ARGS = Object.freeze({
rumble: Object.freeze(["--impersonate", "chrome", "--sleep-requests", "1"]),
youtube: Object.freeze(["--sleep-requests", "1"]),
+ archiveorg: Object.freeze([
+ "--sleep-requests",
+ "2",
+ "--retry-sleep",
+ "http:exp=2:120",
+ "--retry-sleep",
+ "extractor:exp=2:120",
+ "--parse-metadata",
+ "original_url:(?P<webpage_url>https://archive\\.org/(?:details|embed|download)/.+)",
+ ]),
});
/**
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -3,6 +3,8 @@
## [Unreleased]
- **A forum thread can be archived as a posts source.** A XenForo thread URL (`…/threads/<title>.<id>/`; Kiwi Farms is recognised by host) makes a forum-thread channel — platform "xenforo", one channel per thread, each forum post a post — searchable and readable like X and Bluesky posts, in the editor, the export and the MCP (`get_thread` gives a forum post's conversation: the posts it quotes and the posts quoting it). **Fetch posts** reads the thread in a headless browser, newest page first, one page at a time with a 10–20 s pause (the channel key `postPagePauseSeconds` sets it), and stops at already-archived posts; a **Latest N pages** box (`archilyzer posts fetch --pages N`) caps a run, and the next run continues where it stopped. The browser keeps one profile per forum host, so a browser check it clears once (KiwiFlare's proof of work, say) stays cleared; a check that does not clear within a minute, a captcha, a login wall or a refusal stops the run with the reason and keeps its place — never retried at once. **Connect forum session** on the channel page opens that profile in a window on the editor's machine, at the thread, for the operator to clear it or log in. **Import saved pages** (`archilyzer posts import-html <slug> <file-or-dir>…`) reads thread pages saved from a browser ("Save page as", complete or HTML only) through the same parser: new posts are added and a post saved again after an edit is updated. **Capture posts** works on forum posts: a screenshot of the post and its attached files, through the same profile. A forum post keeps its thread title, page, position, author id, last-edit time, quoted posts and its media links; quoted text is marked with "> " lines.
- **A site's reports can be exported as files a reader saves and hosts again.** `archilyzer reports export <site> [--report <id>] [--formats html,pdf,md,zip]`, the `reports-export` job (`POST /api/ops/reports-export`, `pnpm ops reports-export`, and **Export reports** on a site's Reports tab) write each published report, checked as the build checks it, into `.export-index/sites/<site>/report-exports/<report>/`: `report.html`, one self-contained page (its own style, no script, stills and post screenshots inlined and recompressed, clips linked on the site); `report.pdf`, that page printed by headless Chromium, skipped with a note where there is none; `report.md`, plain Markdown with numbered references; and `evidence-pack.zip`, the page with its clips, stills and screenshots as files plus the Markdown and the citations, packed by the system `zip` (a host without it fails that format, naming it). An `export.json` names each file's size and checksum and the checksum of the report.json it was made from; every export ends with the report's date and the start of that checksum. Preparing the evidence media now exports at its end when nothing is missing, on the same queue. The build publishes an export beside the report only when it was made from the report as it is now and is at most 24 MiB — a larger evidence pack stays local — and the Reports tab lists each report's exports, their sizes and which the next build publishes. The 24 MiB limit the source mirror and the evidence clips already kept is now one shared number.
+- **archive.org items are a source (`platform: "archiveorg"`).** A channel can hold recordings imported from archive.org and transcribe them like any transcribe channel. **Import video** takes an item page (`https://archive.org/details/<identifier>`) when the item holds one media file, or ONE file of a multi-file item (`…/details/<identifier>/<file>`); an item with several media files is refused with the way to choose files. `pnpm ops import-archive-org --json '{"slug":…,"item":…,"files":[…]}'` (or `"match": "<regex>"`, `"dryRun": true`) imports chosen files of one item as one drainable job. A whole item's id is its identifier; a file's is `<identifier>__<slug>-<hash>`, stable and unique per file. Each record keeps an `archiveorg.json` sidecar — the item's title, date, creator and collections, its torrent, and for a mirror of a YouTube upload the original's id, URL, title and upload date read from the info.json uploaded beside it — and its metadata takes the file's own page and title (and a mirror's original title and date), recorded in the metadata history as `archiveorg-provenance`. The video page says "Archived on archive.org: <item> · torrent" and, for a mirror, "Originally on YouTube: <url> (uploaded <date>)". The channel form offers archive.org in both platform lists.
+- **Polite to archive.org.** archive.org runs on its own queue (`platform:archiveorg`), one download at a time, with a jittered pause of at least 8 s between files (the channel's or the global `sleepBetweenDownloadsSeconds` when longer). Its metadata API is asked once per item (cached for 6 h), with an identifying User-Agent, at most one request at a time and 2 s apart, honouring `Retry-After` and backing off exponentially on 429/503, stopping after four attempts. yt-dlp's archive.org spawns carry `--sleep-requests 2` and an exponential `--retry-sleep`; nothing downloads in parallel ranges. A file already downloaded is never fetched again, and a bulk import stops on a rate limit or after three failures in a row — re-running it resumes. The "auto" download format on archive.org takes the uploader's original mp4/mkv/webm, or an audio item's MP3.
- **A report video's cue lookup names a site that publishes only its reports.** Pointing a report-to-video manifest at such a site (`corpus.json` `site.scope: "cited"`) used to fail with "channel … is not in corpus.json"; it now says the site publishes no transcripts and to use a full archive or a local corpus. The site form's **Publish** hint says a cited-only site is never listed on the homepage or the hub.
- **A site's build composes its reports, and a site that publishes only its reports ships nothing else.** Every site's compose now writes the reports its `site.json` publishes: each report's page and its citations as `citations.json` and `citations.csv` under `/reports/<id>/`, its cited stills, a page per cited moment with the record, the transcript lines around the span and every report that cites it, and the clips and post captures `archilyzer reports prepare` made for it, only the cited ones. Each quote is checked against the record as it is composed (a span's against its cues within 5 s either side, read from `en-orig` when the `en` track has no cues; a post's against its text) and the score, time and method are written into the citation, replacing any typed by hand. The build stops with the list of every problem before anything is written: an invalid report, a citation of a channel outside the site or of a post the site may not carry, a missing record, still or post, a quote that matches less than 60 % of what the record says, and a citation without prepared media or with media cut for another span (`--allow-missing-media` on `archilyzer compose site` and `build site` lets those two through, without a clip). A site with `publish: "cited"` removes everything corpus-shaped from `export/public` before it writes its reports, and its built `out/` is checked against what a cited site may hold: anything else, a file over 25 MiB or more than 20,000 files fails the build, and every deploy path (the Publish tab, `deploy site`, Build & deploy, Build & deploy all, the container build) refuses it, as it refuses a site set to cited whose last build was a full one. The hub's compose removes a report site's files too.
- **A site has a Reports tab.** `/sites/<site>/reports` lists every report under the site's `reports/` directory — the published ones in their order, then the drafts — with its kind, dates, sections, claims, citations by kind and, for a fact-check, how many claims carry each verdict. Each report's problems, from the same checker the prepare step and the build use, open under it. A draft with no problems can be published, and a published report moved up or down or unpublished; each writes only the site's `reports` list, applied to the list as it is on disk at that moment, so it never overwrites another change to the site. "Prepare evidence media" queues the `reports-prepare` job, and beside it the tab shows the last prepared media (moments by kind, total size, problems by kind) and links the last prepare job. What the site publishes (full or cited) is shown with a link to Settings, where it is changed.
diff --git a/editor/app/api/ops/import-archive-org/route.ts b/editor/app/api/ops/import-archive-org/route.ts
@@ -0,0 +1,44 @@
+import { importArchiveOrgAction } from "../../../channels/[slug]/pipelineActions";
+import {
+ OpsInputError,
+ jobResponse,
+ ops,
+ optBool,
+ optString,
+ reqSlug,
+ reqString,
+ type OpsBody,
+} from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, item, files?: string[], match?: string, dryRun? }
+// -> { ok: true, jobId }
+//
+// Import chosen media files of ONE archive.org item into an existing channel,
+// as one job on archive.org's own queue: one file at a time, a jittered pause
+// between them, files already downloaded skipped, a rate limit or three
+// failures in a row ending it (controller/archiveOrgImport.ts). Exactly one of
+// `files` (exact paths in the item, untrimmed) and `match` (a case-insensitive
+// regex over them). `dryRun` logs what would be fetched and fetches nothing.
+function optFiles(body: OpsBody): string[] | undefined {
+ const v = body.files;
+ if (v === undefined) return undefined;
+ if (!Array.isArray(v) || v.length === 0 || v.some((s) => typeof s !== "string" || !s)) {
+ throw new OpsInputError('"files" must be a non-empty array of file names');
+ }
+ return v as string[];
+}
+
+export async function POST(request: Request) {
+ return ops(request, ["slug", "item", "files", "match", "dryRun"], async (body) =>
+ jobResponse(
+ await importArchiveOrgAction(reqSlug(body, "slug"), {
+ item: reqString(body, "item"),
+ files: optFiles(body),
+ match: optString(body, "match"),
+ dryRun: optBool(body, "dryRun"),
+ }),
+ ),
+ );
+}
diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts
@@ -12,7 +12,14 @@ import {
downloadQueueKey,
resolveQueueKey,
} from "yt-dlp-transcript-common/lib/queueKeys";
-import { detectPlatform } from "yt-dlp-transcript-common/lib/platform";
+import {
+ detectPlatform,
+ platformQueueKey,
+} from "yt-dlp-transcript-common/lib/platform";
+import {
+ resolveArchiveOrgImportUrl,
+ runArchiveOrgImport,
+} from "yt-dlp-transcript-common/controller/archiveOrgImport";
import {
heldPlatformRefusal,
platformCooldownRemainingMs,
@@ -24,7 +31,11 @@ import {
readChannelConfig,
readChannelStat,
} from "yt-dlp-transcript-common/controller/channels";
-import { extractVideoId, runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp";
+import {
+ destinationExists,
+ extractVideoId,
+ runYtdlp,
+} from "yt-dlp-transcript-common/ytdlp/runYtdlp";
import { mergeRosterFile } from "yt-dlp-transcript-common/controller/rosterStore";
import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged";
import { runMetadataScanJob } from "yt-dlp-transcript-common/controller/metadataScanJob";
@@ -496,10 +507,41 @@ export async function importVideoAction(
if (!channelConfig) {
return { ok: false, error: `Channel "${slug}" not found` };
}
- const videoUrl = url.trim();
+ let videoUrl = url.trim();
if (!/^https?:\/\//i.test(videoUrl)) {
return { ok: false, error: "Enter a valid video URL" };
}
+ // AN archive.org URL is resolved first (one cached request for the item's
+ // metadata): an item holding several media files is refused — import a file
+ // of it, or several with import-archive-org — and the URL becomes the
+ // canonical page of what is imported (controller/archiveOrgImport.ts). It
+ // runs on archive.org's own queue whatever the channel's platform, with
+ // archive.org's yt-dlp args, and a file already on disk is not fetched again.
+ const archiveOrg = detectPlatform(videoUrl) === "archiveorg";
+ let downloadConfig = channelConfig;
+ if (archiveOrg) {
+ const refused = await archiveOrgRefusal(paths, "The import");
+ if (refused) return { ok: false, info: true, error: refused };
+ const resolved = await resolveArchiveOrgImportUrl(videoUrl);
+ if (!resolved.ok) return { ok: false, error: resolved.error };
+ videoUrl = resolved.url;
+ if (
+ await destinationExists(
+ path.join(paths.channelsDir, slug, "data"),
+ resolved.id,
+ channelConfig.handling,
+ )
+ ) {
+ return {
+ ok: false,
+ info: true,
+ error: `Already downloaded: data/${resolved.id}/ — archive.org is not asked for it again.`,
+ };
+ }
+ if (channelConfig.platform !== "archiveorg") {
+ downloadConfig = { ...channelConfig, platform: "archiveorg" };
+ }
+ }
// Best-effort canonical id: used only for revalidation/labels. When null,
// downloadOneManaged falls back to %(id)s and the reconcile pass repairs the
// dir, so we don't hard-fail here.
@@ -509,7 +551,10 @@ export async function importVideoAction(
const settings = getSettings();
return runManagedFunction({
kind: "import-one",
- queueKey: resolveQueueKey(downloadQueueKey(channelConfig), queueKey),
+ queueKey: resolveQueueKey(
+ archiveOrg ? platformQueueKey("archiveorg") : downloadQueueKey(channelConfig),
+ queueKey,
+ ),
paths,
channelSlug: slug,
videoId,
@@ -522,7 +567,7 @@ export async function importVideoAction(
try {
await downloadOneManaged({
channelSlug: slug,
- channelConfig,
+ channelConfig: downloadConfig,
paths,
videoUrl,
onLog: task.onLog,
@@ -557,6 +602,93 @@ export async function importVideoAction(
});
}
+// archive.org held, or in a rate-limit cooldown: a sentence, else null. An
+// import asks archive.org nothing while it has asked us to wait.
+async function archiveOrgRefusal(paths: Paths, what: string): Promise<string | null> {
+ const held = await heldPlatformRefusal("archiveorg", what, paths);
+ if (held) return held;
+ const remainingMs = await platformCooldownRemainingMs("archiveorg", paths);
+ if (remainingMs > 0) {
+ return `archive.org is in a rate-limit cooldown (${Math.ceil(remainingMs / 1000)}s remaining). ${what} can run once it lapses.`;
+ }
+ return null;
+}
+
+// IMPORT CHOSEN FILES OF ONE archive.org ITEM (`pnpm ops import-archive-org`):
+// one job on archive.org's own queue that imports the files one at a time,
+// with a jittered pause between them, skipping any already downloaded, and
+// stopping on a rate limit or three failures in a row
+// (controller/archiveOrgImport.ts). `files` names exact paths in the item;
+// `match` is a case-insensitive regex over them. `dryRun` lists what would be
+// fetched and fetches nothing.
+export async function importArchiveOrgAction(
+ slug: string,
+ opts: { item: string; files?: string[]; match?: string; dryRun?: boolean },
+): Promise<StreamActionResult> {
+ const paths = getPaths();
+ const channelConfig = await readChannelConfig(paths, slug);
+ if (!channelConfig) {
+ return { ok: false, error: `Channel "${slug}" not found` };
+ }
+ const item = opts.item.trim();
+ if (!/^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(item)) {
+ return { ok: false, error: `"${item}" is not an archive.org identifier` };
+ }
+ if ((opts.files === undefined) === (opts.match === undefined)) {
+ return { ok: false, error: 'Name the files: exactly one of "files" (a list) or "match" (a regex)' };
+ }
+ if (opts.match !== undefined) {
+ try {
+ new RegExp(opts.match, "i");
+ } catch (e) {
+ return { ok: false, error: `"match" is not a valid regex: ${(e as Error).message}` };
+ }
+ }
+ if (!opts.dryRun) {
+ const err = await lowDiskError(paths, slug);
+ if (err) return err;
+ const refused = await archiveOrgRefusal(paths, "The archive.org import");
+ if (refused) return { ok: false, info: true, error: refused };
+ }
+ const downloadConfig =
+ channelConfig.platform === "archiveorg"
+ ? channelConfig
+ : { ...channelConfig, platform: "archiveorg" as const };
+ return runManagedFunction({
+ kind: "import-archive-org",
+ queueKey: platformQueueKey("archiveorg"),
+ paths,
+ channelSlug: slug,
+ fn: async (onLog, signal, _setProgress, ctx) => {
+ const result = await runArchiveOrgImport({
+ paths,
+ slug,
+ channelConfig: downloadConfig,
+ identifier: item,
+ selection: opts.files !== undefined ? { files: opts.files } : { match: opts.match! },
+ onLog,
+ signal,
+ drainSignal: ctx.drainSignal,
+ dryRun: opts.dryRun,
+ deps: {
+ onImported: (id) => safeRevalidate([`/channels/${slug}/videos/${id}`]),
+ },
+ });
+ safeRevalidate([`/channels/${slug}`, "/channels"]);
+ // The platform's shared pacing state learns what archive.org said, so
+ // the next import (and any other archive.org job) backs off or settles.
+ if (result.rateLimited) {
+ await recordDownloadBackoff("archiveorg", paths, "rate_limit");
+ } else if (result.imported.length > 0 && result.failed.length === 0) {
+ await recordPlatformClean("archiveorg", paths);
+ }
+ if (result.failed.length > 0 && result.imported.length === 0 && !opts.dryRun) {
+ throw new Error(result.stopped ?? `${result.failed.length} file(s) failed`);
+ }
+ },
+ });
+}
+
// A PODCAST CHANNEL'S RECORDS COMPLETED FROM ITS RSS FEED (the
// `feed-metadata` job, controller/feedMetadataBackfill.ts): one fetch of the
// channel's url, then title, date, description and duration written into each
diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx
@@ -11,6 +11,8 @@ import { listClipWindows } from "yt-dlp-transcript-common/lib/clipWindow-server"
import { isDoNotClean } from "yt-dlp-transcript-common/lib/doNotClean-server";
import { isExcludedFromTruncatedCheck } from "yt-dlp-transcript-common/lib/excludeTruncatedCheck-server";
import { loadSavedVideo } from "yt-dlp-transcript-common/lib/savedVideo-server";
+import { loadArchiveOrgProvenance } from "yt-dlp-transcript-common/lib/archiveOrg-server";
+import type { ArchiveOrgProvenance } from "yt-dlp-transcript-common/lib/archiveOrg";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import {
readVideoMetadataForDisplay,
@@ -134,6 +136,9 @@ export default async function VideoDetailPage({
const excludedFromTruncatedCheck =
await isExcludedFromTruncatedCheck(videoDir);
const savedVideo = await loadSavedVideo(videoDir);
+ // Where an archive.org record came from (lib/archiveOrg-server.ts): the
+ // item and its torrent, and a mirror's original. Absent everywhere else.
+ const archiveOrg = await loadArchiveOrgProvenance(videoDir);
// The windows another tool asked this editor to fetch. One readdir of
// data/<id>/clips/ plus a stat per file — and no per-CHANNEL count anywhere,
// because that would be a walk of every video dir to draw one number.
@@ -185,6 +190,7 @@ export default async function VideoDetailPage({
doNotClean,
excludedFromTruncatedCheck,
savedVideo,
+ archiveOrg,
clipWindows,
vttProvenance,
coverage,
@@ -212,6 +218,7 @@ export default async function VideoDetailPage({
doNotClean,
excludedFromTruncatedCheck,
savedVideo,
+ archiveOrg,
clipWindows,
vttProvenance,
coverage,
@@ -277,6 +284,7 @@ export default async function VideoDetailPage({
</span>
)}
</div>
+ {archiveOrg && <ArchiveOrgProvenanceLine prov={archiveOrg} />}
{meta.description && (
<details className="text-sm">
<summary className="cursor-pointer text-muted-foreground hover:text-foreground">
@@ -343,6 +351,40 @@ export default async function VideoDetailPage({
);
}
+// "Archived on archive.org: <item> · torrent", and for a mirror "Originally on
+// YouTube: <url> (uploaded <date>)". Terse, under the header line.
+function ArchiveOrgProvenanceLine({ prov }: { prov: ArchiveOrgProvenance }) {
+ const link = "underline hover:text-foreground";
+ return (
+ <div aria-label="archive.org provenance" className="flex flex-col gap-1 text-sm text-muted-foreground">
+ <span>
+ Archived on archive.org:{" "}
+ <a href={prov.fileUrl ?? prov.itemUrl} target="_blank" rel="noreferrer" className={link}>
+ {prov.item.title ?? prov.identifier}
+ {prov.file ? ` / ${prov.file}` : ""}
+ </a>
+ {prov.torrentUrl && (
+ <>
+ {" · "}
+ <a href={prov.torrentUrl} target="_blank" rel="noreferrer" className={link}>
+ torrent
+ </a>
+ </>
+ )}
+ </span>
+ {prov.mirror && (
+ <span>
+ Originally on YouTube:{" "}
+ <a href={prov.mirror.url} target="_blank" rel="noreferrer" className={link}>
+ {prov.mirror.url}
+ </a>
+ {prov.mirror.uploadDate && ` (uploaded ${formatUploadDate(prov.mirror.uploadDate)})`}
+ </span>
+ )}
+ </div>
+ );
+}
+
function formatUploadDate(s: string): string {
// yt-dlp emits YYYYMMDD. Render as YYYY-MM-DD; pass through anything else.
if (/^\d{8}$/.test(s)) {
diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx
@@ -384,6 +384,7 @@ export function ChannelForm({
<option value="odysee">Odysee</option>
<option value="twitch">Twitch</option>
<option value="kick">Kick</option>
+ <option value="archiveorg">archive.org</option>
<option value="twitter">X / Twitter (posts)</option>
<option value="bluesky">Bluesky (posts)</option>
<option value="xenforo">Forum thread — XenForo (posts)</option>
@@ -504,6 +505,7 @@ export function ChannelForm({
<option value="odysee">Odysee</option>
<option value="twitch">Twitch</option>
<option value="kick">Kick</option>
+ <option value="archiveorg">archive.org</option>
<option value="twitter">X / Twitter (posts)</option>
<option value="bluesky">Bluesky (posts)</option>
<option value="xenforo">Forum thread — XenForo (posts)</option>
diff --git a/editor/app/storage/lib/storeBusy.ts b/editor/app/storage/lib/storeBusy.ts
@@ -40,6 +40,7 @@ const STORE_TOUCHING_KINDS = new Set([
"download-missing",
"download-missing-subs",
"import-one",
+ "import-archive-org",
"redownload-archive",
"redownload-incomplete-bucket",
"retry-bucket",
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -4,6 +4,7 @@
- **A report can be saved whole: as one HTML page, a PDF, Markdown, or an evidence pack.** A report page's download line now reads HTML · PDF · Markdown · Evidence pack · Citations JSON · CSV, each listed only when the site publishes it. The HTML is one file that opens with no network: the report with its verdicts, the document's sentences and the post screenshots inside it, numbered citations, and a reference list giving each quote's speaker, date, record, the original at its time and the moment page on the site. The PDF is that page printed. The Markdown is the same report as plain text with numbered references. The evidence pack is a zip of the page with its clips, stills and screenshots beside it, so the clips play offline. Each ends with a line naming the report's date and the start of its checksum. Needs `reports export` (or prepare) and a rebuild and deploy of each site with reports.
- **A report's claim can carry a flag, its title a byline, and a site with one report names it in the browser tab.** `report.json` claim `flag` (one line, at most 60 characters) shows as a small pill in the accent colour beside the claim's verdict, e.g. "No source given". On a report-only site with one report, the home page's tab title is the report's, as on the report's own page, where it was the site's title alone. A report's page follows its title with a byline from the document under review, "by <author> · <publisher>", in the same line when it fits; the publisher (or, with none, the author) links to the document. Under it, a step apart, the page names its own: "Fact-check by <site title>" ("Report by …"), then the dates, then the subtitle; the kind's label no longer sits above the title on the report's page. A claim's sentence from the document under review no longer links up to the document's box: it sits on a rail in the document's colour, and the box's left edge wears the same colour (a source's `accent`, `"#rrggbb"`; without one, the border colour). A sentence of another document keeps its "from <title>" link. A report's citation can say where its evidence came from (`origin`: `"subject"`, the document under review gave it; `"added"`, the report's author found it). A claim lists what the report added first, each card marked with the Archilyzer mark and "Not in the article" ("Not in the source"), then evidence of unknown origin, then what the document gave itself folded under "In the article (n)"; the reference list marks an added citation with the mark alone, and a claim's flag pill wears the same mark. A fact-check's page reads in three tiers, each opened by a hairline with one, two or three dots and its reading time (at 230 words a minute): the quick take (the tally, the summary, and links to what the check found, every claim and the downloads); **What the check found**, every ruled claim grouped by verdict (contradicted, not found, partly, untestable, corroborated), one line each linking to the claim, with its `gist` (a new optional claim field, one line, at most 240 characters) and its flag; and **Every claim, with its evidence**, which opens with **How it was checked** (`method`, a new optional report field in markdown). A report of kind `sweep` has the first and last tiers only. Needs a rebuild and deploy of the site.
- **Forum posts read like the other posts.** A post from a forum-thread channel shows its place in the thread (#N), an "edited" mark, the thread's title and its media as links, and opening its thread shows its conversation — the posts it quotes and the posts quoting it — rather than the whole forum thread.
+- **archive.org records play and are cited with their downloads.** A record imported from archive.org plays its file in the page's own player, which seeks to a cited second and follows the transcript. A citation of one links "archive.org" (the file's page) and its "torrent"; a citation of an archive.org mirror of a YouTube upload links the original on YouTube at the cited second, then "archive.org" and "torrent", on the citation cards and the moment pages.
- **A report-only site with one report opens on that report.** Its home page is the report itself, its header links nothing, and `/reports/` forwards home: there is no index of one. With more reports the home page is the list, without repeating the site's title under the header; a list entry is the report's name, subtitle and dates (its counts and tally are on its page). Pages a report-only site does not have link home.
- **A report can belong to a series.** `report.json` `series` is shown on its own line above the report's title, in the accent colour, in place of the kind's label ("Fact-check"); a page title, a cited-in link, `llms.txt` and the MCP name it `<series>: <title>`.
- **`pnpm start:export` serves a built site's moment pages.** It used `serve`, which listed a video or audio moment's directory (`3126.00-3151.00`) instead of serving its page; it now runs `export/scripts/serve-out.mjs`, which serves directories as Cloudflare Pages does, on `EXPORT_DEV_PORT` (3000).
diff --git a/export/app/components/reports/MomentArticle.tsx b/export/app/components/reports/MomentArticle.tsx
@@ -124,9 +124,16 @@ export default function MomentArticle({ view }: { view: MomentPageView }) {
<p className="flex flex-wrap gap-x-4 gap-y-1 text-sm">
{r.originalUrl && (
<ExternalLinkText href={r.originalUrl}>
- {isSpan && view.start !== undefined ? `Original at ${formatTimestamp(view.start)}` : "Original post"}
+ {isSpan && view.start !== undefined
+ ? `${r.originalLabel ?? "Original"} at ${formatTimestamp(view.start)}`
+ : "Original post"}
</ExternalLinkText>
)}
+ {r.downloads?.map((d) => (
+ <ExternalLinkText key={d.url} href={d.url}>
+ {d.label}
+ </ExternalLinkText>
+ ))}
{r.corpusUrl && (
<a href={r.corpusUrl} className={textLink}>
Open in the archive
diff --git a/export/app/duplicates/DuplicatesClient.tsx b/export/app/duplicates/DuplicatesClient.tsx
@@ -31,6 +31,7 @@ const PLATFORM_LABEL: Record<string, string> = {
odysee: "Odysee",
twitch: "Twitch",
kick: "Kick",
+ archiveorg: "archive.org",
};
const MATCH_LABEL: Record<DuplicateCluster["matchKind"], string> = {
diff --git a/mcp/README.md b/mcp/README.md
@@ -53,7 +53,7 @@ compact `[mm:ss|<seconds>]`. The expansion rule: **full moment link =
`[title @ 2:36](<moment_base>156)`. The seconds are floored exactly like the
inline links', so both styles cite the identical second. A video whose base
can't be built (no viewer origin and a platform whose time param doesn't take
-raw seconds — Twitch — or doesn't exist — Rumble/Kick) omits the line: cite its
+raw seconds — Twitch — or doesn't exist — Rumble/Kick/archive.org) omits the line: cite its
`- source:` URL plain instead. `open_link` results and `get_transcript` stay
inline-linked (future work).
diff --git a/mcp/src/reports.ts b/mcp/src/reports.ts
@@ -153,7 +153,10 @@ function citationLines(c: CitationView, origin: string | null): string[] {
.join(" · ");
lines.push(`${n} ${c.kind} ${where} @ ${Math.floor(c.start)}–${Math.floor(c.end)} s`);
lines.push(` ${quote}` + (c.speaker ? ` — ${c.speaker}` : ""));
- if (c.record.originalUrl) lines.push(` original: ${c.record.originalUrl}`);
+ if (c.record.originalUrl) {
+ lines.push(` original${c.record.originalLabel ? ` (${c.record.originalLabel})` : ""}: ${c.record.originalUrl}`);
+ }
+ for (const d of c.record.downloads ?? []) lines.push(` ${d.label}: ${d.url}`);
lines.push(` moment: ${absolute(origin, c.href)}`);
break;
}
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -27,6 +27,8 @@
// pnpm ops sync --json '{"slug":"the-quartering"}' --wait
// pnpm ops metadata-scan --json '{"slug":"the-quartering"}'
// pnpm ops feed-metadata --json '{"slug":"demo-podcast","dryRun":true}' --wait
+// pnpm ops import-video --json '{"slug":"demo-archive","url":"https://archive.org/details/example-item"}'
+// pnpm ops import-archive-org --json '{"slug":"demo-archive","item":"example-item","match":"\\.mp4$"}' --wait
// pnpm ops channel-config --json '{"slug":"x","patch":{"downloadFilterExclude":"rerun"}}'
// pnpm ops channel-priority --json '{"slugs":["x"],"operation":"download","tier":"paused"}'
// pnpm ops lane --json '{"lane":"download","held":true}'
@@ -97,6 +99,9 @@ const ACTIONS = [
"channel-config",
"metadata-scan",
"import-video",
+ // Chosen media files of ONE archive.org item ({slug, item, files: [...] |
+ // match: "<regex>", dryRun?}): one job, one file at a time, paced.
+ "import-archive-org",
// A podcast channel's records completed from its RSS feed ({slug, dryRun?}):
// one fetch of the feed, no media.
"feed-metadata",