commit 5a14b2addb79a5c6ea1535e1b00cc3deccff6172
parent 9d8e7303ace208fdf95a13d21810268235868d0e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 6 Oct 2026 09:01:22 -0400
wayback refresh: bring imported Wayback copies up to the rules, offline
`archilyzer wayback refresh <slug> [--titles <file>] [--dry-run]`
(controller/waybackRefresh.ts): for every record whose webpage_url, original_url or
download log names a capture, writes `wayback.json`, renames the dir to its canonical id
through reconcileVideoDirs (new `only` filter; tier links move with the dir, an existing
dir is merged) and moves its roster entry (renameRosterEntries), and with --titles sets a
raw file's title and upload_date through patchMetadataInfo as `wayback-provenance`. A
record a live job names (queued/running meta, writer alive) is held; a live channel-wide
job holds every record. Idempotent; prints old → new.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
7 files changed, 657 insertions(+), 1 deletion(-)
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -351,6 +351,25 @@ export const COMMANDS: Command[] = [
},
},
{
+ path: ["wayback", "refresh"],
+ usage:
+ "<slug> [--titles <json>] [--dry-run] bring a channel's Wayback Machine copies up to the Wayback rules: wayback.json (original URL, capture time), the dir renamed to its canonical id through the snapshot's reconcile (roster moved with it), and with --titles (a file of id → {title, upload_date}) a raw file's title and date; offline, skips a record a live job holds; prints old → new",
+ flags: { titles: "string", "dry-run": "boolean" },
+ maxPositionals: 1,
+ run: async ({ positionals, flags }) => {
+ const [slug] = positionals;
+ if (!slug) {
+ console.error("wayback refresh: which channel? Pass its slug.");
+ return 2;
+ }
+ return (await import("./wayback-refresh")).main({
+ slug,
+ dryRun: flags["dry-run"] === true,
+ ...(typeof flags.titles === "string" ? { titlesFile: flags.titles } : {}),
+ });
+ },
+ },
+ {
path: ["feeds", "backfill-metadata"],
usage:
"<slug> [--feed <url>] [--dry-run] complete a podcast channel's records (title, date, description, duration) from its RSS feed: one fetch of the feed (default: the channel's url), no media; --dry-run counts matched / unmatched / already complete and writes nothing",
diff --git a/common/bin/wayback-refresh.ts b/common/bin/wayback-refresh.ts
@@ -0,0 +1,70 @@
+// `archilyzer wayback refresh <slug> [--titles <json>] [--dry-run]` — every
+// Wayback Machine copy of a channel brought up to the Wayback rules
+// (controller/waybackRefresh.ts): its `wayback.json`, its dir under its
+// canonical id (through the snapshot's own reconcile pass, roster moved with
+// it), and with `--titles` a file mapping id → {title, upload_date} for a raw
+// file that has neither. Offline. Prints old → new per record; a second run
+// changes nothing.
+
+import { readFile } from "node:fs/promises";
+import { isValidChannelSlug } from "../controller/channels";
+import { refreshWaybackRecords, type WaybackTitle } from "../controller/waybackRefresh";
+import { getPaths, type Paths } from "../lib/paths";
+
+async function readTitles(file: string): Promise<Record<string, WaybackTitle>> {
+ const v = JSON.parse(await readFile(file, "utf8")) as unknown;
+ if (!v || typeof v !== "object" || Array.isArray(v)) {
+ throw new Error(`${file}: expected an object of id → {title, upload_date}`);
+ }
+ const out: Record<string, WaybackTitle> = {};
+ for (const [id, raw] of Object.entries(v as Record<string, unknown>)) {
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) throw new Error(`${file}: "${id}" is not an object`);
+ const r = raw as Record<string, unknown>;
+ out[id] = {
+ ...(typeof r.title === "string" ? { title: r.title } : {}),
+ ...(typeof r.upload_date === "string" ? { upload_date: r.upload_date } : {}),
+ };
+ }
+ return out;
+}
+
+export async function main(opts: {
+ slug: string;
+ dryRun: boolean;
+ titlesFile?: string;
+ paths?: Paths;
+}): Promise<number> {
+ if (!isValidChannelSlug(opts.slug)) {
+ console.error(`wayback refresh: "${opts.slug}" is not a channel slug`);
+ return 2;
+ }
+ try {
+ const titles = opts.titlesFile ? await readTitles(opts.titlesFile) : undefined;
+ const r = await refreshWaybackRecords({
+ slug: opts.slug,
+ paths: opts.paths ?? getPaths(),
+ dryRun: opts.dryRun,
+ titles,
+ onLog: (line) => console.log(line.replace(/\n$/, "")),
+ });
+ const would = opts.dryRun ? "would be " : "";
+ const renamed = r.records.filter((x) => x.rename === "renamed" || x.rename === "merged").length;
+ const held = r.records.filter((x) => x.rename === "held" || x.rename === "conflict").length;
+ const sidecars = r.records.filter((x) => x.sidecar).length;
+ const patched = r.records.filter((x) => Object.keys(x.changes).length > 0).length;
+ for (const id of r.unmatchedTitles) console.log(`--titles: no Wayback record "${id}"`);
+ for (const f of r.failed) console.error(`${f.id}: failed — ${f.error}`);
+ console.log(
+ `${opts.dryRun ? "dry run: " : ""}${r.records.length} Wayback records, ` +
+ `${renamed} ${would}renamed, ${sidecars} wayback.json ${would}written, ` +
+ `${patched} ${would}retitled` +
+ (held > 0 ? `, ${held} not renamed` : "") +
+ (r.failed.length > 0 ? `, ${r.failed.length} failed` : "") +
+ ".",
+ );
+ return r.failed.length > 0 ? 1 : 0;
+ } catch (err) {
+ console.error(`wayback refresh: ${(err as Error).message}`);
+ return 1;
+ }
+}
diff --git a/common/controller/reconcileVideoDirs.ts b/common/controller/reconcileVideoDirs.ts
@@ -28,6 +28,10 @@ export type ReconcileOpts = {
channelDir: string;
dryRun?: boolean;
onLog?: (s: string) => void;
+ // Reconcile only the dirs this accepts (default: every dir). A caller that
+ // knows which records it means to move (`archilyzer wayback refresh`,
+ // controller/waybackRefresh.ts) leaves the rest to the snapshot's pass.
+ only?: (dirName: string) => boolean;
};
// Files the canonical dir's copy should win on a name collision: these are
@@ -136,7 +140,10 @@ export async function reconcileVideoDirs(
const entries = await readdir(dataDir, { withFileTypes: true }).catch(
() => [],
);
- const dirNames = entries.filter((e) => e.isDirectory()).map((e) => e.name);
+ const dirNames = entries
+ .filter((e) => e.isDirectory())
+ .map((e) => e.name)
+ .filter((name) => !opts.only || opts.only(name));
for (const name of dirNames) {
try {
diff --git a/common/controller/rosterStore.ts b/common/controller/rosterStore.ts
@@ -141,6 +141,33 @@ export function mergeRoster(
return { ...roster, version: ROSTER_VERSION, updatedAt: now, entries };
}
+// A RENAMED RECORD keeps its roster entry under its new id. Not a removal:
+// the entry moves (url, firstSeenAt, source and all), which is what a dir
+// renamed to its canonical id (reconcileVideoDirs.ts) needs — left under the
+// old id it would read as a video the channel has and nobody downloaded.
+// When both ids have an entry the new one's stays and the earlier
+// firstSeenAt wins. Returns the SAME object when nothing moved.
+export function renameRosterEntries(
+ roster: Roster,
+ renames: ReadonlyArray<{ from: string; to: string }>,
+ now: string,
+): Roster {
+ const entries: Record<string, RosterEntry> = { ...roster.entries };
+ let changed = false;
+ for (const { from, to } of renames) {
+ const prev = entries[from];
+ if (!prev || from === to) continue;
+ const there = entries[to];
+ entries[to] = there
+ ? { ...there, firstSeenAt: prev.firstSeenAt && prev.firstSeenAt < there.firstSeenAt ? prev.firstSeenAt : there.firstSeenAt, url: there.url || prev.url }
+ : prev;
+ delete entries[from];
+ changed = true;
+ }
+ if (!changed) return roster;
+ return { ...roster, version: ROSTER_VERSION, updatedAt: now, entries };
+}
+
// Stamp the outcome of an enumeration. Kept separate from mergeRoster because a
// REJECTED enumeration still merges (additively, losing nothing) while recording
// that its listing was not trusted — that record is what the next enumeration
diff --git a/common/controller/waybackRefresh.test.ts b/common/controller/waybackRefresh.test.ts
@@ -0,0 +1,222 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { lstat, mkdir, mkdtemp, readFile, readdir, readlink, stat, symlink, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import { buildWaybackProvenance } from "../lib/wayback";
+import { loadMetadataHistory } from "../lib/metadataHistory-server";
+import { refreshWaybackRecords } from "./waybackRefresh";
+import { renameRosterEntries, type Roster } from "./rosterStore";
+import { main as refreshCli } from "../bin/wayback-refresh";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/waybackRefresh.test.ts
+//
+// A temp corpus of Wayback copies imported before the app knew what one was:
+// an archived YouTube page in `watch/`, two raw JW Player files in
+// `<jwId>-<rendition>.mp4/`, and a plain record. Every id here is invented.
+
+const SLUG = "demo-wayback";
+const YT = "Abc123def45";
+const PAGE = `https://web.archive.org/web/20220102030405/https://www.youtube.com/watch?v=${YT}`;
+const JW_A = "Qw3rTy12";
+const JW_B = "Zx9vBn34";
+const fileUrl = (jw: string, ts: string) =>
+ `https://web.archive.org/web/${ts}id_/https://videos-fms.jwpsrv.com/content/conversions/AcCt1234/videos/${jw}-12345678.mp4?token=0_abc_0xdef`;
+const FILE_A = fileUrl(JW_A, "20200102030405");
+const FILE_B = fileUrl(JW_B, "20200103030405");
+
+async function withCorpus(
+ fn: (paths: Paths, dataDir: string, channelDir: string) => Promise<void>,
+): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "ttb-wayback-"));
+ const transcriptsDir = path.join(dir, "corpus");
+ const paths = {
+ transcriptsDir,
+ channelsDir: path.join(transcriptsDir, "channels"),
+ jobsDir: path.join(transcriptsDir, ".jobs"),
+ } as Paths;
+ const channelDir = path.join(paths.channelsDir, SLUG);
+ const dataDir = path.join(channelDir, "data");
+ await mkdir(dataDir, { recursive: true });
+ await mkdir(paths.jobsDir, { recursive: true });
+ await writeFile(
+ path.join(channelDir, "config.json"),
+ JSON.stringify({ name: "Demo", platform: "archiveorg", handling: "transcribe" }),
+ );
+ const record = async (name: string, info: Record<string, unknown>) => {
+ await mkdir(path.join(dataDir, name), { recursive: true });
+ await writeFile(path.join(dataDir, name, "metadata.info.json"), JSON.stringify(info));
+ };
+ await record("watch", {
+ id: YT,
+ title: "An archived upload",
+ upload_date: "20090102",
+ extractor_key: "YoutubeWebArchive",
+ webpage_url: PAGE,
+ webpage_url_basename: "watch",
+ });
+ for (const [jw, url] of [[JW_A, FILE_A], [JW_B, FILE_B]] as const) {
+ await record(`${jw}-12345678.mp4`, {
+ id: `${jw}-12345678`,
+ title: `${jw}-12345678`,
+ extractor_key: "Generic",
+ webpage_url: url,
+ webpage_url_basename: `${jw}-12345678.mp4`,
+ });
+ }
+ await record("Plain12345a", { id: "Plain12345a", title: "Plain", upload_date: "20200101", extractor_key: "Youtube", webpage_url: "https://www.youtube.com/watch?v=Plain12345a" });
+ // A tiered file: a relative link into media/<old id>/ (release 17).
+ await mkdir(path.join(channelDir, "media", `${JW_A}-12345678.mp4`), { recursive: true });
+ await writeFile(path.join(channelDir, "media", `${JW_A}-12345678.mp4`, "audio.mp3"), "audio");
+ await symlink(
+ path.join("..", "..", "media", `${JW_A}-12345678.mp4`, "audio.mp3"),
+ path.join(dataDir, `${JW_A}-12345678.mp4`, "audio.mp3"),
+ );
+ const entry = (url: string) => ({ url, firstSeenAt: "2026-01-01T00:00:00.000Z", lastListedAt: "2026-01-01T00:00:00.000Z", source: "import" });
+ await writeFile(
+ path.join(channelDir, "roster.json"),
+ JSON.stringify({
+ version: 1,
+ updatedAt: "2026-01-01T00:00:00.000Z",
+ lastSweep: null,
+ entries: {
+ watch: entry(PAGE),
+ [`${JW_A}-12345678.mp4`]: entry(FILE_A),
+ [`${JW_B}-12345678.mp4`]: entry(FILE_B),
+ Plain12345a: entry("https://www.youtube.com/watch?v=Plain12345a"),
+ },
+ }),
+ );
+ await fn(paths, dataDir, channelDir);
+}
+
+const titles = {
+ [JW_A]: { title: "Show: First Guest", upload_date: "2019-03-03" },
+ [`${JW_B}-12345678.mp4`]: { title: "Show: Second Guest", upload_date: "20190310" },
+ [YT]: { title: "Not applied: the page has a title", upload_date: "20000101" },
+ nobody: { title: "No such record" },
+};
+
+test("a dry run reports every rename and title and writes nothing", async () => {
+ await withCorpus(async (paths, dataDir, channelDir) => {
+ const rosterBefore = await readFile(path.join(channelDir, "roster.json"), "utf8");
+ const r = await refreshWaybackRecords({ slug: SLUG, paths, dryRun: true, titles });
+ assert.deepEqual(
+ r.records.map((x) => [x.from, x.to, x.rename, x.sidecar]),
+ [
+ [`${JW_A}-12345678.mp4`, JW_A, "renamed", true],
+ [`${JW_B}-12345678.mp4`, JW_B, "renamed", true],
+ ["watch", YT, "renamed", true],
+ ],
+ );
+ assert.deepEqual(r.records[0].changes, {
+ title: { from: `${JW_A}-12345678`, to: "Show: First Guest" },
+ upload_date: { from: null, to: "20190303" },
+ });
+ assert.deepEqual(r.records[2].changes, {});
+ assert.deepEqual(r.unmatchedTitles, ["nobody"]);
+ assert.deepEqual((await readdir(dataDir)).sort(), [`${JW_A}-12345678.mp4`, `${JW_B}-12345678.mp4`, "Plain12345a", "watch"].sort());
+ assert.equal(await readFile(path.join(channelDir, "roster.json"), "utf8"), rosterBefore);
+ await assert.rejects(stat(path.join(dataDir, "watch", "wayback.json")));
+ });
+});
+
+test("a real run renames through the reconcile pass, moves the roster, writes sidecars and titles; a second run changes nothing", async () => {
+ await withCorpus(async (paths, dataDir, channelDir) => {
+ const r = await refreshWaybackRecords({ slug: SLUG, paths, titles, now: () => new Date("2026-02-02T00:00:00.000Z") });
+ assert.equal(r.failed.length, 0);
+ assert.deepEqual((await readdir(dataDir)).sort(), [JW_A, JW_B, "Plain12345a", YT].sort());
+
+ // The tier link moved with its dir and still resolves.
+ const link = path.join(dataDir, JW_A, "audio.mp3");
+ assert.ok((await lstat(link)).isSymbolicLink());
+ assert.equal(await readlink(link), path.join("..", "..", "media", `${JW_A}-12345678.mp4`, "audio.mp3"));
+ assert.equal(await readFile(link, "utf8"), "audio");
+
+ const roster = JSON.parse(await readFile(path.join(channelDir, "roster.json"), "utf8"));
+ assert.deepEqual(Object.keys(roster.entries).sort(), [JW_A, JW_B, "Plain12345a", YT].sort());
+ assert.equal(roster.entries[YT].url, PAGE);
+ assert.equal(roster.entries[YT].firstSeenAt, "2026-01-01T00:00:00.000Z");
+
+ assert.deepEqual(JSON.parse(await readFile(path.join(dataDir, YT, "wayback.json"), "utf8")), buildWaybackProvenance(PAGE));
+ assert.equal(JSON.parse(await readFile(path.join(dataDir, JW_A, "wayback.json"), "utf8")).originalUrl, FILE_A.replace(/^.*?id_\//, ""));
+ await assert.rejects(stat(path.join(dataDir, "Plain12345a", "wayback.json")));
+
+ const info = JSON.parse(await readFile(path.join(dataDir, JW_B, "metadata.info.json"), "utf8"));
+ assert.equal(info.title, "Show: Second Guest");
+ assert.equal(info.upload_date, "20190310");
+ const page = JSON.parse(await readFile(path.join(dataDir, YT, "metadata.info.json"), "utf8"));
+ assert.equal(page.title, "An archived upload");
+ assert.equal(page.upload_date, "20090102");
+ const history = await loadMetadataHistory(path.join(dataDir, JW_A));
+ assert.equal(history?.entries.at(-1)?.by, "wayback-provenance");
+
+ const again = await refreshWaybackRecords({ slug: SLUG, paths, titles });
+ assert.deepEqual(
+ again.records.map((x) => [x.from, x.to, x.rename ?? null, x.sidecar, Object.keys(x.changes).length]),
+ [
+ [YT, YT, null, false, 0],
+ [JW_A, JW_A, null, false, 0],
+ [JW_B, JW_B, null, false, 0],
+ ],
+ );
+ });
+});
+
+test("a record a live job names is held; a dead writer's job is not", async () => {
+ await withCorpus(async (paths, dataDir) => {
+ const meta = (id: string, videoId: string, pid: number) =>
+ writeFile(
+ path.join(paths.jobsDir, `${id}.meta.json`),
+ JSON.stringify({ id, kind: "transcribe-one", queueKey: "transcription", channelSlug: SLUG, videoId, status: "running", queuedAt: 1, pid }),
+ );
+ await meta("01AAAAAAAAAAAAAAAAAAAAAAAA", `${JW_A}-12345678.mp4`, process.ppid);
+ await meta("01BBBBBBBBBBBBBBBBBBBBBBBB", "watch", 2 ** 22 + 12345);
+ const r = await refreshWaybackRecords({ slug: SLUG, paths, titles });
+ const a = r.records.find((x) => x.from === `${JW_A}-12345678.mp4`)!;
+ assert.equal(a.rename, "held");
+ assert.equal(a.to, a.from);
+ assert.deepEqual(a.changes, {});
+ assert.equal(r.records.find((x) => x.from === "watch")!.rename, "renamed");
+ assert.deepEqual((await readdir(dataDir)).sort(), [`${JW_A}-12345678.mp4`, JW_B, "Plain12345a", YT].sort());
+ });
+});
+
+test("renameRosterEntries moves an entry and keeps the earlier sighting", () => {
+ const e = (url: string, firstSeenAt: string) => ({ url, firstSeenAt, lastListedAt: firstSeenAt, source: "import" as const });
+ const roster: Roster = {
+ version: 1,
+ updatedAt: "",
+ lastSweep: null,
+ entries: { old: e("u-old", "2026-01-01"), both: e("u-both", "2026-01-01"), keep: e("u-keep", "2026-03-03") },
+ };
+ const next = renameRosterEntries(roster, [{ from: "old", to: "new" }, { from: "both", to: "keep" }], "now");
+ assert.deepEqual(Object.keys(next.entries).sort(), ["keep", "new"]);
+ assert.equal(next.entries.new.url, "u-old");
+ assert.equal(next.entries.keep.firstSeenAt, "2026-01-01");
+ assert.equal(next.entries.keep.url, "u-keep");
+ assert.equal(renameRosterEntries(next, [{ from: "gone", to: "x" }], "now"), next);
+});
+
+test("the CLI: --titles from a file, old → new printed, exit 0", async () => {
+ await withCorpus(async (paths, dataDir) => {
+ const file = path.join(paths.transcriptsDir, "titles.json");
+ await writeFile(file, JSON.stringify(titles));
+ const lines: string[] = [];
+ const orig = console.log;
+ console.log = (...a: unknown[]) => void lines.push(a.join(" "));
+ try {
+ assert.equal(await refreshCli({ slug: SLUG, dryRun: false, titlesFile: file, paths }), 0);
+ } finally {
+ console.log = orig;
+ }
+ const out = lines.join("\n");
+ assert.match(out, new RegExp(`watch → ${YT}`));
+ assert.match(out, new RegExp(`${JW_A}-12345678\\.mp4 → ${JW_A}`));
+ assert.match(out, /3 Wayback records, 3 renamed, 3 wayback\.json written, 2 retitled\./);
+ assert.ok((await readdir(dataDir)).includes(YT));
+ assert.equal(await refreshCli({ slug: "Not A Slug", dryRun: true, paths }), 2);
+ });
+});
diff --git a/common/controller/waybackRefresh.ts b/common/controller/waybackRefresh.ts
@@ -0,0 +1,307 @@
+// WAYBACK MACHINE COPIES, BROUGHT UP TO THE WAYBACK RULES — offline.
+//
+// A record imported from a Wayback capture (lib/wayback.ts) before the app
+// knew what one was carries no `wayback.json`, and its dir is named by the last
+// segment of the capture URL: `watch` for an archived YouTube page (a name
+// every such capture shares), `<jwId>-<rendition>.mp4` for a raw JW Player
+// file. This brings every such record of a channel up to the rules:
+//
+// wayback.json written from the capture URL (lib/wayback-server.ts)
+// data/<id>/ renamed to its canonical id (lib/videoId.ts) by the
+// snapshot's own pass, reconcileVideoDirs — the media
+// tier's relative links move with the dir, and a dir
+// already there is merged, never overwritten
+// roster.json the entry moved to the new id (renameRosterEntries)
+// metadata.info.json with `titles`: the title and upload_date the operator
+// found for a record that has none (a raw file's title
+// is its file name), through `patchMetadataInfo`, so the
+// change is in metadata.history.json as
+// `wayback-provenance`
+//
+// A record a live job names (a `.jobs/` meta, queued or running, whose writer
+// is alive) is skipped and reported; a live job over the whole channel holds
+// every record. Only what differs is written, so a second run writes nothing.
+// No network.
+
+import path from "node:path";
+import { readdir, readFile } from "node:fs/promises";
+import { getPaths, type Paths } from "../lib/paths";
+import { readJsonFile } from "../lib/jsonFile-server";
+import { assertChannelTextReadable, readRelocationMarker } from "../lib/channelMedia";
+import { parseWaybackUrl } from "../lib/wayback";
+import { ensureWaybackProvenance } from "../lib/wayback-server";
+import { extractVideoId } from "../lib/videoId";
+import { patchMetadataInfo } from "../lib/metadataHistory-server";
+import { readChannelConfig } from "./channels";
+import { listChannelVideoIds } from "./keptVideos";
+import { reconcileVideoDirs } from "./reconcileVideoDirs";
+import { loadRoster, renameRosterEntries, writeRoster } from "./rosterStore";
+import { writerIsGone } from "../jobs/bootQueuedJobs";
+import type { JobMeta } from "../jobs/jobMeta";
+
+// What the operator found for a record: its title and the day it is of.
+export type WaybackTitle = { title?: string; upload_date?: string };
+
+export type WaybackRecordResult = {
+ // The dir's name before, and after (the same when it was not renamed).
+ from: string;
+ to: string;
+ // The capture the record was fetched from.
+ captureUrl: string;
+ rename?: "renamed" | "merged" | "conflict" | "held";
+ // Why a rename did not happen (a live job, a conflict, a capture that is
+ // not the record's webpage_url).
+ note?: string;
+ // wayback.json was (or would be) written.
+ sidecar: boolean;
+ // Per key: the value before and after.
+ changes: Partial<Record<"title" | "upload_date", { from: unknown; to: unknown }>>;
+};
+
+export type WaybackRefreshResult = {
+ records: WaybackRecordResult[];
+ // `titles` entries no Wayback record of the channel matched.
+ unmatchedTitles: string[];
+ failed: { id: string; error: string }[];
+};
+
+type Info = Record<string, unknown>;
+
+async function readInfo(videoDir: string): Promise<Info | null> {
+ const read = await readJsonFile(path.join(videoDir, "metadata.info.json"));
+ return read.ok && read.value && typeof read.value === "object" && !Array.isArray(read.value)
+ ? (read.value as Info)
+ : null;
+}
+
+// The URL the managed download ran yt-dlp on: the last argument of the
+// command line it logged (`$ yt-dlp … -- <url>`).
+async function urlFromDownloadLog(videoDir: string): Promise<string | null> {
+ let head: string;
+ try {
+ head = (await readFile(path.join(videoDir, "download.log"), "utf8")).slice(0, 16 * 1024);
+ } catch {
+ return null;
+ }
+ for (const line of head.split("\n")) {
+ const m = /^\$ yt-dlp .* -- (\S+)\s*$/.exec(line);
+ if (m && parseWaybackUrl(m[1])) return m[1];
+ }
+ return null;
+}
+
+// The capture a record was fetched from: its webpage_url, its original_url,
+// or the URL its download log ran on.
+async function captureUrlOf(videoDir: string, info: Info | null): Promise<string | null> {
+ for (const key of ["webpage_url", "original_url"]) {
+ const v = info?.[key];
+ if (typeof v === "string" && parseWaybackUrl(v)) return v;
+ }
+ return urlFromDownloadLog(videoDir);
+}
+
+// The live jobs on a channel: the record ids they name, and whether one of
+// them covers the whole channel.
+async function liveJobs(
+ paths: Paths,
+ slug: string,
+): Promise<{ ids: Set<string>; channelWide: string[] }> {
+ const ids = new Set<string>();
+ const channelWide: string[] = [];
+ const names = paths.jobsDir ? await readdir(paths.jobsDir).catch(() => [] as string[]) : [];
+ for (const name of names) {
+ if (!name.endsWith(".meta.json")) continue;
+ const read = await readJsonFile(path.join(paths.jobsDir, name));
+ if (!read.ok || !read.value || typeof read.value !== "object") continue;
+ const meta = read.value as JobMeta;
+ if (meta.channelSlug !== slug) continue;
+ if (meta.status !== "queued" && meta.status !== "running") continue;
+ if (writerIsGone(meta)) continue;
+ if (meta.videoId) ids.add(meta.videoId);
+ else channelWide.push(`${meta.kind} ${meta.id}`);
+ }
+ return { ids, channelWide };
+}
+
+// A title that is not one: absent, or the file's own name (what yt-dlp's
+// generic extractor titles a raw file with).
+function hasRealTitle(info: Info, names: string[]): boolean {
+ const t = typeof info.title === "string" ? info.title.trim() : "";
+ if (!t) return false;
+ const own = new Set<string>(names);
+ for (const key of ["id", "display_id", "webpage_url_basename"]) {
+ const v = info[key];
+ if (typeof v === "string") {
+ own.add(v);
+ own.add(v.replace(/\.[A-Za-z0-9]{2,4}$/, ""));
+ }
+ }
+ return !own.has(t);
+}
+
+// `YYYYMMDD` from `YYYYMMDD` or `YYYY-MM-DD`; null otherwise.
+export function normalizeUploadDate(v: unknown): string | null {
+ if (typeof v !== "string") return null;
+ const s = v.trim();
+ if (/^\d{8}$/.test(s)) return s;
+ const m = /^(\d{4})-(\d{2})-(\d{2})$/.exec(s);
+ return m ? `${m[1]}${m[2]}${m[3]}` : null;
+}
+
+export async function refreshWaybackRecords(opts: {
+ slug: string;
+ paths?: Paths;
+ dryRun?: boolean;
+ // Record id (its new id, or its old dir name) → its title and date.
+ titles?: Record<string, WaybackTitle>;
+ onLog?: (line: string) => void;
+ now?: () => Date;
+}): Promise<WaybackRefreshResult> {
+ const paths = opts.paths ?? getPaths();
+ const log = opts.onLog ?? (() => {});
+ const dryRun = opts.dryRun === true;
+ const config = await readChannelConfig(paths, opts.slug);
+ if (!config) throw new Error(`Channel "${opts.slug}" not found`);
+ // An unreadable text tier is not an empty channel (AGENTS.md).
+ await assertChannelTextReadable(paths, opts.slug, config);
+ if (await readRelocationMarker(paths, opts.slug)) {
+ throw new Error(`Channel "${opts.slug}" is relocating (.relocating.json) — run again when the move is done`);
+ }
+
+ const channelDir = path.join(paths.channelsDir, opts.slug);
+ const dataDir = path.join(channelDir, "data");
+ const result: WaybackRefreshResult = { records: [], unmatchedTitles: [], failed: [] };
+ const jobs = await liveJobs(paths, opts.slug);
+
+ // ─── Find the Wayback records ───
+ type Found = { rec: WaybackRecordResult; info: Info | null; canonical: string | null };
+ const found: Found[] = [];
+ for (const name of (await listChannelVideoIds(paths, opts.slug)).sort()) {
+ const videoDir = path.join(dataDir, name);
+ try {
+ const info = await readInfo(videoDir);
+ const captureUrl = await captureUrlOf(videoDir, info);
+ if (!captureUrl) continue;
+ // The snapshot names a dir by its webpage_url (reconcileVideoDirs.ts),
+ // so that is the only name a rename can give it that lasts.
+ const webpageUrl = typeof info?.webpage_url === "string" ? info.webpage_url : null;
+ const canonical = webpageUrl && parseWaybackUrl(webpageUrl) ? extractVideoId(webpageUrl) : null;
+ const rec: WaybackRecordResult = { from: name, to: name, captureUrl, sidecar: false, changes: {} };
+ if (canonical && canonical !== name) rec.to = canonical;
+ else if (!canonical && webpageUrl !== captureUrl) {
+ rec.note = "webpage_url is not the capture; the dir keeps its name";
+ }
+ found.push({ rec, info, canonical });
+ } catch (err) {
+ result.failed.push({ id: name, error: (err as Error).message });
+ }
+ }
+
+ const held = (rec: WaybackRecordResult): string | null => {
+ if (jobs.channelWide.length > 0) return `a live job holds the channel (${jobs.channelWide.join(", ")})`;
+ if (jobs.ids.has(rec.from) || jobs.ids.has(rec.to)) return "a live job names this record";
+ return null;
+ };
+
+ // ─── Rename, through the snapshot's own pass ───
+ const toRename = new Set<string>();
+ for (const { rec } of found) {
+ if (rec.to === rec.from) continue;
+ const why = held(rec);
+ if (why) {
+ rec.rename = "held";
+ rec.note = why;
+ rec.to = rec.from;
+ continue;
+ }
+ toRename.add(rec.from);
+ }
+ if (toRename.size > 0) {
+ const r = await reconcileVideoDirs({ channelDir, dryRun, only: (name) => toRename.has(name) });
+ const byFrom = new Map(found.map((f) => [f.rec.from, f.rec]));
+ for (const x of r.renamed) {
+ const rec = byFrom.get(x.from);
+ if (rec) rec.rename = "renamed";
+ }
+ for (const x of r.merged) {
+ const rec = byFrom.get(x.from);
+ if (rec) {
+ rec.rename = "merged";
+ rec.note = `merged into the existing ${x.to}/ (${x.movedFiles.length} files)`;
+ }
+ }
+ for (const x of r.conflicts) {
+ const rec = byFrom.get(x.from);
+ if (rec) {
+ rec.rename = "conflict";
+ rec.note = x.reason;
+ rec.to = rec.from;
+ }
+ }
+ const moved = [...r.renamed, ...r.merged].map((x) => ({ from: x.from, to: x.to }));
+ if (!dryRun && moved.length > 0) {
+ const now = (opts.now?.() ?? new Date()).toISOString();
+ const before = await loadRoster(paths, opts.slug);
+ const after = renameRosterEntries(before, moved, now);
+ if (after !== before) await writeRoster(paths, opts.slug, after);
+ }
+ }
+
+ // ─── The sidecar, and the titles ───
+ const titles = opts.titles ?? {};
+ const usedTitles = new Set<string>();
+ for (const { rec, info } of found) {
+ // A dry run renamed nothing: the record is still under its old name.
+ const videoDir = path.join(dataDir, dryRun ? rec.from : rec.to);
+ try {
+ const s = await ensureWaybackProvenance(videoDir, rec.captureUrl, { dryRun });
+ rec.sidecar = s.written;
+ const key = rec.to in titles ? rec.to : rec.from in titles ? rec.from : null;
+ if (key === null || !info) continue;
+ usedTitles.add(key);
+ if (held(rec)) {
+ rec.note = rec.note ?? held(rec)!;
+ continue;
+ }
+ const want = titles[key];
+ const patch: Record<string, unknown> = {};
+ const title = typeof want.title === "string" ? want.title.trim() : "";
+ if (title && !hasRealTitle(info, [rec.from, rec.to]) && info.title !== title) {
+ patch.title = title;
+ rec.changes.title = { from: info.title ?? null, to: title };
+ }
+ const date = normalizeUploadDate(want.upload_date);
+ if (want.upload_date !== undefined && !date) {
+ rec.note = `upload_date "${String(want.upload_date)}" is not YYYYMMDD or YYYY-MM-DD`;
+ } else if (date && normalizeUploadDate(info.upload_date) === null) {
+ patch.upload_date = date;
+ rec.changes.upload_date = { from: info.upload_date ?? null, to: date };
+ }
+ if (Object.keys(patch).length > 0 && !dryRun) {
+ await patchMetadataInfo(videoDir, patch, { by: "wayback-provenance", requestedBy: "cli", onLog: log });
+ }
+ } catch (err) {
+ result.failed.push({ id: rec.from, error: (err as Error).message });
+ }
+ }
+ result.unmatchedTitles = Object.keys(titles).filter((k) => !usedTitles.has(k)).sort();
+ result.records = found.map((f) => f.rec);
+ for (const rec of result.records) log(formatWaybackRecord(rec, dryRun));
+ return result;
+}
+
+// One record as the CLI prints it: `old → new`, then what was written.
+export function formatWaybackRecord(rec: WaybackRecordResult, dryRun: boolean): string {
+ const would = dryRun ? "would be " : "";
+ const head =
+ rec.from === rec.to
+ ? `${rec.from} (name kept)`
+ : `${rec.from} → ${rec.to}${rec.rename === "merged" ? " (merged)" : ""}`;
+ const lines = [head];
+ if (rec.note) lines.push(` ${rec.rename === "held" || rec.rename === "conflict" ? "NOT RENAMED: " : ""}${rec.note}`);
+ if (rec.sidecar) lines.push(` wayback.json ${would}written`);
+ for (const [k, c] of Object.entries(rec.changes)) {
+ lines.push(` ${k}: ${JSON.stringify(c!.from)} → ${JSON.stringify(c!.to)}`);
+ }
+ return lines.join("\n");
+}
diff --git a/common/lib/metadataHistory.ts b/common/lib/metadataHistory.ts
@@ -68,6 +68,10 @@ export const METADATA_HISTORY_WRITERS = [
// metadata API (controller/archiveOrgDownload.ts) — archive.org records are
// not fetched by yt-dlp.
"archiveorg-import",
+ // A Wayback Machine copy's title and date, set from what the operator found
+ // for it (`archilyzer wayback refresh --titles`, controller/waybackRefresh.ts)
+ // — a raw media file captured by the Wayback Machine carries no title.
+ "wayback-provenance",
] as const;
export type MetadataHistoryWriter = (typeof METADATA_HISTORY_WRITERS)[number];