import { test } from "node:test"; import assert from "node:assert/strict"; import { lstat, mkdir, mkdtemp, readFile, readdir, readlink, stat, symlink, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import type { Paths } from "../lib/paths"; import { buildWaybackProvenance } from "../lib/wayback"; import { loadMetadataHistory } from "../lib/metadataHistory-server"; import { refreshWaybackRecords } from "./waybackRefresh"; import { renameRosterEntries, type Roster } from "./rosterStore"; import { main as refreshCli } from "../bin/wayback-refresh"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test controller/waybackRefresh.test.ts // // A temp corpus of Wayback copies imported before the app knew what one was: // an archived YouTube page in `watch/`, two raw JW Player files in // `-.mp4/`, and a plain record. Every id here is invented. const SLUG = "demo-wayback"; const YT = "Abc123def45"; const PAGE = `https://web.archive.org/web/20220102030405/https://www.youtube.com/watch?v=${YT}`; const JW_A = "Qw3rTy12"; const JW_B = "Zx9vBn34"; const fileUrl = (jw: string, ts: string) => `https://web.archive.org/web/${ts}id_/https://videos-fms.jwpsrv.com/content/conversions/AcCt1234/videos/${jw}-12345678.mp4?token=0_abc_0xdef`; const FILE_A = fileUrl(JW_A, "20200102030405"); const FILE_B = fileUrl(JW_B, "20200103030405"); async function withCorpus( fn: (paths: Paths, dataDir: string, channelDir: string) => Promise, ): Promise { const dir = await mkdtemp(path.join(tmpdir(), "ttb-wayback-")); const transcriptsDir = path.join(dir, "corpus"); const paths = { transcriptsDir, channelsDir: path.join(transcriptsDir, "channels"), jobsDir: path.join(transcriptsDir, ".jobs"), } as Paths; const channelDir = path.join(paths.channelsDir, SLUG); const dataDir = path.join(channelDir, "data"); await mkdir(dataDir, { recursive: true }); await mkdir(paths.jobsDir, { recursive: true }); await writeFile( path.join(channelDir, "config.json"), JSON.stringify({ name: "Demo", platform: "archiveorg", handling: "transcribe" }), ); const record = async (name: string, info: Record) => { await mkdir(path.join(dataDir, name), { recursive: true }); await writeFile(path.join(dataDir, name, "metadata.info.json"), JSON.stringify(info)); }; await record("watch", { id: YT, title: "An archived upload", upload_date: "20090102", extractor_key: "YoutubeWebArchive", webpage_url: PAGE, webpage_url_basename: "watch", }); for (const [jw, url] of [[JW_A, FILE_A], [JW_B, FILE_B]] as const) { await record(`${jw}-12345678.mp4`, { id: `${jw}-12345678`, title: `${jw}-12345678`, extractor_key: "Generic", webpage_url: url, webpage_url_basename: `${jw}-12345678.mp4`, }); } await record("Plain12345a", { id: "Plain12345a", title: "Plain", upload_date: "20200101", extractor_key: "Youtube", webpage_url: "https://www.youtube.com/watch?v=Plain12345a" }); // A tiered file: a relative link into media// (release 17). await mkdir(path.join(channelDir, "media", `${JW_A}-12345678.mp4`), { recursive: true }); await writeFile(path.join(channelDir, "media", `${JW_A}-12345678.mp4`, "audio.mp3"), "audio"); await symlink( path.join("..", "..", "media", `${JW_A}-12345678.mp4`, "audio.mp3"), path.join(dataDir, `${JW_A}-12345678.mp4`, "audio.mp3"), ); const entry = (url: string) => ({ url, firstSeenAt: "2026-01-01T00:00:00.000Z", lastListedAt: "2026-01-01T00:00:00.000Z", source: "import" }); await writeFile( path.join(channelDir, "roster.json"), JSON.stringify({ version: 1, updatedAt: "2026-01-01T00:00:00.000Z", lastSweep: null, entries: { watch: entry(PAGE), [`${JW_A}-12345678.mp4`]: entry(FILE_A), [`${JW_B}-12345678.mp4`]: entry(FILE_B), Plain12345a: entry("https://www.youtube.com/watch?v=Plain12345a"), }, }), ); await fn(paths, dataDir, channelDir); } const titles = { [JW_A]: { title: "Show: First Guest", upload_date: "2019-03-03" }, [`${JW_B}-12345678.mp4`]: { title: "Show: Second Guest", upload_date: "20190310" }, [YT]: { title: "Not applied: the page has a title", upload_date: "20000101" }, nobody: { title: "No such record" }, }; test("a dry run reports every rename and title and writes nothing", async () => { await withCorpus(async (paths, dataDir, channelDir) => { const rosterBefore = await readFile(path.join(channelDir, "roster.json"), "utf8"); const r = await refreshWaybackRecords({ slug: SLUG, paths, dryRun: true, titles }); assert.deepEqual( r.records.map((x) => [x.from, x.to, x.rename, x.sidecar]), [ [`${JW_A}-12345678.mp4`, JW_A, "renamed", true], [`${JW_B}-12345678.mp4`, JW_B, "renamed", true], ["watch", YT, "renamed", true], ], ); assert.deepEqual(r.records[0].changes, { title: { from: `${JW_A}-12345678`, to: "Show: First Guest" }, upload_date: { from: null, to: "20190303" }, }); assert.deepEqual(r.records[2].changes, {}); assert.deepEqual(r.unmatchedTitles, ["nobody"]); assert.deepEqual((await readdir(dataDir)).sort(), [`${JW_A}-12345678.mp4`, `${JW_B}-12345678.mp4`, "Plain12345a", "watch"].sort()); assert.equal(await readFile(path.join(channelDir, "roster.json"), "utf8"), rosterBefore); await assert.rejects(stat(path.join(dataDir, "watch", "wayback.json"))); }); }); test("a real run renames through the reconcile pass, moves the roster, writes sidecars and titles; a second run changes nothing", async () => { await withCorpus(async (paths, dataDir, channelDir) => { const r = await refreshWaybackRecords({ slug: SLUG, paths, titles, now: () => new Date("2026-02-02T00:00:00.000Z") }); assert.equal(r.failed.length, 0); assert.deepEqual((await readdir(dataDir)).sort(), [JW_A, JW_B, "Plain12345a", YT].sort()); // The tier link moved with its dir and still resolves. const link = path.join(dataDir, JW_A, "audio.mp3"); assert.ok((await lstat(link)).isSymbolicLink()); assert.equal(await readlink(link), path.join("..", "..", "media", `${JW_A}-12345678.mp4`, "audio.mp3")); assert.equal(await readFile(link, "utf8"), "audio"); const roster = JSON.parse(await readFile(path.join(channelDir, "roster.json"), "utf8")); assert.deepEqual(Object.keys(roster.entries).sort(), [JW_A, JW_B, "Plain12345a", YT].sort()); assert.equal(roster.entries[YT].url, PAGE); assert.equal(roster.entries[YT].firstSeenAt, "2026-01-01T00:00:00.000Z"); assert.deepEqual(JSON.parse(await readFile(path.join(dataDir, YT, "wayback.json"), "utf8")), buildWaybackProvenance(PAGE)); assert.equal(JSON.parse(await readFile(path.join(dataDir, JW_A, "wayback.json"), "utf8")).originalUrl, FILE_A.replace(/^.*?id_\//, "")); await assert.rejects(stat(path.join(dataDir, "Plain12345a", "wayback.json"))); const info = JSON.parse(await readFile(path.join(dataDir, JW_B, "metadata.info.json"), "utf8")); assert.equal(info.title, "Show: Second Guest"); assert.equal(info.upload_date, "20190310"); const page = JSON.parse(await readFile(path.join(dataDir, YT, "metadata.info.json"), "utf8")); assert.equal(page.title, "An archived upload"); assert.equal(page.upload_date, "20090102"); const history = await loadMetadataHistory(path.join(dataDir, JW_A)); assert.equal(history?.entries.at(-1)?.by, "wayback-provenance"); const again = await refreshWaybackRecords({ slug: SLUG, paths, titles }); assert.deepEqual( again.records.map((x) => [x.from, x.to, x.rename ?? null, x.sidecar, Object.keys(x.changes).length]), [ [YT, YT, null, false, 0], [JW_A, JW_A, null, false, 0], [JW_B, JW_B, null, false, 0], ], ); }); }); test("a record a live job names is held; a dead writer's job is not", async () => { await withCorpus(async (paths, dataDir) => { const meta = (id: string, videoId: string, pid: number) => writeFile( path.join(paths.jobsDir, `${id}.meta.json`), JSON.stringify({ id, kind: "transcribe-one", queueKey: "transcription", channelSlug: SLUG, videoId, status: "running", queuedAt: 1, pid }), ); await meta("01AAAAAAAAAAAAAAAAAAAAAAAA", `${JW_A}-12345678.mp4`, process.ppid); await meta("01BBBBBBBBBBBBBBBBBBBBBBBB", "watch", 2 ** 22 + 12345); // An auto-queue lane's newest pick is in flight; an old one is not. paths.autoQueueStateFile = path.join(paths.transcriptsDir, ".auto-queue", "state.json"); await mkdir(path.dirname(paths.autoQueueStateFile), { recursive: true }); const at = Date.parse("2026-02-02T00:00:00.000Z"); await writeFile( paths.autoQueueStateFile, JSON.stringify({ transcription: { picks: [ { at, leafId: "x", videoId: `${JW_B}-12345678.mp4`, channelSlug: SLUG }, // Finished: its outcome sidecar is newer than the pick (below). { at: at - 1000, leafId: "x", videoId: "watch", channelSlug: SLUG }, ], }, // Too old to be in flight. download: { picks: [{ at: at - 7 * 3600 * 1000, leafId: "x", videoId: "watch", channelSlug: SLUG }] }, }), ); await writeFile(path.join(dataDir, "watch", "transcribe-outcome.json"), "{}"); const r = await refreshWaybackRecords({ slug: SLUG, paths, titles, now: () => new Date(at + 60_000) }); assert.equal(r.records.find((x) => x.from === `${JW_B}-12345678.mp4`)!.rename, "held"); const a = r.records.find((x) => x.from === `${JW_A}-12345678.mp4`)!; assert.equal(a.rename, "held"); assert.equal(a.to, a.from); assert.deepEqual(a.changes, {}); assert.equal(r.records.find((x) => x.from === "watch")!.rename, "renamed"); assert.deepEqual((await readdir(dataDir)).sort(), [`${JW_A}-12345678.mp4`, `${JW_B}-12345678.mp4`, "Plain12345a", YT].sort()); }); }); test("renameRosterEntries moves an entry and keeps the earlier sighting", () => { const e = (url: string, firstSeenAt: string) => ({ url, firstSeenAt, lastListedAt: firstSeenAt, source: "import" as const }); const roster: Roster = { version: 1, updatedAt: "", lastSweep: null, entries: { old: e("u-old", "2026-01-01"), both: e("u-both", "2026-01-01"), keep: e("u-keep", "2026-03-03") }, }; const next = renameRosterEntries(roster, [{ from: "old", to: "new" }, { from: "both", to: "keep" }], "now"); assert.deepEqual(Object.keys(next.entries).sort(), ["keep", "new"]); assert.equal(next.entries.new.url, "u-old"); assert.equal(next.entries.keep.firstSeenAt, "2026-01-01"); assert.equal(next.entries.keep.url, "u-keep"); assert.equal(renameRosterEntries(next, [{ from: "gone", to: "x" }], "now"), next); }); test("the CLI: --titles from a file, old → new printed, exit 0", async () => { await withCorpus(async (paths, dataDir) => { const file = path.join(paths.transcriptsDir, "titles.json"); await writeFile(file, JSON.stringify(titles)); const lines: string[] = []; const orig = console.log; console.log = (...a: unknown[]) => void lines.push(a.join(" ")); try { assert.equal(await refreshCli({ slug: SLUG, dryRun: false, titlesFile: file, paths }), 0); } finally { console.log = orig; } const out = lines.join("\n"); assert.match(out, new RegExp(`watch → ${YT}`)); assert.match(out, new RegExp(`${JW_A}-12345678\\.mp4 → ${JW_A}`)); assert.match(out, /3 Wayback records, 3 renamed, 3 wayback\.json written, 2 retitled\./); assert.ok((await readdir(dataDir)).includes(YT)); assert.equal(await refreshCli({ slug: "Not A Slug", dryRun: true, paths }), 2); }); });