Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 5a14b2addb79a5c6ea1535e1b00cc3deccff6172
parent 9d8e7303ace208fdf95a13d21810268235868d0e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  6 Oct 2026 09:01:22 -0400

wayback refresh: bring imported Wayback copies up to the rules, offline

`archilyzer wayback refresh <slug> [--titles <file>] [--dry-run]`
(controller/waybackRefresh.ts): for every record whose webpage_url, original_url or
download log names a capture, writes `wayback.json`, renames the dir to its canonical id
through reconcileVideoDirs (new `only` filter; tier links move with the dir, an existing
dir is merged) and moves its roster entry (renameRosterEntries), and with --titles sets a
raw file's title and upload_date through patchMetadataInfo as `wayback-provenance`. A
record a live job names (queued/running meta, writer alive) is held; a live channel-wide
job holds every record. Idempotent; prints old → new.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/bin/archilyzer.ts | 19+++++++++++++++++++
Acommon/bin/wayback-refresh.ts | 70++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/reconcileVideoDirs.ts | 9++++++++-
Mcommon/controller/rosterStore.ts | 27+++++++++++++++++++++++++++
Acommon/controller/waybackRefresh.test.ts | 222+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/waybackRefresh.ts | 307+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/metadataHistory.ts | 4++++
7 files changed, 657 insertions(+), 1 deletion(-)

diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts @@ -351,6 +351,25 @@ export const COMMANDS: Command[] = [ }, }, { + path: ["wayback", "refresh"], + usage: + "<slug> [--titles <json>] [--dry-run] bring a channel's Wayback Machine copies up to the Wayback rules: wayback.json (original URL, capture time), the dir renamed to its canonical id through the snapshot's reconcile (roster moved with it), and with --titles (a file of id → {title, upload_date}) a raw file's title and date; offline, skips a record a live job holds; prints old → new", + flags: { titles: "string", "dry-run": "boolean" }, + maxPositionals: 1, + run: async ({ positionals, flags }) => { + const [slug] = positionals; + if (!slug) { + console.error("wayback refresh: which channel? Pass its slug."); + return 2; + } + return (await import("./wayback-refresh")).main({ + slug, + dryRun: flags["dry-run"] === true, + ...(typeof flags.titles === "string" ? { titlesFile: flags.titles } : {}), + }); + }, + }, + { path: ["feeds", "backfill-metadata"], usage: "<slug> [--feed <url>] [--dry-run] complete a podcast channel's records (title, date, description, duration) from its RSS feed: one fetch of the feed (default: the channel's url), no media; --dry-run counts matched / unmatched / already complete and writes nothing", diff --git a/common/bin/wayback-refresh.ts b/common/bin/wayback-refresh.ts @@ -0,0 +1,70 @@ +// `archilyzer wayback refresh <slug> [--titles <json>] [--dry-run]` — every +// Wayback Machine copy of a channel brought up to the Wayback rules +// (controller/waybackRefresh.ts): its `wayback.json`, its dir under its +// canonical id (through the snapshot's own reconcile pass, roster moved with +// it), and with `--titles` a file mapping id → {title, upload_date} for a raw +// file that has neither. Offline. Prints old → new per record; a second run +// changes nothing. + +import { readFile } from "node:fs/promises"; +import { isValidChannelSlug } from "../controller/channels"; +import { refreshWaybackRecords, type WaybackTitle } from "../controller/waybackRefresh"; +import { getPaths, type Paths } from "../lib/paths"; + +async function readTitles(file: string): Promise<Record<string, WaybackTitle>> { + const v = JSON.parse(await readFile(file, "utf8")) as unknown; + if (!v || typeof v !== "object" || Array.isArray(v)) { + throw new Error(`${file}: expected an object of id → {title, upload_date}`); + } + const out: Record<string, WaybackTitle> = {}; + for (const [id, raw] of Object.entries(v as Record<string, unknown>)) { + if (!raw || typeof raw !== "object" || Array.isArray(raw)) throw new Error(`${file}: "${id}" is not an object`); + const r = raw as Record<string, unknown>; + out[id] = { + ...(typeof r.title === "string" ? { title: r.title } : {}), + ...(typeof r.upload_date === "string" ? { upload_date: r.upload_date } : {}), + }; + } + return out; +} + +export async function main(opts: { + slug: string; + dryRun: boolean; + titlesFile?: string; + paths?: Paths; +}): Promise<number> { + if (!isValidChannelSlug(opts.slug)) { + console.error(`wayback refresh: "${opts.slug}" is not a channel slug`); + return 2; + } + try { + const titles = opts.titlesFile ? await readTitles(opts.titlesFile) : undefined; + const r = await refreshWaybackRecords({ + slug: opts.slug, + paths: opts.paths ?? getPaths(), + dryRun: opts.dryRun, + titles, + onLog: (line) => console.log(line.replace(/\n$/, "")), + }); + const would = opts.dryRun ? "would be " : ""; + const renamed = r.records.filter((x) => x.rename === "renamed" || x.rename === "merged").length; + const held = r.records.filter((x) => x.rename === "held" || x.rename === "conflict").length; + const sidecars = r.records.filter((x) => x.sidecar).length; + const patched = r.records.filter((x) => Object.keys(x.changes).length > 0).length; + for (const id of r.unmatchedTitles) console.log(`--titles: no Wayback record "${id}"`); + for (const f of r.failed) console.error(`${f.id}: failed — ${f.error}`); + console.log( + `${opts.dryRun ? "dry run: " : ""}${r.records.length} Wayback records, ` + + `${renamed} ${would}renamed, ${sidecars} wayback.json ${would}written, ` + + `${patched} ${would}retitled` + + (held > 0 ? `, ${held} not renamed` : "") + + (r.failed.length > 0 ? `, ${r.failed.length} failed` : "") + + ".", + ); + return r.failed.length > 0 ? 1 : 0; + } catch (err) { + console.error(`wayback refresh: ${(err as Error).message}`); + return 1; + } +} diff --git a/common/controller/reconcileVideoDirs.ts b/common/controller/reconcileVideoDirs.ts @@ -28,6 +28,10 @@ export type ReconcileOpts = { channelDir: string; dryRun?: boolean; onLog?: (s: string) => void; + // Reconcile only the dirs this accepts (default: every dir). A caller that + // knows which records it means to move (`archilyzer wayback refresh`, + // controller/waybackRefresh.ts) leaves the rest to the snapshot's pass. + only?: (dirName: string) => boolean; }; // Files the canonical dir's copy should win on a name collision: these are @@ -136,7 +140,10 @@ export async function reconcileVideoDirs( const entries = await readdir(dataDir, { withFileTypes: true }).catch( () => [], ); - const dirNames = entries.filter((e) => e.isDirectory()).map((e) => e.name); + const dirNames = entries + .filter((e) => e.isDirectory()) + .map((e) => e.name) + .filter((name) => !opts.only || opts.only(name)); for (const name of dirNames) { try { diff --git a/common/controller/rosterStore.ts b/common/controller/rosterStore.ts @@ -141,6 +141,33 @@ export function mergeRoster( return { ...roster, version: ROSTER_VERSION, updatedAt: now, entries }; } +// A RENAMED RECORD keeps its roster entry under its new id. Not a removal: +// the entry moves (url, firstSeenAt, source and all), which is what a dir +// renamed to its canonical id (reconcileVideoDirs.ts) needs — left under the +// old id it would read as a video the channel has and nobody downloaded. +// When both ids have an entry the new one's stays and the earlier +// firstSeenAt wins. Returns the SAME object when nothing moved. +export function renameRosterEntries( + roster: Roster, + renames: ReadonlyArray<{ from: string; to: string }>, + now: string, +): Roster { + const entries: Record<string, RosterEntry> = { ...roster.entries }; + let changed = false; + for (const { from, to } of renames) { + const prev = entries[from]; + if (!prev || from === to) continue; + const there = entries[to]; + entries[to] = there + ? { ...there, firstSeenAt: prev.firstSeenAt && prev.firstSeenAt < there.firstSeenAt ? prev.firstSeenAt : there.firstSeenAt, url: there.url || prev.url } + : prev; + delete entries[from]; + changed = true; + } + if (!changed) return roster; + return { ...roster, version: ROSTER_VERSION, updatedAt: now, entries }; +} + // Stamp the outcome of an enumeration. Kept separate from mergeRoster because a // REJECTED enumeration still merges (additively, losing nothing) while recording // that its listing was not trusted — that record is what the next enumeration diff --git a/common/controller/waybackRefresh.test.ts b/common/controller/waybackRefresh.test.ts @@ -0,0 +1,222 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { lstat, mkdir, mkdtemp, readFile, readdir, readlink, stat, symlink, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import { buildWaybackProvenance } from "../lib/wayback"; +import { loadMetadataHistory } from "../lib/metadataHistory-server"; +import { refreshWaybackRecords } from "./waybackRefresh"; +import { renameRosterEntries, type Roster } from "./rosterStore"; +import { main as refreshCli } from "../bin/wayback-refresh"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/waybackRefresh.test.ts +// +// A temp corpus of Wayback copies imported before the app knew what one was: +// an archived YouTube page in `watch/`, two raw JW Player files in +// `<jwId>-<rendition>.mp4/`, and a plain record. Every id here is invented. + +const SLUG = "demo-wayback"; +const YT = "Abc123def45"; +const PAGE = `https://web.archive.org/web/20220102030405/https://www.youtube.com/watch?v=${YT}`; +const JW_A = "Qw3rTy12"; +const JW_B = "Zx9vBn34"; +const fileUrl = (jw: string, ts: string) => + `https://web.archive.org/web/${ts}id_/https://videos-fms.jwpsrv.com/content/conversions/AcCt1234/videos/${jw}-12345678.mp4?token=0_abc_0xdef`; +const FILE_A = fileUrl(JW_A, "20200102030405"); +const FILE_B = fileUrl(JW_B, "20200103030405"); + +async function withCorpus( + fn: (paths: Paths, dataDir: string, channelDir: string) => Promise<void>, +): Promise<void> { + const dir = await mkdtemp(path.join(tmpdir(), "ttb-wayback-")); + const transcriptsDir = path.join(dir, "corpus"); + const paths = { + transcriptsDir, + channelsDir: path.join(transcriptsDir, "channels"), + jobsDir: path.join(transcriptsDir, ".jobs"), + } as Paths; + const channelDir = path.join(paths.channelsDir, SLUG); + const dataDir = path.join(channelDir, "data"); + await mkdir(dataDir, { recursive: true }); + await mkdir(paths.jobsDir, { recursive: true }); + await writeFile( + path.join(channelDir, "config.json"), + JSON.stringify({ name: "Demo", platform: "archiveorg", handling: "transcribe" }), + ); + const record = async (name: string, info: Record<string, unknown>) => { + await mkdir(path.join(dataDir, name), { recursive: true }); + await writeFile(path.join(dataDir, name, "metadata.info.json"), JSON.stringify(info)); + }; + await record("watch", { + id: YT, + title: "An archived upload", + upload_date: "20090102", + extractor_key: "YoutubeWebArchive", + webpage_url: PAGE, + webpage_url_basename: "watch", + }); + for (const [jw, url] of [[JW_A, FILE_A], [JW_B, FILE_B]] as const) { + await record(`${jw}-12345678.mp4`, { + id: `${jw}-12345678`, + title: `${jw}-12345678`, + extractor_key: "Generic", + webpage_url: url, + webpage_url_basename: `${jw}-12345678.mp4`, + }); + } + await record("Plain12345a", { id: "Plain12345a", title: "Plain", upload_date: "20200101", extractor_key: "Youtube", webpage_url: "https://www.youtube.com/watch?v=Plain12345a" }); + // A tiered file: a relative link into media/<old id>/ (release 17). + await mkdir(path.join(channelDir, "media", `${JW_A}-12345678.mp4`), { recursive: true }); + await writeFile(path.join(channelDir, "media", `${JW_A}-12345678.mp4`, "audio.mp3"), "audio"); + await symlink( + path.join("..", "..", "media", `${JW_A}-12345678.mp4`, "audio.mp3"), + path.join(dataDir, `${JW_A}-12345678.mp4`, "audio.mp3"), + ); + const entry = (url: string) => ({ url, firstSeenAt: "2026-01-01T00:00:00.000Z", lastListedAt: "2026-01-01T00:00:00.000Z", source: "import" }); + await writeFile( + path.join(channelDir, "roster.json"), + JSON.stringify({ + version: 1, + updatedAt: "2026-01-01T00:00:00.000Z", + lastSweep: null, + entries: { + watch: entry(PAGE), + [`${JW_A}-12345678.mp4`]: entry(FILE_A), + [`${JW_B}-12345678.mp4`]: entry(FILE_B), + Plain12345a: entry("https://www.youtube.com/watch?v=Plain12345a"), + }, + }), + ); + await fn(paths, dataDir, channelDir); +} + +const titles = { + [JW_A]: { title: "Show: First Guest", upload_date: "2019-03-03" }, + [`${JW_B}-12345678.mp4`]: { title: "Show: Second Guest", upload_date: "20190310" }, + [YT]: { title: "Not applied: the page has a title", upload_date: "20000101" }, + nobody: { title: "No such record" }, +}; + +test("a dry run reports every rename and title and writes nothing", async () => { + await withCorpus(async (paths, dataDir, channelDir) => { + const rosterBefore = await readFile(path.join(channelDir, "roster.json"), "utf8"); + const r = await refreshWaybackRecords({ slug: SLUG, paths, dryRun: true, titles }); + assert.deepEqual( + r.records.map((x) => [x.from, x.to, x.rename, x.sidecar]), + [ + [`${JW_A}-12345678.mp4`, JW_A, "renamed", true], + [`${JW_B}-12345678.mp4`, JW_B, "renamed", true], + ["watch", YT, "renamed", true], + ], + ); + assert.deepEqual(r.records[0].changes, { + title: { from: `${JW_A}-12345678`, to: "Show: First Guest" }, + upload_date: { from: null, to: "20190303" }, + }); + assert.deepEqual(r.records[2].changes, {}); + assert.deepEqual(r.unmatchedTitles, ["nobody"]); + assert.deepEqual((await readdir(dataDir)).sort(), [`${JW_A}-12345678.mp4`, `${JW_B}-12345678.mp4`, "Plain12345a", "watch"].sort()); + assert.equal(await readFile(path.join(channelDir, "roster.json"), "utf8"), rosterBefore); + await assert.rejects(stat(path.join(dataDir, "watch", "wayback.json"))); + }); +}); + +test("a real run renames through the reconcile pass, moves the roster, writes sidecars and titles; a second run changes nothing", async () => { + await withCorpus(async (paths, dataDir, channelDir) => { + const r = await refreshWaybackRecords({ slug: SLUG, paths, titles, now: () => new Date("2026-02-02T00:00:00.000Z") }); + assert.equal(r.failed.length, 0); + assert.deepEqual((await readdir(dataDir)).sort(), [JW_A, JW_B, "Plain12345a", YT].sort()); + + // The tier link moved with its dir and still resolves. + const link = path.join(dataDir, JW_A, "audio.mp3"); + assert.ok((await lstat(link)).isSymbolicLink()); + assert.equal(await readlink(link), path.join("..", "..", "media", `${JW_A}-12345678.mp4`, "audio.mp3")); + assert.equal(await readFile(link, "utf8"), "audio"); + + const roster = JSON.parse(await readFile(path.join(channelDir, "roster.json"), "utf8")); + assert.deepEqual(Object.keys(roster.entries).sort(), [JW_A, JW_B, "Plain12345a", YT].sort()); + assert.equal(roster.entries[YT].url, PAGE); + assert.equal(roster.entries[YT].firstSeenAt, "2026-01-01T00:00:00.000Z"); + + assert.deepEqual(JSON.parse(await readFile(path.join(dataDir, YT, "wayback.json"), "utf8")), buildWaybackProvenance(PAGE)); + assert.equal(JSON.parse(await readFile(path.join(dataDir, JW_A, "wayback.json"), "utf8")).originalUrl, FILE_A.replace(/^.*?id_\//, "")); + await assert.rejects(stat(path.join(dataDir, "Plain12345a", "wayback.json"))); + + const info = JSON.parse(await readFile(path.join(dataDir, JW_B, "metadata.info.json"), "utf8")); + assert.equal(info.title, "Show: Second Guest"); + assert.equal(info.upload_date, "20190310"); + const page = JSON.parse(await readFile(path.join(dataDir, YT, "metadata.info.json"), "utf8")); + assert.equal(page.title, "An archived upload"); + assert.equal(page.upload_date, "20090102"); + const history = await loadMetadataHistory(path.join(dataDir, JW_A)); + assert.equal(history?.entries.at(-1)?.by, "wayback-provenance"); + + const again = await refreshWaybackRecords({ slug: SLUG, paths, titles }); + assert.deepEqual( + again.records.map((x) => [x.from, x.to, x.rename ?? null, x.sidecar, Object.keys(x.changes).length]), + [ + [YT, YT, null, false, 0], + [JW_A, JW_A, null, false, 0], + [JW_B, JW_B, null, false, 0], + ], + ); + }); +}); + +test("a record a live job names is held; a dead writer's job is not", async () => { + await withCorpus(async (paths, dataDir) => { + const meta = (id: string, videoId: string, pid: number) => + writeFile( + path.join(paths.jobsDir, `${id}.meta.json`), + JSON.stringify({ id, kind: "transcribe-one", queueKey: "transcription", channelSlug: SLUG, videoId, status: "running", queuedAt: 1, pid }), + ); + await meta("01AAAAAAAAAAAAAAAAAAAAAAAA", `${JW_A}-12345678.mp4`, process.ppid); + await meta("01BBBBBBBBBBBBBBBBBBBBBBBB", "watch", 2 ** 22 + 12345); + const r = await refreshWaybackRecords({ slug: SLUG, paths, titles }); + const a = r.records.find((x) => x.from === `${JW_A}-12345678.mp4`)!; + assert.equal(a.rename, "held"); + assert.equal(a.to, a.from); + assert.deepEqual(a.changes, {}); + assert.equal(r.records.find((x) => x.from === "watch")!.rename, "renamed"); + assert.deepEqual((await readdir(dataDir)).sort(), [`${JW_A}-12345678.mp4`, JW_B, "Plain12345a", YT].sort()); + }); +}); + +test("renameRosterEntries moves an entry and keeps the earlier sighting", () => { + const e = (url: string, firstSeenAt: string) => ({ url, firstSeenAt, lastListedAt: firstSeenAt, source: "import" as const }); + const roster: Roster = { + version: 1, + updatedAt: "", + lastSweep: null, + entries: { old: e("u-old", "2026-01-01"), both: e("u-both", "2026-01-01"), keep: e("u-keep", "2026-03-03") }, + }; + const next = renameRosterEntries(roster, [{ from: "old", to: "new" }, { from: "both", to: "keep" }], "now"); + assert.deepEqual(Object.keys(next.entries).sort(), ["keep", "new"]); + assert.equal(next.entries.new.url, "u-old"); + assert.equal(next.entries.keep.firstSeenAt, "2026-01-01"); + assert.equal(next.entries.keep.url, "u-keep"); + assert.equal(renameRosterEntries(next, [{ from: "gone", to: "x" }], "now"), next); +}); + +test("the CLI: --titles from a file, old → new printed, exit 0", async () => { + await withCorpus(async (paths, dataDir) => { + const file = path.join(paths.transcriptsDir, "titles.json"); + await writeFile(file, JSON.stringify(titles)); + const lines: string[] = []; + const orig = console.log; + console.log = (...a: unknown[]) => void lines.push(a.join(" ")); + try { + assert.equal(await refreshCli({ slug: SLUG, dryRun: false, titlesFile: file, paths }), 0); + } finally { + console.log = orig; + } + const out = lines.join("\n"); + assert.match(out, new RegExp(`watch → ${YT}`)); + assert.match(out, new RegExp(`${JW_A}-12345678\\.mp4 → ${JW_A}`)); + assert.match(out, /3 Wayback records, 3 renamed, 3 wayback\.json written, 2 retitled\./); + assert.ok((await readdir(dataDir)).includes(YT)); + assert.equal(await refreshCli({ slug: "Not A Slug", dryRun: true, paths }), 2); + }); +}); diff --git a/common/controller/waybackRefresh.ts b/common/controller/waybackRefresh.ts @@ -0,0 +1,307 @@ +// WAYBACK MACHINE COPIES, BROUGHT UP TO THE WAYBACK RULES — offline. +// +// A record imported from a Wayback capture (lib/wayback.ts) before the app +// knew what one was carries no `wayback.json`, and its dir is named by the last +// segment of the capture URL: `watch` for an archived YouTube page (a name +// every such capture shares), `<jwId>-<rendition>.mp4` for a raw JW Player +// file. This brings every such record of a channel up to the rules: +// +// wayback.json written from the capture URL (lib/wayback-server.ts) +// data/<id>/ renamed to its canonical id (lib/videoId.ts) by the +// snapshot's own pass, reconcileVideoDirs — the media +// tier's relative links move with the dir, and a dir +// already there is merged, never overwritten +// roster.json the entry moved to the new id (renameRosterEntries) +// metadata.info.json with `titles`: the title and upload_date the operator +// found for a record that has none (a raw file's title +// is its file name), through `patchMetadataInfo`, so the +// change is in metadata.history.json as +// `wayback-provenance` +// +// A record a live job names (a `.jobs/` meta, queued or running, whose writer +// is alive) is skipped and reported; a live job over the whole channel holds +// every record. Only what differs is written, so a second run writes nothing. +// No network. + +import path from "node:path"; +import { readdir, readFile } from "node:fs/promises"; +import { getPaths, type Paths } from "../lib/paths"; +import { readJsonFile } from "../lib/jsonFile-server"; +import { assertChannelTextReadable, readRelocationMarker } from "../lib/channelMedia"; +import { parseWaybackUrl } from "../lib/wayback"; +import { ensureWaybackProvenance } from "../lib/wayback-server"; +import { extractVideoId } from "../lib/videoId"; +import { patchMetadataInfo } from "../lib/metadataHistory-server"; +import { readChannelConfig } from "./channels"; +import { listChannelVideoIds } from "./keptVideos"; +import { reconcileVideoDirs } from "./reconcileVideoDirs"; +import { loadRoster, renameRosterEntries, writeRoster } from "./rosterStore"; +import { writerIsGone } from "../jobs/bootQueuedJobs"; +import type { JobMeta } from "../jobs/jobMeta"; + +// What the operator found for a record: its title and the day it is of. +export type WaybackTitle = { title?: string; upload_date?: string }; + +export type WaybackRecordResult = { + // The dir's name before, and after (the same when it was not renamed). + from: string; + to: string; + // The capture the record was fetched from. + captureUrl: string; + rename?: "renamed" | "merged" | "conflict" | "held"; + // Why a rename did not happen (a live job, a conflict, a capture that is + // not the record's webpage_url). + note?: string; + // wayback.json was (or would be) written. + sidecar: boolean; + // Per key: the value before and after. + changes: Partial<Record<"title" | "upload_date", { from: unknown; to: unknown }>>; +}; + +export type WaybackRefreshResult = { + records: WaybackRecordResult[]; + // `titles` entries no Wayback record of the channel matched. + unmatchedTitles: string[]; + failed: { id: string; error: string }[]; +}; + +type Info = Record<string, unknown>; + +async function readInfo(videoDir: string): Promise<Info | null> { + const read = await readJsonFile(path.join(videoDir, "metadata.info.json")); + return read.ok && read.value && typeof read.value === "object" && !Array.isArray(read.value) + ? (read.value as Info) + : null; +} + +// The URL the managed download ran yt-dlp on: the last argument of the +// command line it logged (`$ yt-dlp … -- <url>`). +async function urlFromDownloadLog(videoDir: string): Promise<string | null> { + let head: string; + try { + head = (await readFile(path.join(videoDir, "download.log"), "utf8")).slice(0, 16 * 1024); + } catch { + return null; + } + for (const line of head.split("\n")) { + const m = /^\$ yt-dlp .* -- (\S+)\s*$/.exec(line); + if (m && parseWaybackUrl(m[1])) return m[1]; + } + return null; +} + +// The capture a record was fetched from: its webpage_url, its original_url, +// or the URL its download log ran on. +async function captureUrlOf(videoDir: string, info: Info | null): Promise<string | null> { + for (const key of ["webpage_url", "original_url"]) { + const v = info?.[key]; + if (typeof v === "string" && parseWaybackUrl(v)) return v; + } + return urlFromDownloadLog(videoDir); +} + +// The live jobs on a channel: the record ids they name, and whether one of +// them covers the whole channel. +async function liveJobs( + paths: Paths, + slug: string, +): Promise<{ ids: Set<string>; channelWide: string[] }> { + const ids = new Set<string>(); + const channelWide: string[] = []; + const names = paths.jobsDir ? await readdir(paths.jobsDir).catch(() => [] as string[]) : []; + for (const name of names) { + if (!name.endsWith(".meta.json")) continue; + const read = await readJsonFile(path.join(paths.jobsDir, name)); + if (!read.ok || !read.value || typeof read.value !== "object") continue; + const meta = read.value as JobMeta; + if (meta.channelSlug !== slug) continue; + if (meta.status !== "queued" && meta.status !== "running") continue; + if (writerIsGone(meta)) continue; + if (meta.videoId) ids.add(meta.videoId); + else channelWide.push(`${meta.kind} ${meta.id}`); + } + return { ids, channelWide }; +} + +// A title that is not one: absent, or the file's own name (what yt-dlp's +// generic extractor titles a raw file with). +function hasRealTitle(info: Info, names: string[]): boolean { + const t = typeof info.title === "string" ? info.title.trim() : ""; + if (!t) return false; + const own = new Set<string>(names); + for (const key of ["id", "display_id", "webpage_url_basename"]) { + const v = info[key]; + if (typeof v === "string") { + own.add(v); + own.add(v.replace(/\.[A-Za-z0-9]{2,4}$/, "")); + } + } + return !own.has(t); +} + +// `YYYYMMDD` from `YYYYMMDD` or `YYYY-MM-DD`; null otherwise. +export function normalizeUploadDate(v: unknown): string | null { + if (typeof v !== "string") return null; + const s = v.trim(); + if (/^\d{8}$/.test(s)) return s; + const m = /^(\d{4})-(\d{2})-(\d{2})$/.exec(s); + return m ? `${m[1]}${m[2]}${m[3]}` : null; +} + +export async function refreshWaybackRecords(opts: { + slug: string; + paths?: Paths; + dryRun?: boolean; + // Record id (its new id, or its old dir name) → its title and date. + titles?: Record<string, WaybackTitle>; + onLog?: (line: string) => void; + now?: () => Date; +}): Promise<WaybackRefreshResult> { + const paths = opts.paths ?? getPaths(); + const log = opts.onLog ?? (() => {}); + const dryRun = opts.dryRun === true; + const config = await readChannelConfig(paths, opts.slug); + if (!config) throw new Error(`Channel "${opts.slug}" not found`); + // An unreadable text tier is not an empty channel (AGENTS.md). + await assertChannelTextReadable(paths, opts.slug, config); + if (await readRelocationMarker(paths, opts.slug)) { + throw new Error(`Channel "${opts.slug}" is relocating (.relocating.json) — run again when the move is done`); + } + + const channelDir = path.join(paths.channelsDir, opts.slug); + const dataDir = path.join(channelDir, "data"); + const result: WaybackRefreshResult = { records: [], unmatchedTitles: [], failed: [] }; + const jobs = await liveJobs(paths, opts.slug); + + // ─── Find the Wayback records ─── + type Found = { rec: WaybackRecordResult; info: Info | null; canonical: string | null }; + const found: Found[] = []; + for (const name of (await listChannelVideoIds(paths, opts.slug)).sort()) { + const videoDir = path.join(dataDir, name); + try { + const info = await readInfo(videoDir); + const captureUrl = await captureUrlOf(videoDir, info); + if (!captureUrl) continue; + // The snapshot names a dir by its webpage_url (reconcileVideoDirs.ts), + // so that is the only name a rename can give it that lasts. + const webpageUrl = typeof info?.webpage_url === "string" ? info.webpage_url : null; + const canonical = webpageUrl && parseWaybackUrl(webpageUrl) ? extractVideoId(webpageUrl) : null; + const rec: WaybackRecordResult = { from: name, to: name, captureUrl, sidecar: false, changes: {} }; + if (canonical && canonical !== name) rec.to = canonical; + else if (!canonical && webpageUrl !== captureUrl) { + rec.note = "webpage_url is not the capture; the dir keeps its name"; + } + found.push({ rec, info, canonical }); + } catch (err) { + result.failed.push({ id: name, error: (err as Error).message }); + } + } + + const held = (rec: WaybackRecordResult): string | null => { + if (jobs.channelWide.length > 0) return `a live job holds the channel (${jobs.channelWide.join(", ")})`; + if (jobs.ids.has(rec.from) || jobs.ids.has(rec.to)) return "a live job names this record"; + return null; + }; + + // ─── Rename, through the snapshot's own pass ─── + const toRename = new Set<string>(); + for (const { rec } of found) { + if (rec.to === rec.from) continue; + const why = held(rec); + if (why) { + rec.rename = "held"; + rec.note = why; + rec.to = rec.from; + continue; + } + toRename.add(rec.from); + } + if (toRename.size > 0) { + const r = await reconcileVideoDirs({ channelDir, dryRun, only: (name) => toRename.has(name) }); + const byFrom = new Map(found.map((f) => [f.rec.from, f.rec])); + for (const x of r.renamed) { + const rec = byFrom.get(x.from); + if (rec) rec.rename = "renamed"; + } + for (const x of r.merged) { + const rec = byFrom.get(x.from); + if (rec) { + rec.rename = "merged"; + rec.note = `merged into the existing ${x.to}/ (${x.movedFiles.length} files)`; + } + } + for (const x of r.conflicts) { + const rec = byFrom.get(x.from); + if (rec) { + rec.rename = "conflict"; + rec.note = x.reason; + rec.to = rec.from; + } + } + const moved = [...r.renamed, ...r.merged].map((x) => ({ from: x.from, to: x.to })); + if (!dryRun && moved.length > 0) { + const now = (opts.now?.() ?? new Date()).toISOString(); + const before = await loadRoster(paths, opts.slug); + const after = renameRosterEntries(before, moved, now); + if (after !== before) await writeRoster(paths, opts.slug, after); + } + } + + // ─── The sidecar, and the titles ─── + const titles = opts.titles ?? {}; + const usedTitles = new Set<string>(); + for (const { rec, info } of found) { + // A dry run renamed nothing: the record is still under its old name. + const videoDir = path.join(dataDir, dryRun ? rec.from : rec.to); + try { + const s = await ensureWaybackProvenance(videoDir, rec.captureUrl, { dryRun }); + rec.sidecar = s.written; + const key = rec.to in titles ? rec.to : rec.from in titles ? rec.from : null; + if (key === null || !info) continue; + usedTitles.add(key); + if (held(rec)) { + rec.note = rec.note ?? held(rec)!; + continue; + } + const want = titles[key]; + const patch: Record<string, unknown> = {}; + const title = typeof want.title === "string" ? want.title.trim() : ""; + if (title && !hasRealTitle(info, [rec.from, rec.to]) && info.title !== title) { + patch.title = title; + rec.changes.title = { from: info.title ?? null, to: title }; + } + const date = normalizeUploadDate(want.upload_date); + if (want.upload_date !== undefined && !date) { + rec.note = `upload_date "${String(want.upload_date)}" is not YYYYMMDD or YYYY-MM-DD`; + } else if (date && normalizeUploadDate(info.upload_date) === null) { + patch.upload_date = date; + rec.changes.upload_date = { from: info.upload_date ?? null, to: date }; + } + if (Object.keys(patch).length > 0 && !dryRun) { + await patchMetadataInfo(videoDir, patch, { by: "wayback-provenance", requestedBy: "cli", onLog: log }); + } + } catch (err) { + result.failed.push({ id: rec.from, error: (err as Error).message }); + } + } + result.unmatchedTitles = Object.keys(titles).filter((k) => !usedTitles.has(k)).sort(); + result.records = found.map((f) => f.rec); + for (const rec of result.records) log(formatWaybackRecord(rec, dryRun)); + return result; +} + +// One record as the CLI prints it: `old → new`, then what was written. +export function formatWaybackRecord(rec: WaybackRecordResult, dryRun: boolean): string { + const would = dryRun ? "would be " : ""; + const head = + rec.from === rec.to + ? `${rec.from} (name kept)` + : `${rec.from} → ${rec.to}${rec.rename === "merged" ? " (merged)" : ""}`; + const lines = [head]; + if (rec.note) lines.push(` ${rec.rename === "held" || rec.rename === "conflict" ? "NOT RENAMED: " : ""}${rec.note}`); + if (rec.sidecar) lines.push(` wayback.json ${would}written`); + for (const [k, c] of Object.entries(rec.changes)) { + lines.push(` ${k}: ${JSON.stringify(c!.from)} → ${JSON.stringify(c!.to)}`); + } + return lines.join("\n"); +} diff --git a/common/lib/metadataHistory.ts b/common/lib/metadataHistory.ts @@ -68,6 +68,10 @@ export const METADATA_HISTORY_WRITERS = [ // metadata API (controller/archiveOrgDownload.ts) — archive.org records are // not fetched by yt-dlp. "archiveorg-import", + // A Wayback Machine copy's title and date, set from what the operator found + // for it (`archilyzer wayback refresh --titles`, controller/waybackRefresh.ts) + // — a raw media file captured by the Wayback Machine carries no title. + "wayback-provenance", ] as const; export type MetadataHistoryWriter = (typeof METADATA_HISTORY_WRITERS)[number];