// WAYBACK MACHINE COPIES, BROUGHT UP TO THE WAYBACK RULES — offline. // // A record imported from a Wayback capture (lib/wayback.ts) before the app // knew what one was carries no `wayback.json`, and its dir is named by the last // segment of the capture URL: `watch` for an archived YouTube page (a name // every such capture shares), `-.mp4` for a raw JW Player // file. This brings every such record of a channel up to the rules: // // wayback.json written from the capture URL (lib/wayback-server.ts) // data// renamed to its canonical id (lib/videoId.ts) by the // snapshot's own pass, reconcileVideoDirs — the media // tier's relative links move with the dir, and a dir // already there is merged, never overwritten // roster.json the entry moved to the new id (renameRosterEntries) // metadata.info.json with `titles`: the title and upload_date the operator // found for a record that has none (a raw file's title // is its file name), through `patchMetadataInfo`, so the // change is in metadata.history.json as // `wayback-provenance` // // A record a live job names (a `.jobs/` meta, queued or running, whose writer // is alive, or one of an auto-queue lane's newest picks) is skipped and // reported; a live job over the whole channel holds every record. Only what differs is written, so a second run writes nothing. // No network. import path from "node:path"; import { readdir, readFile, stat } from "node:fs/promises"; import { getPaths, type Paths } from "../lib/paths"; import { readJsonFile } from "../lib/jsonFile-server"; import { assertChannelTextReadable, readRelocationMarker } from "../lib/channelMedia"; import { parseWaybackUrl } from "../lib/wayback"; import { ensureWaybackProvenance } from "../lib/wayback-server"; import { extractVideoId } from "../lib/videoId"; import { patchMetadataInfo } from "../lib/metadataHistory-server"; import { readChannelConfig } from "./channels"; import { listChannelVideoIds } from "./keptVideos"; import { reconcileVideoDirs } from "./reconcileVideoDirs"; import { loadRoster, renameRosterEntries, writeRoster } from "./rosterStore"; import { writerIsGone } from "../jobs/bootQueuedJobs"; import type { JobMeta } from "../jobs/jobMeta"; import type { AutoQueuePick } from "../jobs/autoQueueState"; // What the operator found for a record: its title and the day it is of. export type WaybackTitle = { title?: string; upload_date?: string }; export type WaybackRecordResult = { // The dir's name before, and after (the same when it was not renamed). from: string; to: string; // The capture the record was fetched from. captureUrl: string; rename?: "renamed" | "merged" | "conflict" | "held"; // Why a rename did not happen (a live job, a conflict, a capture that is // not the record's webpage_url). note?: string; // wayback.json was (or would be) written. sidecar: boolean; // Per key: the value before and after. changes: Partial>; }; export type WaybackRefreshResult = { records: WaybackRecordResult[]; // `titles` entries no Wayback record of the channel matched. unmatchedTitles: string[]; failed: { id: string; error: string }[]; }; type Info = Record; async function readInfo(videoDir: string): Promise { const read = await readJsonFile(path.join(videoDir, "metadata.info.json")); return read.ok && read.value && typeof read.value === "object" && !Array.isArray(read.value) ? (read.value as Info) : null; } // The URL the managed download ran yt-dlp on: the last argument of the // command line it logged (`$ yt-dlp … -- `). async function urlFromDownloadLog(videoDir: string): Promise { let head: string; try { head = (await readFile(path.join(videoDir, "download.log"), "utf8")).slice(0, 16 * 1024); } catch { return null; } for (const line of head.split("\n")) { const m = /^\$ yt-dlp .* -- (\S+)\s*$/.exec(line); if (m && parseWaybackUrl(m[1])) return m[1]; } return null; } // The capture a record was fetched from: its webpage_url, its original_url, // or the URL its download log ran on. async function captureUrlOf(videoDir: string, info: Info | null): Promise { for (const key of ["webpage_url", "original_url"]) { const v = info?.[key]; if (typeof v === "string" && parseWaybackUrl(v)) return v; } return urlFromDownloadLog(videoDir); } // The live jobs on a channel: the record ids they name, and whether one of // them covers the whole channel. async function liveJobs( paths: Paths, slug: string, now: number, ): Promise<{ ids: Set; channelWide: string[] }> { const ids = new Set(); const channelWide: string[] = []; const names = paths.jobsDir ? await readdir(paths.jobsDir).catch(() => [] as string[]) : []; for (const name of names) { if (!name.endsWith(".meta.json")) continue; const read = await readJsonFile(path.join(paths.jobsDir, name)); if (!read.ok || !read.value || typeof read.value !== "object") continue; const meta = read.value as JobMeta; if (meta.channelSlug !== slug) continue; if (meta.status !== "queued" && meta.status !== "running") continue; if (writerIsGone(meta)) continue; if (meta.videoId) ids.add(meta.videoId); else channelWide.push(`${meta.kind} ${meta.id}`); } // The auto-queue lanes are jobs of no channel, and the record each is on // is its newest pick (jobs/autoQueueState.ts). A lane runs a few workers, so // the newest few recent picks of this channel are held. if (paths.autoQueueStateFile) { const read = await readJsonFile(paths.autoQueueStateFile); const state = read.ok && read.value && typeof read.value === "object" ? (read.value as Record) : {}; const since = now - LANE_PICK_HOLD_MS; for (const [kind, kindState] of Object.entries(state)) { const picks = (kindState as { picks?: unknown } | null)?.picks; if (!Array.isArray(picks)) continue; for (const pick of picks.slice(0, LANE_PICKS_HELD) as Partial[]) { if (pick?.channelSlug !== slug || typeof pick.videoId !== "string") continue; const at = pick.at ?? 0; if (at < since) continue; // A pick whose outcome sidecar was written since is finished. const outcome = LANE_OUTCOME[kind]; if (outcome) { const done = await stat(path.join(paths.channelsDir, slug, "data", pick.videoId, outcome)).catch(() => null); if (done && done.mtimeMs >= at) continue; } ids.add(pick.videoId); } } } return { ids, channelWide }; } // How many of a lane's newest picks may still be in flight, and for how long. const LANE_PICKS_HELD = 4; const LANE_PICK_HOLD_MS = 6 * 60 * 60 * 1000; // The sidecar a lane's run writes when it ends, by lane kind. const LANE_OUTCOME: Record = { transcription: "transcribe-outcome.json", download: "download-outcome.json", }; // A title that is not one: absent, or the file's own name (what yt-dlp's // generic extractor titles a raw file with). function hasRealTitle(info: Info, names: string[]): boolean { const t = typeof info.title === "string" ? info.title.trim() : ""; if (!t) return false; const own = new Set(names); for (const key of ["id", "display_id", "webpage_url_basename"]) { const v = info[key]; if (typeof v === "string") { own.add(v); own.add(v.replace(/\.[A-Za-z0-9]{2,4}$/, "")); } } return !own.has(t); } // `YYYYMMDD` from `YYYYMMDD` or `YYYY-MM-DD`; null otherwise. export function normalizeUploadDate(v: unknown): string | null { if (typeof v !== "string") return null; const s = v.trim(); if (/^\d{8}$/.test(s)) return s; const m = /^(\d{4})-(\d{2})-(\d{2})$/.exec(s); return m ? `${m[1]}${m[2]}${m[3]}` : null; } export async function refreshWaybackRecords(opts: { slug: string; paths?: Paths; dryRun?: boolean; // Record id (its new id, or its old dir name) → its title and date. titles?: Record; onLog?: (line: string) => void; now?: () => Date; }): Promise { const paths = opts.paths ?? getPaths(); const log = opts.onLog ?? (() => {}); const dryRun = opts.dryRun === true; const config = await readChannelConfig(paths, opts.slug); if (!config) throw new Error(`Channel "${opts.slug}" not found`); // An unreadable text tier is not an empty channel (AGENTS.md). await assertChannelTextReadable(paths, opts.slug, config); if (await readRelocationMarker(paths, opts.slug)) { throw new Error(`Channel "${opts.slug}" is relocating (.relocating.json) — run again when the move is done`); } const channelDir = path.join(paths.channelsDir, opts.slug); const dataDir = path.join(channelDir, "data"); const result: WaybackRefreshResult = { records: [], unmatchedTitles: [], failed: [] }; const jobs = await liveJobs(paths, opts.slug, (opts.now?.() ?? new Date()).getTime()); // ─── Find the Wayback records ─── type Found = { rec: WaybackRecordResult; info: Info | null; canonical: string | null }; const found: Found[] = []; for (const name of (await listChannelVideoIds(paths, opts.slug)).sort()) { const videoDir = path.join(dataDir, name); try { const info = await readInfo(videoDir); const captureUrl = await captureUrlOf(videoDir, info); if (!captureUrl) continue; // The snapshot names a dir by its webpage_url (reconcileVideoDirs.ts), // so that is the only name a rename can give it that lasts. const webpageUrl = typeof info?.webpage_url === "string" ? info.webpage_url : null; const canonical = webpageUrl && parseWaybackUrl(webpageUrl) ? extractVideoId(webpageUrl) : null; const rec: WaybackRecordResult = { from: name, to: name, captureUrl, sidecar: false, changes: {} }; if (canonical && canonical !== name) rec.to = canonical; else if (!canonical && webpageUrl !== captureUrl) { rec.note = "webpage_url is not the capture; the dir keeps its name"; } found.push({ rec, info, canonical }); } catch (err) { result.failed.push({ id: name, error: (err as Error).message }); } } const held = (rec: WaybackRecordResult): string | null => { if (jobs.channelWide.length > 0) return `a live job holds the channel (${jobs.channelWide.join(", ")})`; if (jobs.ids.has(rec.from) || jobs.ids.has(rec.to)) return "a live job names this record"; return null; }; // ─── Rename, through the snapshot's own pass ─── const toRename = new Set(); for (const { rec } of found) { if (rec.to === rec.from) continue; const why = held(rec); if (why) { rec.rename = "held"; rec.note = why; rec.to = rec.from; continue; } toRename.add(rec.from); } if (toRename.size > 0) { const r = await reconcileVideoDirs({ channelDir, dryRun, only: (name) => toRename.has(name) }); const byFrom = new Map(found.map((f) => [f.rec.from, f.rec])); for (const x of r.renamed) { const rec = byFrom.get(x.from); if (rec) rec.rename = "renamed"; } for (const x of r.merged) { const rec = byFrom.get(x.from); if (rec) { rec.rename = "merged"; rec.note = `merged into the existing ${x.to}/ (${x.movedFiles.length} files)`; } } for (const x of r.conflicts) { const rec = byFrom.get(x.from); if (rec) { rec.rename = "conflict"; rec.note = x.reason; rec.to = rec.from; } } const moved = [...r.renamed, ...r.merged].map((x) => ({ from: x.from, to: x.to })); if (!dryRun && moved.length > 0) { const now = (opts.now?.() ?? new Date()).toISOString(); const before = await loadRoster(paths, opts.slug); const after = renameRosterEntries(before, moved, now); if (after !== before) await writeRoster(paths, opts.slug, after); } } // ─── The sidecar, and the titles ─── const titles = opts.titles ?? {}; const usedTitles = new Set(); for (const { rec, info } of found) { // A dry run renamed nothing: the record is still under its old name. const videoDir = path.join(dataDir, dryRun ? rec.from : rec.to); try { const s = await ensureWaybackProvenance(videoDir, rec.captureUrl, { dryRun }); rec.sidecar = s.written; const key = rec.to in titles ? rec.to : rec.from in titles ? rec.from : null; if (key === null || !info) continue; usedTitles.add(key); if (held(rec)) { rec.note = rec.note ?? held(rec)!; continue; } const want = titles[key]; const patch: Record = {}; const title = typeof want.title === "string" ? want.title.trim() : ""; if (title && !hasRealTitle(info, [rec.from, rec.to]) && info.title !== title) { patch.title = title; rec.changes.title = { from: info.title ?? null, to: title }; } const date = normalizeUploadDate(want.upload_date); if (want.upload_date !== undefined && !date) { rec.note = `upload_date "${String(want.upload_date)}" is not YYYYMMDD or YYYY-MM-DD`; } else if (date && normalizeUploadDate(info.upload_date) === null) { patch.upload_date = date; rec.changes.upload_date = { from: info.upload_date ?? null, to: date }; } if (Object.keys(patch).length > 0 && !dryRun) { await patchMetadataInfo(videoDir, patch, { by: "wayback-provenance", requestedBy: "cli", onLog: log }); } } catch (err) { result.failed.push({ id: rec.from, error: (err as Error).message }); } } result.unmatchedTitles = Object.keys(titles).filter((k) => !usedTitles.has(k)).sort(); result.records = found.map((f) => f.rec); for (const rec of result.records) log(formatWaybackRecord(rec, dryRun)); return result; } // One record as the CLI prints it: `old → new`, then what was written. export function formatWaybackRecord(rec: WaybackRecordResult, dryRun: boolean): string { const would = dryRun ? "would be " : ""; const head = rec.from === rec.to ? `${rec.from} (name kept)` : `${rec.from} → ${rec.to}${rec.rename === "merged" ? " (merged)" : ""}`; const lines = [head]; if (rec.note) lines.push(` ${rec.rename === "held" || rec.rename === "conflict" ? "NOT RENAMED: " : ""}${rec.note}`); if (rec.sidecar) lines.push(` wayback.json ${would}written`); for (const [k, c] of Object.entries(rec.changes)) { lines.push(` ${k}: ${JSON.stringify(c!.from)} → ${JSON.stringify(c!.to)}`); } return lines.join("\n"); }