import { test } from "node:test"; import assert from "node:assert/strict"; import { mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import type { Paths } from "../lib/paths"; import { buildArchiveOrgProvenance, parseArchiveOrgItemMetadata } from "../lib/archiveOrg"; import { archiveOrgVideoId } from "../lib/archiveOrgId"; import { loadMetadataHistory } from "../lib/metadataHistory-server"; import { refreshArchiveOrgRecords } from "./archiveOrgRefresh"; import { main as refreshCli } from "../bin/archive-org-refresh"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test controller/archiveOrgRefresh.test.ts // // A temp corpus of archive.org records written before file-name titles: the // raw name as the title, the item's date, and no mirror title or date. Every name here is invented. const SLUG = "demo-archive"; const ITEM = "example-channel-archive"; const RAW = "extras_20210102 A Talk _ Some Show [4321 views]-Pqr678stu_4.mp4"; const TITLED = "Second Upload-Def456uvw_8.mp4"; const MIRRORED = "Third Upload-Ghi789rst_7.mp4"; const item = parseArchiveOrgItemMetadata({ metadata: { identifier: ITEM, title: "Example Channel Archive", date: "2024-01-02" }, files: [ { name: RAW, source: "original" }, { name: TITLED, source: "original", title: "Second, as titled on archive.org" }, { name: MIRRORED, source: "original" }, ], })!; const stem = (f: string) => f.replace(/\.mp4$/, ""); // The sidecar as an older import wrote it: no title or date for a name-only mirror. function oldProvenance(file: string, infoJson?: unknown) { const prov = buildArchiveOrgProvenance({ ref: { identifier: ITEM, file }, item, infoJson, fetchedAt: "2026-01-01T00:00:00.000Z", }); if (prov.mirror?.from === "file-name") { delete prov.mirror.title; delete prov.mirror.uploadDate; } return prov; } async function withCorpus(fn: (paths: Paths, dataDir: string) => Promise): Promise { const dir = await mkdtemp(path.join(tmpdir(), "ttb-ia-titles-")); const transcriptsDir = path.join(dir, "corpus"); const paths = { transcriptsDir, channelsDir: path.join(transcriptsDir, "channels") } as Paths; const channelDir = path.join(paths.channelsDir, SLUG); const dataDir = path.join(channelDir, "data"); await mkdir(dataDir, { recursive: true }); await writeFile( path.join(channelDir, "config.json"), JSON.stringify({ name: "Demo", platform: "archiveorg", handling: "transcribe" }), ); const seed: [string, unknown, Record][] = [ [archiveOrgVideoId({ identifier: ITEM, file: RAW }), oldProvenance(RAW), { title: stem(RAW), timestamp: 1704153600 }], [archiveOrgVideoId({ identifier: ITEM, file: TITLED }), oldProvenance(TITLED), { title: "Second, as titled on archive.org" }], [ archiveOrgVideoId({ identifier: ITEM, file: MIRRORED }), oldProvenance(MIRRORED, { id: "Ghi789rst_7", extractor_key: "Youtube", title: "The Original Title" }), { title: "The Original Title" }, ], // A whole item: not one file of many, never re-titled. [ "example-single", buildArchiveOrgProvenance({ ref: { identifier: "example-single" }, item: { ...item, metadata: { identifier: "example-single", title: "Single" } }, fetchedAt: "2026-01-01T00:00:00.000Z", }), { title: "Single" }, ], ]; for (const [id, prov, info] of seed) { await mkdir(path.join(dataDir, id), { recursive: true }); await writeFile(path.join(dataDir, id, "archiveorg.json"), JSON.stringify(prov)); await writeFile( path.join(dataDir, id, "metadata.info.json"), JSON.stringify({ id, extractor_key: "ArchiveOrg", ...info, upload_date: "20240102" }), ); } try { await fn(paths, dataDir); } finally { await rm(dir, { recursive: true, force: true }); } } const RAW_ID = archiveOrgVideoId({ identifier: ITEM, file: RAW }); test("refresh: a dry run lists old → new and writes nothing", async () => { await withCorpus(async (paths, dataDir) => { const before = await readFile(path.join(dataDir, RAW_ID, "metadata.info.json"), "utf8"); const beforeProv = await readFile(path.join(dataDir, RAW_ID, "archiveorg.json"), "utf8"); const r = await refreshArchiveOrgRecords({ paths, slug: SLUG, dryRun: true }); assert.equal(r.fileRecords, 3); assert.deepEqual(r.changed, [ { id: RAW_ID, changes: { title: { from: stem(RAW), to: "A Talk | Some Show" }, upload_date: { from: "20240102", to: "20210102" }, timestamp: { from: 1704153600, to: null }, }, }, ]); assert.equal(r.sidecars, 1); assert.equal(await readFile(path.join(dataDir, RAW_ID, "metadata.info.json"), "utf8"), before); assert.equal(await readFile(path.join(dataDir, RAW_ID, "archiveorg.json"), "utf8"), beforeProv); assert.equal(await loadMetadataHistory(path.join(dataDir, RAW_ID)), null); // The CLI over the same corpus: prints the change, exits 0, writes nothing. const lines: string[] = []; const orig = console.log; console.log = (...a: unknown[]) => void lines.push(a.join(" ")); try { assert.equal(await refreshCli({ slug: SLUG, dryRun: true, paths }), 0); } finally { console.log = orig; } const out = lines.join("\n"); assert.match(out, /→ "A Talk \| Some Show"/); assert.match(out, /upload_date: "20240102"\n {4}→ "20210102"/); assert.match(out, /dry run: 3 archive\.org file records, 1 would be refreshed/); assert.equal(await readFile(path.join(dataDir, RAW_ID, "metadata.info.json"), "utf8"), before); }); }); test("refresh: title and date go through the metadata history, the raw name stays as `file`; idempotent", async () => { await withCorpus(async (paths, dataDir) => { const r = await refreshArchiveOrgRecords({ paths, slug: SLUG }); assert.equal(r.changed.length, 1); assert.deepEqual(r.failed, []); const dir = path.join(dataDir, RAW_ID); const info = JSON.parse(await readFile(path.join(dir, "metadata.info.json"), "utf8")); assert.equal(info.title, "A Talk | Some Show"); assert.equal(info.upload_date, "20210102"); assert.equal(info.timestamp ?? null, null); assert.equal(Object.keys(info).at(-1), "upload_date"); const prov = JSON.parse(await readFile(path.join(dir, "archiveorg.json"), "utf8")); assert.equal(prov.file, RAW); assert.equal(prov.mirror.title, "A Talk | Some Show"); assert.equal(prov.mirror.uploadDate, "20210102"); const entry = (await loadMetadataHistory(dir))?.entries.at(-1); assert.equal(entry?.by, "archiveorg-provenance"); assert.deepEqual(entry?.changed.title, { from: stem(RAW), to: "A Talk | Some Show" }); assert.deepEqual(entry?.changed.upload_date, { from: "20240102", to: "20210102" }); const again = await refreshArchiveOrgRecords({ paths, slug: SLUG }); assert.deepEqual(again.changed, []); assert.equal(again.sidecars, 0); assert.equal((await loadMetadataHistory(dir))?.entries.length, 1); }); }); test("refresh: an unknown channel is refused", async () => { await withCorpus(async (paths) => { await assert.rejects(refreshArchiveOrgRecords({ paths, slug: "no-such-channel" }), /not found/); }); });