import { test } from "node:test"; import assert from "node:assert/strict"; import { mkdir, mkdtemp, rm, unlink, utimes, writeFile, readdir } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import type { Paths } from "../lib/paths"; import type { TranscriptSummary } from "../lib/transcripts"; import { upsertMetadataScan } from "./metadataScanStore"; import { readChannelVideoTitles, readVideoMetadataForDisplay, resetVideoTitleMemo, videoTitleMetadataReadCount, } from "./videoTitles"; // Run with: pnpm -C common exec tsx --test "controller/videoTitles.test.ts" // // A REAL LMDB file written the way buildIndex writes it (compression on, one // fat description) — the reader must open it with compression too, and a fake // would not catch that (curatedTagsPreview.test.ts's lesson). const SLUG = "ch"; async function fixture(prefix: string): Promise<{ dir: string; paths: Paths }> { const dir = await mkdtemp(path.join(tmpdir(), prefix)); const channelsDir = path.join(dir, "channels"); await mkdir(path.join(channelsDir, SLUG, "data"), { recursive: true }); return { dir, paths: { lmdbPath: path.join(dir, "index.mdb"), channelsDir } as Paths, }; } function summary(id: string, title: string, uploadDate: string): TranscriptSummary { return { slug: `${SLUG}/${id}`, id, channelSlug: SLUG, title, uploadDate, duration: 60, channel: "Ch", // Over lmdb-js's ~1 KB compression threshold. description: "d".repeat(4000), tags: [], isLivestream: false, ageRestricted: false, platform: "youtube", webpageUrl: `https://www.youtube.com/watch?v=${id}`, } as TranscriptSummary; } async function writeIndex( lmdbPath: string, rows: ReadonlyArray<{ slug: string; summary: TranscriptSummary }>, ): Promise { const { open } = await import("lmdb"); const root = open({ path: lmdbPath, maxDbs: 18, compression: true }); const sums = root.openDB({ name: "sums", encoding: "msgpack", }); const byChannel = root.openDB({ name: "byChannel", encoding: "msgpack", }); for (const { slug, summary: s } of rows) { await sums.put([s.uploadDate, slug, s.id], s); await byChannel.put([slug, s.uploadDate, s.id], 1); } await sums.flushed; await byChannel.flushed; await root.close(); } async function writeInfo(paths: Paths, id: string, info: object): Promise { const dir = path.join(paths.channelsDir, SLUG, "data", id); await mkdir(dir, { recursive: true }); await writeFile(path.join(dir, "metadata.info.json"), JSON.stringify(info)); } async function scan( paths: Paths, entries: Record, ): Promise { const full: Parameters[2]["entries"] = {}; for (const [id, e] of Object.entries(entries)) { full[id] = { title: e.title, description: e.description ?? "", uploadDate: "20260101", duration: e.duration, scannedAt: "2026-09-25T00:00:00.000Z", }; } await upsertMetadataScan(paths, SLUG, { entries: full }, "2026-09-25T00:00:00.000Z"); } test("merges index, scan and metadata in that order; first hit wins; bare ids are absent", async () => { const { dir, paths } = await fixture("vtitles-merge-"); try { await writeIndex(paths.lmdbPath, [ { slug: SLUG, summary: summary("idx", "From the index", "20260101") }, { slug: SLUG, summary: summary("both", "Index beats scan", "20260102") }, // Another channel's id must not leak into this one. { slug: "other", summary: summary("foreign", "Other channel", "20260103") }, ]); await scan(paths, { both: { title: "Scan loses" }, scanned: { title: "From the scan" }, }); await writeInfo(paths, "disk", { id: "disk", title: "From metadata.info.json" }); await writeInfo(paths, "idx", { id: "idx", title: "Metadata loses" }); const got = await readChannelVideoTitles(paths, SLUG, [ "idx", "both", "scanned", "disk", "bare", "foreign", ]); assert.deepEqual(Object.fromEntries(got), { idx: { title: "From the index", source: "index" }, both: { title: "Index beats scan", source: "index" }, scanned: { title: "From the scan", source: "scan" }, disk: { title: "From metadata.info.json", source: "metadata" }, }); } finally { await rm(dir, { recursive: true, force: true }); } }); test("a missing index and a missing scan store are not errors", async () => { const { dir, paths } = await fixture("vtitles-missing-"); try { await writeInfo(paths, "a", { id: "a", title: "Only on disk" }); const got = await readChannelVideoTitles(paths, SLUG, ["a", "b"]); assert.deepEqual(Object.fromEntries(got), { a: { title: "Only on disk", source: "metadata" }, }); assert.equal((await readChannelVideoTitles(paths, SLUG, [])).size, 0); } finally { await rm(dir, { recursive: true, force: true }); } }); test("an index title equal to the id (summarize's fallback) is not a title", async () => { const { dir, paths } = await fixture("vtitles-fallback-"); try { await writeIndex(paths.lmdbPath, [ { slug: SLUG, summary: summary("noname", "noname", "20260101") }, ]); await scan(paths, { noname: { title: "Named by the scan" } }); const got = await readChannelVideoTitles(paths, SLUG, ["noname"]); assert.deepEqual(got.get("noname"), { title: "Named by the scan", source: "scan", }); } finally { await rm(dir, { recursive: true, force: true }); } }); test("metadata head read: title past the head and escaped titles still resolve", async () => { const { dir, paths } = await fixture("vtitles-head-"); try { // A title that JSON must unescape. await writeInfo(paths, "esc", { id: "esc", title: 'He said "hi" \\ bye ✓' }); // A title beyond the 16 KB head: the full-parse fallback finds it. await writeInfo(paths, "late", { id: "late", pad: "x".repeat(40_000), title: "Late title" }); // The scan never creates a video dir, so reading must not either. const got = await readChannelVideoTitles(paths, SLUG, ["esc", "late", "ghost"]); assert.equal(got.get("esc")?.title, 'He said "hi" \\ bye ✓'); assert.equal(got.get("late")?.title, "Late title"); assert.equal(got.has("ghost"), false); const dirs = await readdir(path.join(paths.channelsDir, SLUG, "data")); assert.deepEqual(dirs.sort(), ["esc", "late"]); } finally { await rm(dir, { recursive: true, force: true }); } }); test("readVideoMetadataForDisplay: metadata.info.json, else the scan entry, else none", async () => { const { dir, paths } = await fixture("vtitles-display-"); try { await writeInfo(paths, "dl", { id: "dl", title: "Downloaded", description: "Full description", webpage_url: "https://www.youtube.com/watch?v=dl", uploader: "Uploader", upload_date: "20250102", duration: 125, }); await scan(paths, { dl: { title: "Scan must lose" }, un: { title: "Listed only", description: "Scan description", duration: 61 }, }); assert.deepEqual(await readVideoMetadataForDisplay(paths, SLUG, "dl"), { title: "Downloaded", description: "Full description", webpageUrl: "https://www.youtube.com/watch?v=dl", uploader: "Uploader", uploadDate: "20250102", duration: 125, source: "metadata", }); assert.deepEqual(await readVideoMetadataForDisplay(paths, SLUG, "un"), { title: "Listed only", description: "Scan description", uploadDate: "20260101", duration: 61, source: "scan", }); assert.deepEqual(await readVideoMetadataForDisplay(paths, SLUG, "nope"), { source: "none", }); } finally { await rm(dir, { recursive: true, force: true }); } }); // The cost bar: a 5,000-id channel, all three sources exercised. Recorded in // plans/release-8.md; asserted only loosely so a slow CI box does not flake. test("cost: 5,000 ids across the three sources", async () => { const { dir, paths } = await fixture("vtitles-cost-"); try { const ids = Array.from({ length: 5000 }, (_, i) => `v${String(i).padStart(5, "0")}`); // 3,000 in the index, 1,500 in the scan, 400 on disk only, 100 bare. await writeIndex( paths.lmdbPath, ids.slice(0, 3000).map((id, i) => ({ slug: SLUG, summary: summary(id, `Indexed ${id}`, `2025${String((i % 12) + 1).padStart(2, "0")}01`), })), ); await scan( paths, Object.fromEntries( ids.slice(3000, 4500).map((id) => [id, { title: `Scanned ${id}`, description: "x".repeat(500) }]), ), ); for (const id of ids.slice(4500, 4900)) { await writeInfo(paths, id, { id, title: `Disk ${id}`, formats: "f".repeat(50_000) }); } const t0 = performance.now(); const got = await readChannelVideoTitles(paths, SLUG, ids); const ms = performance.now() - t0; assert.equal(got.size, 4900); const bySource = { index: 0, scan: 0, metadata: 0 }; for (const v of got.values()) bySource[v.source]++; assert.deepEqual(bySource, { index: 3000, scan: 1500, metadata: 400 }); const t1 = performance.now(); await readChannelVideoTitles(paths, SLUG, ids); const memoMs = performance.now() - t1; console.log( `readChannelVideoTitles 5,000 ids: ${ms.toFixed(1)} ms first, ${memoMs.toFixed(1)} ms memoized`, ); assert.ok(ms < 5000, `took ${ms} ms`); } finally { await rm(dir, { recursive: true, force: true }); } }); test("metadata titles are memoized per channel until data/ changes", async () => { const { dir, paths } = await fixture("vtitles-memo-"); try { resetVideoTitleMemo(); const dataDir = path.join(paths.channelsDir, SLUG, "data"); await writeInfo(paths, "a", { id: "a", title: "Alpha" }); await writeInfo(paths, "b", { id: "b", title: "Bravo" }); // Pin the dir's mtime so the second call's key is certainly unchanged. const pinned = new Date("2026-01-01T00:00:00Z"); await utimes(dataDir, pinned, pinned); const r0 = videoTitleMetadataReadCount(); const first = await readChannelVideoTitles(paths, SLUG, ["a", "b"]); assert.equal(first.get("a")?.title, "Alpha"); assert.equal(videoTitleMetadataReadCount() - r0, 2); // Same data/ mtime: served from the memo, nothing read — even with the // file gone (proof that it was not read). await unlink(path.join(dataDir, "a", "metadata.info.json")); await utimes(dataDir, pinned, pinned); const r1 = videoTitleMetadataReadCount(); const second = await readChannelVideoTitles(paths, SLUG, ["a", "b"]); assert.equal(second.get("a")?.title, "Alpha"); assert.equal(second.get("b")?.title, "Bravo"); assert.equal(videoTitleMetadataReadCount() - r1, 0); // A new video dir changes data/'s mtime: the memo is REBASED — the dirs // still there keep their titles and only the new one is read (release 8 // review, V). "a"'s dir is still there, so its memoized title stands. await writeInfo(paths, "c", { id: "c", title: "Charlie" }); const r2 = videoTitleMetadataReadCount(); const third = await readChannelVideoTitles(paths, SLUG, ["a", "b", "c"]); assert.equal(third.get("a")?.title, "Alpha"); assert.equal(third.get("b")?.title, "Bravo"); assert.equal(third.get("c")?.title, "Charlie"); assert.equal(videoTitleMetadataReadCount() - r2, 1); // A REMOVED dir takes its title with it: the rebase keeps only the dirs // still present, so "a" falls through, is read, and has no title. await rm(path.join(dataDir, "a"), { recursive: true, force: true }); // Two changes inside one filesystem timestamp tick would share an mtime; // pin a distinct one so the rebase certainly fires. const later = new Date("2026-01-02T00:00:00Z"); await utimes(dataDir, later, later); const r3 = videoTitleMetadataReadCount(); const fourth = await readChannelVideoTitles(paths, SLUG, ["a", "b", "c"]); assert.equal(fourth.has("a"), false); assert.equal(fourth.get("c")?.title, "Charlie"); assert.equal(videoTitleMetadataReadCount() - r3, 1); // A miss is not memoized: a title that lands later is found. await mkdir(path.join(dataDir, "a"), { recursive: true }); await writeFile( path.join(dataDir, "a", "metadata.info.json"), JSON.stringify({ id: "a", title: "Alpha again" }), ); const fifth = await readChannelVideoTitles(paths, SLUG, ["a"]); assert.equal(fifth.get("a")?.title, "Alpha again"); } finally { resetVideoTitleMemo(); await rm(dir, { recursive: true, force: true }); } });