import { test } from "node:test"; import assert from "node:assert/strict"; import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import { type RecencyKey, buildRecencyKeys, clearRecencyCache, interpolateFromPlaylist, makeRecencyComparator, } from "./recencyIndex"; import type { Paths } from "../lib/paths"; // Run with: node_modules/.bin/tsx --test common/controller/recencyIndex.test.ts function run( playlist: string[], dates: Record, wanted: string[], ): Record { const out = new Map(); interpolateFromPlaylist( playlist, new Map(Object.entries(dates)), new Set(wanted), out, ); return Object.fromEntries(out); } // --- Layer 2: playlist-neighbour interpolation ------------------------------- test("interpolation: an undated id takes the nearest PRECEDING date", () => { // The playlist is newest-first, so the entry above an undownloaded video is // an upper bound on its age — the best estimate available, since an // undownloaded video has no metadata.info.json to read a real date from. const keys = run( ["a", "gap1", "gap2", "b", "gap3"], { a: "20260801", b: "20260101" }, ["gap1", "gap2", "gap3"], ); assert.deepEqual(keys.gap1, { key: "20260801", estimated: true }); assert.deepEqual(keys.gap2, { key: "20260801", estimated: true }); assert.deepEqual(keys.gap3, { key: "20260101", estimated: true }); }); test("interpolation: an id ahead of every date sorts above everything", () => { // The point of the whole feature: a video uploaded today, not yet downloaded, // sits at the head of a newest-first playlist with nothing dated above it. It // must jump the backlog rather than land next to the channel's oldest work. const keys = run(["brand-new", "a"], { a: "20260801" }, ["brand-new"]); assert.equal(keys["brand-new"].estimated, true); const cmp = makeRecencyComparator( new Map(Object.entries(keys) as [string, RecencyKey][]).set("old", { key: "20261231", estimated: false, }), "newest", )!; assert.deepEqual(["old", "brand-new"].sort(cmp), ["brand-new", "old"]); }); test("interpolation: an anchorless playlist is left alone", () => { // With no dated entry anywhere we know nothing about the channel's timeline. // Handing every id the top sentinel would let one freshly-added channel with // 9,000 undownloaded videos monopolize the head of the queue, so this bails // and lets the dir-prefix fallback (layer 3) decide instead. assert.deepEqual(run(["x", "y"], {}, ["x", "y"]), {}); }); test("interpolation: only ids the caller asked for are keyed, once each", () => { const wanted = new Set(["gap1"]); const out = new Map(); interpolateFromPlaylist( ["a", "gap1", "gap2"], new Map([["a", "20260801"]]), wanted, out, ); assert.deepEqual([...out.keys()], ["gap1"]); // The wanted set is drained as it is satisfied, so the caller can stop early. assert.equal(wanted.size, 0); }); // --- Layer 2: the metadata.info.json tail read ------------------------------- // Write a metadata.info.json whose upload_date sits near the END of the file, // behind `padBytes` of other JSON — which is how yt-dlp actually writes them, // and the reason a tail read works at all. async function writeMeta( channelsDir: string, slug: string, id: string, uploadDate: string | null, padBytes = 0, ): Promise { const dir = path.join(channelsDir, slug, "data", id); await mkdir(dir, { recursive: true }); const body: Record = { id, description: "x".repeat(padBytes) }; if (uploadDate) body.upload_date = uploadDate; await writeFile(path.join(dir, "metadata.info.json"), JSON.stringify(body)); } test("tail read dates a downloaded-but-untranscribed video", async () => { // The case the LMDB layer structurally cannot serve: auto-transcribe's whole // candidate set is videos with no transcript, so none of them are in the // transcript index. Measured on the live corpus, layer 1 covered 122 of 870 // and this layer covered 868. const dir = await mkdtemp(path.join(tmpdir(), "recency-tail-")); try { clearRecencyCache(); const channelsDir = path.join(dir, "channels"); await writeMeta(channelsDir, "ch", "vid-new", "20260812", 400); await writeMeta(channelsDir, "ch", "vid-old", "20240101", 400); // No upload_date at all, and one with no metadata file whatsoever. await writeMeta(channelsDir, "ch", "vid-undated", null, 100); const paths = { lmdbPath: path.join(dir, "none.mdb"), channelsDir, } as Paths; const keys = await buildRecencyKeys({ paths, meta: [{ slug: "ch" }], candidateIds: new Set(["vid-new", "vid-old", "vid-undated", "vid-absent"]), owner: new Map([ ["vid-new", "ch"], ["vid-old", "ch"], ["vid-undated", "ch"], ["vid-absent", "ch"], ]), fresh: true, }); assert.deepEqual(keys.get("vid-new"), { key: "20260812", estimated: false }); assert.deepEqual(keys.get("vid-old"), { key: "20240101", estimated: false }); // Both misses fall through to layer 4 rather than vanishing from the queue. assert.deepEqual(keys.get("vid-undated"), { key: "", estimated: false }); assert.deepEqual(keys.get("vid-absent"), { key: "", estimated: false }); assert.deepEqual( ["vid-old", "vid-new"].sort(makeRecencyComparator(keys, "newest")!), ["vid-new", "vid-old"], ); } finally { await rm(dir, { recursive: true, force: true }); } }); test("tail read is skipped for ids with no known owner", async () => { // Without an owner there is no channel dir to look in; those ids must fall // through rather than probe every channel. const dir = await mkdtemp(path.join(tmpdir(), "recency-noowner-")); try { clearRecencyCache(); const channelsDir = path.join(dir, "channels"); await writeMeta(channelsDir, "ch", "vid", "20260812"); const keys = await buildRecencyKeys({ paths: { lmdbPath: path.join(dir, "none.mdb"), channelsDir } as Paths, meta: [{ slug: "ch" }], candidateIds: new Set(["vid"]), fresh: true, }); assert.deepEqual(keys.get("vid"), { key: "", estimated: false }); } finally { await rm(dir, { recursive: true, force: true }); } }); // --- Layer 4 + end-to-end degradation --------------------------------------- test("buildRecencyKeys: no index and no metadata still keys every candidate", async () => { // A missing/locked index must never fail a dispatch — it just means fewer // known dates. The YYYYMMDD_ dir-name prefix is the last real signal. const dir = await mkdtemp(path.join(tmpdir(), "recency-")); try { clearRecencyCache(); const paths = { lmdbPath: path.join(dir, "does-not-exist.mdb"), channelsDir: path.join(dir, "channels"), } as Paths; const keys = await buildRecencyKeys({ paths, meta: [{ slug: "ch" }], candidateIds: new Set(["20260812_talk", "20240101_old", "dQw4w9WgXcQ"]), fresh: true, }); assert.deepEqual(keys.get("20260812_talk"), { key: "20260812", estimated: false, }); assert.deepEqual(keys.get("20240101_old"), { key: "20240101", estimated: false, }); // An undatable id sorts oldest rather than being dropped from the queue. assert.deepEqual(keys.get("dQw4w9WgXcQ"), { key: "", estimated: false }); } finally { await rm(dir, { recursive: true, force: true }); } }); test("buildRecencyKeys: an empty candidate set does no work", async () => { const keys = await buildRecencyKeys({ paths: { lmdbPath: "/nope", channelsDir: "/nope" } as Paths, meta: [{ slug: "ch" }], candidateIds: new Set(), fresh: true, }); assert.equal(keys.size, 0); }); // --- The comparator ---------------------------------------------------------- test("makeRecencyComparator: newest is descending, oldest ascending", () => { const keys = new Map([ ["old", { key: "20240101", estimated: false }], ["mid", { key: "20250601", estimated: false }], ["new", { key: "20260812", estimated: false }], ]); assert.deepEqual( ["old", "new", "mid"].sort(makeRecencyComparator(keys, "newest")!), ["new", "mid", "old"], ); assert.deepEqual( ["new", "old", "mid"].sort(makeRecencyComparator(keys, "oldest")!), ["old", "mid", "new"], ); }); test("makeRecencyComparator: listed returns null, so nothing is sorted", () => { // Not an identity comparator — null, so buildPendingByLeaf skips the sort // entirely and today's order is reproduced by not touching it. assert.equal(makeRecencyComparator(new Map(), "listed"), null); }); test("makeRecencyComparator: an unknown id sorts oldest, never crashes", () => { const keys = new Map([ ["known", { key: "20260101", estimated: false }], ]); assert.deepEqual( ["ghost", "known"].sort(makeRecencyComparator(keys, "newest")!), ["known", "ghost"], ); }); // --- Layer 1: the index scan, and the two key spaces it straddles ------------ // Write a byChannel sub-DB the way buildIndex does: key [slug, YYYYMMDD, id], // constant value, msgpack. That id is the METADATA id, which is not always the // directory name — see the layer-1 comment in recencyIndex.ts. async function writeIndex( lmdbPath: string, rows: ReadonlyArray<[string, string, string]>, ): Promise { const { open } = await import("lmdb"); const root = open({ path: lmdbPath, maxDbs: 14 }); const byChannel = root.openDB({ name: "byChannel", encoding: "msgpack", }); for (const row of rows) await byChannel.put(row, 1); await byChannel.flushed; await root.close(); } test("layer 1 is scoped to the owning channel, not the whole corpus", async () => { // The bug this pins: candidate ids are DIRECTORY NAMES, the index is keyed by // METADATA ids, and on this corpus they diverge for ~14.5% of videos (a // Rumble dir is the URL slug, its metadata id is the embed id). A flat // corpus-wide map lets channel A's directory name collide with channel B's // metadata id and inherit B's date — a wrong answer, not a missing one. const dir = await mkdtemp(path.join(tmpdir(), "recency-scope-")); try { clearRecencyCache(); const lmdbPath = path.join(dir, "index.mdb"); const channelsDir = path.join(dir, "channels"); // "v1007ay" is a metadata id in channel `other`, and a DIRECTORY name in // channel `rumble` whose metadata id is `vxe1ae`. await writeIndex(lmdbPath, [ ["other", "20200101", "v1007ay"], ["rumble", "20260812", "vxe1ae"], ]); await writeMeta(channelsDir, "rumble", "v1007ay", "20260812"); const keys = await buildRecencyKeys({ paths: { lmdbPath, channelsDir } as Paths, meta: [{ slug: "other" }, { slug: "rumble" }], candidateIds: new Set(["v1007ay"]), owner: new Map([["v1007ay", "rumble"]]), interpolate: false, fresh: true, }); // Not 20200101 — that is the other channel's video. Layer 1 misses, and // layer 2 reads the truth out of the directory the id actually names. assert.deepEqual(keys.get("v1007ay"), { key: "20260812", estimated: false, }); } finally { await rm(dir, { recursive: true, force: true }); } }); test("layer 1 still serves an id that IS its channel's metadata id", async () => { // The 85.5% case must not regress: same-space ids are answered by the scan, // with no tail read at all (there is no metadata.info.json to read here). const dir = await mkdtemp(path.join(tmpdir(), "recency-scope-hit-")); try { clearRecencyCache(); const lmdbPath = path.join(dir, "index.mdb"); await writeIndex(lmdbPath, [["ch", "20260812", "vid"]]); const keys = await buildRecencyKeys({ paths: { lmdbPath, channelsDir: path.join(dir, "channels") } as Paths, meta: [{ slug: "ch" }], candidateIds: new Set(["vid"]), owner: new Map([["vid", "ch"]]), interpolate: false, fresh: true, }); assert.deepEqual(keys.get("vid"), { key: "20260812", estimated: false }); } finally { await rm(dir, { recursive: true, force: true }); } }); test('"cheapest" is not a recency order: no comparator, so no silent oldest-first', () => { // It is keyed by DURATION and its comparator comes from the lane's runner // (slice 1.2). Falling through to the date sort would have made it mean // oldest-first — a real ordering nobody asked for. const keys = new Map([ ["a", { key: "20260101", estimated: false }], ["b", { key: "20261231", estimated: false }], ]); assert.equal(makeRecencyComparator(keys, "cheapest"), null); assert.equal(makeRecencyComparator(keys, "listed"), null); // And the two orders this module DOES answer still do. assert.ok(makeRecencyComparator(keys, "newest")); assert.ok(makeRecencyComparator(keys, "oldest")); });