import { test } from "node:test"; import assert from "node:assert/strict"; import { mkdtempSync, rmSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; import { open } from "lmdb"; import type { Paths } from "../lib/paths"; import type { TranscriptSummary } from "../lib/transcripts"; import type { CuratedTagDef } from "../lib/curatedTags"; import { previewTagRule, recordIdsByVideoDir, ruleHitsForVideo, } from "./curatedTagsPreview"; // Run with: pnpm -C common exec tsx --test "controller/curatedTagsPreview.test.ts" // // A REAL LMDB FILE, not a stand-in — which is the whole point of this file. // The bug this suite exists to stop was invisible to a fake: the preview opened // the index WITHOUT `compression: true` while buildIndex writes with it, and // lmdb-js only reads the compressed status byte when the reading store carries // a compression object. Every value over its ~1 KB threshold — that is, every // real description and every transcript — then threw "Data read, but end of // buffer not reached", and because the video page awaits ruleHitsForVideo // during its render, one rule was enough to 500 every video-detail page. // // So the fixture writes the way buildIndex writes, and one record is // deliberately fat. const KEEP = process.env.KEEP_TAG_PREVIEW_FIXTURE === "1"; // Comfortably over lmdb-js's 1000-byte compression threshold. const FAT = "a fat description that keeps going. ".repeat(60); type IndexKey = [string, string, string]; type Fixture = { paths: Paths; cleanup(): void; }; async function buildIndexFixture( records: { slug: string; dir?: string; id: string; uploadDate: string; title?: string; description?: string; cues?: { start: number; end: number; text: string }[]; chat?: { start: number; end: number; text: string }[]; }[], ): Promise { const dir = mkdtempSync(path.join(tmpdir(), "tag-preview-")); const lmdbPath = path.join(dir, "index.mdb"); // maxDbs and compression EXACTLY as buildIndex opens it. const root = open({ path: lmdbPath, maxDbs: 18, compression: true }); const sums = root.openDB({ name: "sums", encoding: "msgpack", }); const cues = root.openDB({ name: "cues", encoding: "msgpack" }); const subs = root.openDB({ name: "subs", encoding: "msgpack" }); const byChannel = root.openDB({ name: "byChannel", encoding: "msgpack" }); const mtimes = root.openDB({ name: "mtimes", encoding: "msgpack" }); let last: Promise | undefined; for (const r of records) { const key: IndexKey = [r.uploadDate, r.slug, r.id]; const summary = { slug: `${r.slug}/${r.id}`, id: r.id, channelSlug: r.slug, title: r.title ?? r.id, uploadDate: r.uploadDate, duration: 60, channel: r.slug, description: r.description ?? "", tags: [], isLivestream: false, ageRestricted: false, platform: "youtube", webpageUrl: `https://example.invalid/${r.id}`, } as unknown as TranscriptSummary; last = sums.put(key, summary); byChannel.put([r.slug, r.uploadDate, r.id], 1); mtimes.put([r.slug, r.dir ?? r.id], { metaMs: 1, indexKey: key }); if (r.cues) last = cues.put(key, r.cues); if (r.chat) last = subs.put(key, [{ track: "live_chat", cues: r.chat }]); } await last; await root.close(); return { paths: { lmdbPath } as Paths, cleanup: () => { if (!KEEP) rmSync(dir, { recursive: true, force: true }); }, }; } function def( id: string, rule: Partial & { kind: "metadata" | "caption" | "chat-author"; pattern: string; }, ): CuratedTagDef { return { id, label: id, rules: [{ id: "r1", enabled: true, ...rule }], }; } test("a record far over the compression threshold is read, not thrown", async () => { const f = await buildIndexFixture([ { slug: "lm", id: "big", uploadDate: "20260101", title: "An ordinary title", description: `${FAT} elfpire guested here`, cues: Array.from({ length: 400 }, (_, i) => ({ start: i, end: i + 1, text: `a caption line number ${i} with enough words to matter`, })), }, ]); try { const meta = previewTagRule(f.paths, { def: def("t", { kind: "metadata", pattern: "elfpire" }), channels: [], allSlugs: ["lm"], }); assert.equal(meta.indexAvailable, true); // THE ASSERTION THAT WOULD HAVE CAUGHT IT: without compression on the read // side this is 0 matches and 1 unreadable, not 1 and 0. assert.equal(meta.unreadable, 0); assert.deepEqual( meta.matches.map((m) => m.id), ["big"], ); // And the caption path, which decodes a four-hundred-cue value. const caption = previewTagRule(f.paths, { def: def("t", { kind: "caption", pattern: "caption line number 399" }), channels: [], allSlugs: ["lm"], }); assert.equal(caption.unreadable, 0); assert.equal(caption.matches.length, 1); // Same file, the single-record path the video page uses. const hits = ruleHitsForVideo( f.paths, [def("t", { kind: "metadata", pattern: "elfpire" })], "lm", "big", "20260101", ); assert.deepEqual(hits, { hits: ["t"], indexAvailable: true, indexed: true }); } finally { f.cleanup(); } }); test("a chat-author rule reads the live_chat track and no other", async () => { const f = await buildIndexFixture([ { slug: "lm", id: "chatty", uploadDate: "20260101", chat: [ { start: 1, end: 2, text: "SomeoneElse: hello" }, { start: 3, end: 4, text: "ElfpireEva: hi" }, ], }, { slug: "lm", id: "quiet", uploadDate: "20260102", // The NAME IS IN THE MESSAGE, not in the author prefix — a chat-author // rule must not match it. chat: [{ start: 1, end: 2, text: "SomeoneElse: elfpire was here" }], }, ]); try { const r = previewTagRule(f.paths, { def: def("t", { kind: "chat-author", pattern: "elfpire" }), channels: [], allSlugs: ["lm"], }); assert.deepEqual( r.matches.map((m) => m.id), ["chatty"], ); } finally { f.cleanup(); } }); test("the scan cap stops the call, and the cursor resumes past the boundary key", async () => { // 300 records, all matching, with a CAPTION rule — whose per-call cap is 250. const records = Array.from({ length: 300 }, (_, i) => ({ slug: "lm", id: `v${String(i).padStart(3, "0")}`, uploadDate: `2026${String(100 + i).padStart(4, "0")}`, cues: [{ start: 0, end: 1, text: "the needle is here" }], })); const f = await buildIndexFixture(records); try { const first = previewTagRule(f.paths, { def: def("t", { kind: "caption", pattern: "needle" }), channels: [], allSlugs: ["lm"], }); // The match cap (100) bites before the scan cap here, which is itself the // contract: a preview is for judging a rule, not for enumerating its hits. assert.equal(first.stoppedBy, "match-cap"); assert.equal(first.matches.length, 100); assert.ok(first.nextCursor); const second = previewTagRule(f.paths, { def: def("t", { kind: "caption", pattern: "needle" }), channels: [], allSlugs: ["lm"], cursor: first.nextCursor, }); // NO OVERLAP. getRange's start is inclusive, so a resume that did not skip // the boundary key would re-list the row the operator has just acted on. const firstIds = new Set(first.matches.map((m) => m.id)); assert.equal( second.matches.some((m) => firstIds.has(m.id)), false, "the resumed page re-listed a record from the first page", ); assert.equal(second.matches[0].id, "v100"); } finally { f.cleanup(); } }); test("the scan cap is reported when the matches do not fill a page", async () => { // 300 records, only the last one matching, caption kind -> the 250-record // scan cap stops the call before the match is reached. const records = Array.from({ length: 300 }, (_, i) => ({ slug: "lm", id: `v${String(i).padStart(3, "0")}`, uploadDate: `2026${String(100 + i).padStart(4, "0")}`, cues: [{ start: 0, end: 1, text: i === 299 ? "the needle" : "hay" }], })); const f = await buildIndexFixture(records); try { const first = previewTagRule(f.paths, { def: def("t", { kind: "caption", pattern: "needle" }), channels: [], allSlugs: ["lm"], }); assert.equal(first.stoppedBy, "scan-cap"); assert.equal(first.scanned, 250); assert.equal(first.matches.length, 0); const second = previewTagRule(f.paths, { def: def("t", { kind: "caption", pattern: "needle" }), channels: [], allSlugs: ["lm"], cursor: first.nextCursor, }); assert.equal(second.scanned, 50); assert.deepEqual( second.matches.map((m) => m.id), ["v299"], ); assert.equal(second.stoppedBy, "end"); assert.equal(second.nextCursor, null); } finally { f.cleanup(); } }); test("a channel-scoped rule scans only its own channel", async () => { const f = await buildIndexFixture([ { slug: "lm", id: "a1", uploadDate: "20260101", title: "needle here" }, { slug: "other", id: "b1", uploadDate: "20260101", title: "needle here" }, { slug: "other", id: "b2", uploadDate: "20260102", title: "needle here" }, ]); try { const scoped = previewTagRule(f.paths, { def: { id: "t", label: "t", rules: [ { id: "r1", kind: "metadata", pattern: "needle", enabled: true, channels: ["lm"], }, ], }, channels: ["lm"], allSlugs: ["lm", "other"], }); assert.deepEqual( scoped.matches.map((m) => m.id), ["a1"], ); // SCANNED, not merely unmatched: the other channel's two records were never // looked at, which is the entire point of scoping a rule to a channel. assert.equal(scoped.scanned, 1); const unscoped = previewTagRule(f.paths, { def: def("t", { kind: "metadata", pattern: "needle" }), channels: [], allSlugs: ["lm", "other"], }); assert.equal(unscoped.scanned, 3); assert.equal(unscoped.matches.length, 3); } finally { f.cleanup(); } }); test("recordIdsByVideoDir maps a directory name to the record id", async () => { const f = await buildIndexFixture([ { slug: "rum", dir: "a-url-slug", id: "v2embedid", uploadDate: "20260101" }, { slug: "rum", id: "plain", uploadDate: "20260102" }, { slug: "other", dir: "elsewhere", id: "nope", uploadDate: "20260103" }, ]); try { const map = recordIdsByVideoDir(f.paths, "rum"); assert.equal(map.get("a-url-slug"), "v2embedid"); // A directory whose name IS the id still appears — callers fall back to the // name only when the index has never seen the video at all. assert.equal(map.get("plain"), "plain"); // Another channel's directories are not in this channel's map. assert.equal(map.get("elsewhere"), undefined); assert.equal(map.size, 2); } finally { f.cleanup(); } }); test("no index is an empty answer, never a throw", () => { const paths = { lmdbPath: path.join(tmpdir(), "definitely-not-here.mdb") } as Paths; const r = previewTagRule(paths, { def: def("t", { kind: "metadata", pattern: "x" }), channels: [], allSlugs: ["lm"], }); assert.equal(r.indexAvailable, false); assert.deepEqual(r.matches, []); assert.deepEqual(recordIdsByVideoDir(paths, "lm").size, 0); assert.deepEqual(ruleHitsForVideo(paths, [def("t", { kind: "metadata", pattern: "x" })], "lm", "v1"), { hits: [], indexAvailable: false, indexed: false, }); });