// Integration: a curated-tag edit reaching the published files, through the // REAL buildIndex and the real compose script, over a temp corpus. // // The unit tests around reapplyCuratedTags prove the derivation. This proves // the thing that actually breaks in production: a tag edit moves no video and // no mtime, so every incremental short-circuit in the build has a chance to // decide nothing happened and ship yesterday's shards. // // Run with: node_modules/.bin/tsx --test common/controller/curatedTagsBuild.test.ts import { test } from "node:test"; import assert from "node:assert/strict"; import { execFile } from "node:child_process"; import { mkdtempSync, writeFileSync, mkdirSync, readFileSync, existsSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; import { fileURLToPath } from "node:url"; import { promisify } from "node:util"; import { open } from "lmdb"; // getPaths() is lazy and cached, and nothing above calls it at import time, so // pointing the whole path graph at a temp root here is enough to isolate this // file's process from the real corpus. Every path the build touches is derived // from these three. const ROOT = mkdtempSync(path.join(tmpdir(), "curated-tags-build-")); process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts"); process.env.EXPORT_PUBLIC_DIR = path.join(ROOT, "public"); process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); const { getPaths } = await import("../lib/paths"); const { buildIndex } = await import("./buildIndex"); const { writeGlobalTags, writeSiteTags } = await import("../lib/curatedTagsStore"); const { curatedTagsRuntime, reapplyCuratedTags, META_PAGES_PENDING, } = await import("./curatedTagsIndex"); const paths = getPaths(); const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", ".."); const exec = promisify(execFile); // compose-site.ts is a SCRIPT (it runs main() on import), so it is exercised // the way the build runs it: as a child process with SITE_ID, inheriting this // file's temp-root env. `.bin/tsx` is a shell wrapper, so the CLI entry is // invoked directly. const composeSite = () => exec( process.execPath, [ path.join(REPO, "node_modules", "tsx", "dist", "cli.mjs"), path.join(REPO, "common", "bin", "compose-site.ts"), ], { env: { ...process.env, SITE_ID: SITE }, cwd: REPO }, ); const CHANNEL = "test-channel"; const SITE = "testsite"; const HIT = "vid-hit"; const MISS = "vid-miss"; function writeJson(file: string, value: unknown): void { mkdirSync(path.dirname(file), { recursive: true }); writeFileSync(file, JSON.stringify(value, null, 2)); } function videoDir(id: string): string { return path.join(paths.channelsDir, CHANNEL, "data", id); } function seedCorpus(): void { writeFileSync(paths.settingsFile, JSON.stringify({})); writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), { handling: "youtube", name: "Test Channel", url: "https://www.youtube.com/@example/videos", }); const video = (id: string, title: string, uploadDate: string) => { writeJson(path.join(videoDir(id), "metadata.info.json"), { id, title, channel: "Test Channel", upload_date: uploadDate, duration: 120, description: "fixture", webpage_url: `https://www.youtube.com/watch?v=${id}`, extractor_key: "Youtube", }); writeFileSync( path.join(videoDir(id), "transcript.en.vtt"), "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nA line of transcript.\n", ); }; video(HIT, "Stream with Elfpire Eva", "20260102"); video(MISS, "An ordinary stream", "20260101"); writeJson(path.join(paths.sitesDir, SITE, "site.json"), { siteId: SITE, siteTitle: "Test Site", siteDescription: "fixture", headerTitle: "Test Site", homeTagline: "", socialLinks: [], groups: [{ id: "default", name: "All channels", selectedByDefault: true }], defaultGroupId: "default", channels: [{ slug: CHANNEL, groupId: "default" }], }); } const build = () => buildIndex({ paths, onLog: () => {} }); function transcriptPage(): { id: string; curatedTags?: string[] }[] { return JSON.parse( readFileSync( path.join(paths.exportSharedTranscriptsDir, CHANNEL, "page-0000.json"), "utf8", ), ); } function summariesPage(): { id: string; curatedTags?: string[] }[] { return JSON.parse( readFileSync( path.join(paths.exportSitesIndexDir, SITE, "summaries", "page-0000.json"), "utf8", ), ); } function tagCounts(): { tags: Record }> } { return JSON.parse( readFileSync( path.join(paths.exportSitesIndexDir, SITE, "tag-counts.json"), "utf8", ), ); } const recordIn = (page: { id: string; curatedTags?: string[] }[], id: string) => page.find((r) => r.id === id)!; const EVA = { version: 1, tags: [ { id: "eva-collab", label: "Collab", group: "eva", groupLabel: "Eva", order: 1, rules: [ { id: "meta", kind: "metadata" as const, pattern: "elfpire", enabled: true, }, ], }, ], assignments: {}, }; // The whole slice, in one sequence, because each step's fixture is the previous // step's output. node:test runs top-level tests in order within a file. test("build 1: an untagged corpus ships no curatedTags key at all", async () => { seedCorpus(); const res = await build(); assert.equal(res.totalCount, 2); for (const page of [transcriptPage(), summariesPage()]) { for (const record of page) { assert.equal( "curatedTags" in record, false, "omitted-when-empty, or every untagged corpus re-pages on upgrade", ); } } assert.deepEqual(tagCounts().tags, {}); }); test("build 2: a rule edit with NO mtime change reaches the shards AND the site pages", async () => { // The only thing that changes between build 1 and build 2. No video // directory is touched, so the mtime diff sees +0 ~0 -0. writeGlobalTags(paths, EVA); const res = await build(); assert.equal(res.added, 0); assert.equal(res.changed, 0); assert.equal(res.removed, 0); assert.deepEqual(recordIn(transcriptPage(), HIT).curatedTags, ["eva-collab"]); assert.equal("curatedTags" in recordIn(transcriptPage(), MISS), false); assert.deepEqual(recordIn(summariesPage(), HIT).curatedTags, ["eva-collab"]); assert.equal("curatedTags" in recordIn(summariesPage(), MISS), false); assert.deepEqual(tagCounts().tags["eva-collab"], { count: 1, channels: { [CHANNEL]: 1 }, }); }); test("compose 1: the site publishes /tags.json and corpus.json points at it", async () => { await composeSite(); const tagsFile = path.join(paths.exportPublicDir, "tags.json"); const published = JSON.parse(readFileSync(tagsFile, "utf8")); assert.deepEqual(published.tags, [ { id: "eva-collab", label: "Collab", group: "eva", groupLabel: "Eva", order: 1, count: 1, channels: { [CHANNEL]: 1 }, }, ]); const corpus = JSON.parse( readFileSync(path.join(paths.exportPublicDir, "corpus.json"), "utf8"), ); assert.equal(corpus.spec, 5); assert.equal(corpus.tags.videoField, "curatedTags"); assert.match(corpus.tags.url, /\/tags\.json$/); assert.match( readFileSync(path.join(paths.exportPublicDir, "llms.txt"), "utf8"), /tags\.json/, ); }); test("compose 1b: a colour-only site overlay publishes the CORPUS label", async () => { // The site file is written through the same store the editor uses, so this // covers the whole path the overlay takes: sanitize-on-write, sanitize-on- // read, mergeTagDefs, publishedTagsFrom. A row that only recolours must not // rename the tag to its id on this site. writeSiteTags(paths, SITE, { version: 1, tags: [{ id: "eva-collab", color: "#b48ead", order: 3 }], assignments: {}, }); await composeSite(); const published = JSON.parse( readFileSync(path.join(paths.exportPublicDir, "tags.json"), "utf8"), ); assert.deepEqual(published.tags, [ { id: "eva-collab", label: "Collab", group: "eva", groupLabel: "Eva", color: "#b48ead", order: 3, count: 1, channels: { [CHANNEL]: 1 }, }, ]); // …and a corpus rename follows onto the site with no site edit at all. writeGlobalTags(paths, { ...EVA, tags: [{ ...EVA.tags[0], label: "On mic" }], }); await composeSite(); assert.equal( JSON.parse( readFileSync(path.join(paths.exportPublicDir, "tags.json"), "utf8"), ).tags[0].label, "On mic", ); // Put the fixture back for the tests below. writeGlobalTags(paths, EVA); writeSiteTags(paths, SITE, { version: 1, tags: [], assignments: {} }); }); test("build 3 + compose 2: dropping the vocabulary removes the file and the pointer", async () => { writeGlobalTags(paths, { version: 1, tags: [], assignments: {} }); await build(); assert.equal("curatedTags" in recordIn(transcriptPage(), HIT), false); assert.deepEqual(tagCounts().tags, {}); await composeSite(); assert.equal( existsSync(path.join(paths.exportPublicDir, "tags.json")), false, "a site with nothing to say must stop serving the file, not serve an empty one", ); const corpus = JSON.parse( readFileSync(path.join(paths.exportPublicDir, "corpus.json"), "utf8"), ); assert.equal(corpus.tags, undefined); }); test("an interrupted build's page debt is paid on the next build", async () => { // Reconstruct exactly what a Ctrl-C between the re-derivation and the page // build leaves behind: records re-derived in LMDB, hashes recorded, // curatedPagesPending set — and pages still holding the OLD content. writeGlobalTags(paths, EVA); const runtime = curatedTagsRuntime(EVA); const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true }); const sums = root.openDB({ name: "sums", encoding: "msgpack" }); const cues = root.openDB({ name: "cues", encoding: "msgpack" }); const subs = root.openDB({ name: "subs", encoding: "msgpack" }); const byChannel = root.openDB({ name: "byChannel", encoding: "msgpack" }); const meta = root.openDB({ name: "meta", encoding: "msgpack" }); const res = await reapplyCuratedTags({ runtime, // The sub-DB handles are structurally what the pass needs. sums: sums as never, cues: cues as never, subs: subs as never, byChannel: byChannel as never, meta: meta as never, }); assert.equal(res.changedCount, 1, "the interrupted build did re-derive a record"); await sums.flushed; await meta.flushed; assert.equal(meta.get(META_PAGES_PENDING), true); await root.close(); // The shards are stale: the interrupted build never wrote them. assert.equal("curatedTags" in recordIn(transcriptPage(), HIT), false); // Next build: no mtime moved AND the hashes now match, so nothing but the // pending flag can save these pages. await build(); assert.deepEqual(recordIn(transcriptPage(), HIT).curatedTags, ["eva-collab"]); assert.deepEqual(recordIn(summariesPage(), HIT).curatedTags, ["eva-collab"]); const after = open({ path: paths.lmdbPath, maxDbs: 18, compression: true }); const meta2 = after.openDB({ name: "meta", encoding: "msgpack" }); assert.equal(meta2.get(META_PAGES_PENDING), false, "the debt is settled"); await after.close(); }); test("a site-scoped tag: derived only on its sites' channels, and dropped from another site's summaries and counts", async () => { // A second channel whose only site is OTHER, with a video the rule matches, // and OTHER also carrying the first channel (a channel on both sites). const OTHER = "othersite"; const OTHER_CHANNEL = "other-channel"; const OTHER_HIT = "vid-other"; writeJson(path.join(paths.channelsDir, OTHER_CHANNEL, "config.json"), { handling: "youtube", name: "Other Channel", url: "https://www.youtube.com/@other/videos", }); const dir = path.join(paths.channelsDir, OTHER_CHANNEL, "data", OTHER_HIT); writeJson(path.join(dir, "metadata.info.json"), { id: OTHER_HIT, title: "Elfpire mentioned on another channel", channel: "Other Channel", upload_date: "20260103", duration: 120, description: "fixture", webpage_url: `https://www.youtube.com/watch?v=${OTHER_HIT}`, extractor_key: "Youtube", }); writeFileSync(path.join(dir, "transcript.en.vtt"), "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nAnother line.\n"); writeJson(path.join(paths.sitesDir, OTHER, "site.json"), { siteId: OTHER, siteTitle: "Other Site", siteDescription: "fixture", headerTitle: "Other Site", homeTagline: "", socialLinks: [], groups: [{ id: "default", name: "All channels", selectedByDefault: true }], defaultGroupId: "default", channels: [ { slug: CHANNEL, groupId: "default" }, { slug: OTHER_CHANNEL, groupId: "default" }, ], }); const read = (file: string) => JSON.parse(readFileSync(file, "utf8")); const summaries = (site: string): { id: string; curatedTags?: string[] }[] => read(path.join(paths.exportSitesIndexDir, site, "summaries", "page-0000.json")); const counts = (site: string): { tags: Record } => read(path.join(paths.exportSitesIndexDir, site, "tag-counts.json")); const shard = (slug: string): { id: string; curatedTags?: string[] }[] => read(path.join(paths.exportSharedTranscriptsDir, slug, "page-0000.json")); // Scoped to SITE: it exists there only. writeGlobalTags(paths, { ...EVA, tags: [{ ...EVA.tags[0], sites: [SITE] }] }); await build(); assert.equal("curatedTags" in recordIn(shard(OTHER_CHANNEL), OTHER_HIT), false, "not derived off SITE's channels"); assert.deepEqual(recordIn(summaries(SITE), HIT).curatedTags, ["eva-collab"]); assert.equal(counts(SITE).tags["eva-collab"]?.count, 1); // OTHER carries the shared channel too: the tag is dropped from its own // summaries and counts (so compose publishes no /tags.json entry for it). assert.equal("curatedTags" in recordIn(summaries(OTHER), HIT), false); assert.equal("curatedTags" in recordIn(summaries(OTHER), OTHER_HIT), false); assert.equal(counts(OTHER).tags["eva-collab"], undefined); // Unscoped again: the same rule now reaches the other channel and the other // site — the scope, not the pattern, is what kept it off. writeGlobalTags(paths, EVA); await build(); assert.deepEqual(recordIn(shard(OTHER_CHANNEL), OTHER_HIT).curatedTags, ["eva-collab"]); assert.equal(counts(OTHER).tags["eva-collab"]?.count, 2); });