// Integration: the caption-track pass (CAPTION_TRACK_RULE_VERSION) re-reads, // through the REAL buildIndex over a temp corpus, exactly the records an older // caption-track rule may have read differently — one with two English VTTs, // one whose stored cues are empty — and leaves every other record alone. // // Run with: node_modules/.bin/tsx --test common/controller/buildIndexCaptionTrack.test.ts import { after, test } from "node:test"; import assert from "node:assert/strict"; import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-captions-")); const PINNED: Record = { TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), SITES_DIR: path.join(ROOT, "transcripts", "sites"), SETTINGS_FILE: path.join(ROOT, "settings.json"), EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"), EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"), CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"), SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"), CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"), ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"), }; Object.assign(process.env, PINNED); delete process.env.ARCHILYZER_INDEX_ALLOW_HELD; after(() => rmSync(ROOT, { recursive: true, force: true })); const { getPaths } = await import("../lib/paths"); const { buildIndex } = await import("./buildIndex"); const { open } = await import("lmdb"); const paths = getPaths(); const CHANNEL = "example-channel"; const SITE = "testsite"; const BOTH = "BothTracks01"; // served en (cue blocks) + en-orig (rolling) const BLOCKS = "CueBlocks001"; // a lone cue-block en const ROLL = "RollingOnly1"; // a lone rolling en — nothing for the pass to do const fixture = (name: string) => readFileSync(path.join(import.meta.dirname, "..", "lib", "__fixtures__", name), "utf8"); const ROLLING = fixture("vtt-rolling.vtt"); const CUE_BLOCKS = fixture("vtt-cue-blocks.vtt"); const writeJson = (file: string, value: unknown) => { mkdirSync(path.dirname(file), { recursive: true }); writeFileSync(file, JSON.stringify(value, null, 2)); }; const dirOf = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id); function seed(): void { rmSync(paths.transcriptsDir, { recursive: true, force: true }); rmSync(PINNED.EXPORT_INDEX_DIR, { recursive: true, force: true }); writeFileSync(paths.settingsFile, "{}"); writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), { handling: "youtube", name: CHANNEL, }); writeJson(path.join(paths.sitesDir, SITE, "site.json"), { siteId: SITE, siteTitle: "Test Site", siteDescription: "fixture", headerTitle: "Test Site", homeTagline: "", socialLinks: [], groups: [{ id: "default", name: "All channels", selectedByDefault: true }], defaultGroupId: "default", channels: [{ slug: CHANNEL, groupId: "default" }], }); for (const [i, id] of [BOTH, BLOCKS, ROLL].entries()) { writeJson(path.join(dirOf(id), "metadata.info.json"), { id, title: `Video ${id}`, upload_date: `2026060${i + 1}`, duration: 30, webpage_url: `https://www.youtube.com/watch?v=${id}`, extractor_key: "Youtube", }); } writeFileSync(path.join(dirOf(BOTH), "transcript.en.vtt"), CUE_BLOCKS); writeFileSync(path.join(dirOf(BOTH), "transcript.en-orig.vtt"), ROLLING); writeFileSync(path.join(dirOf(BLOCKS), "transcript.en.vtt"), CUE_BLOCKS); writeFileSync(path.join(dirOf(ROLL), "transcript.en.vtt"), ROLLING); } type Summary = { id: string }; type Cue = { start: number; end: number; text: string }; function withIndex(fn: (db: (name: string) => ReturnType["openDB"]>) => T): T { const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true }); try { return fn((name) => root.openDB({ name, encoding: "msgpack" })); } finally { root.close(); } } const keyOf = (db: (name: string) => ReturnType["openDB"]>, id: string) => { for (const { key, value } of db("sums").getRange()) { if ((value as Summary).id === id) return key; } throw new Error(`${id} is not indexed`); }; const cuesOf = (id: string) => withIndex((db) => db("cues").get(keyOf(db, id)) as Cue[] | undefined); async function runIndex(): Promise { const log: string[] = []; await buildIndex({ paths, onLog: (s) => log.push(s) }); return log; } test("a fresh index reads en-orig beside a served en, and a lone cue-block en as text", async () => { seed(); const log = await runIndex(); assert.equal(cuesOf(BOTH)?.[0].text, "are talking about the harbor"); assert.equal(cuesOf(BLOCKS)?.length, 7); assert.equal(cuesOf(ROLL)?.length, 3); // A first build has nothing indexed under an older rule: no pass. assert.equal(log.some((l) => /Caption track/.test(l)), false, log.join("\n")); }); test("an index built under the old rule is re-read once, for exactly the records the rule reaches", async () => { // The index as an older build left it: the served en's text for BOTH, no // cues for BLOCKS, and no caption-track version recorded. const heldRoll = cuesOf(ROLL); withIndex((db) => { const cues = db("cues"); cues.putSync(keyOf(db, BOTH), [{ start: 0, end: 6, text: "served words" }]); cues.putSync(keyOf(db, BLOCKS), []); // ROLL's record is marked so a re-read would show. cues.putSync(keyOf(db, ROLL), [{ start: 0, end: 1, text: "untouched" }]); db("meta").removeSync("captionTrackRule"); }); const log = await runIndex(); assert.ok(log.includes("Caption track v1: 2 record(s) re-read."), log.join("\n")); assert.ok( log.includes( `Caption track v1: ${CHANNEL}: 2 re-read, 2 now read different text, 1 had no text and now do.`, ), log.join("\n"), ); assert.equal(cuesOf(BOTH)?.[0].text, "are talking about the harbor"); assert.equal(cuesOf(BLOCKS)?.length, 7); assert.deepEqual(cuesOf(ROLL), [{ start: 0, end: 1, text: "untouched" }]); assert.notDeepEqual(heldRoll, cuesOf(ROLL)); // Recorded: the next build does not look again. const again = await runIndex(); assert.equal(again.some((l) => /Caption track/.test(l)), false, again.join("\n")); }); // THE 2026-10-01 BUG, CLOSED (release 20 D3): a preferred track that parses to // no cues never hides a later one with text — in either order of the two // tracks the bug was about, and past `en` to a regional track. Added to the // index built above, so the records are read under the current rule. test("a preferred English track with no cues falls through to the next one with text", async () => { const EMPTY_VTT = "WEBVTT\nKind: captions\nLanguage: en\n\n"; const ORIG_EMPTY = "OrigEmpty001"; // en-orig empty, en with text const EN_EMPTY = "EnEmptyUS001"; // en empty, no en-orig, en-US with text for (const [i, id] of [ORIG_EMPTY, EN_EMPTY].entries()) { writeJson(path.join(dirOf(id), "metadata.info.json"), { id, title: `Video ${id}`, upload_date: `2026070${i + 1}`, duration: 30, webpage_url: `https://www.youtube.com/watch?v=${id}`, extractor_key: "Youtube", }); } writeFileSync(path.join(dirOf(ORIG_EMPTY), "transcript.en-orig.vtt"), EMPTY_VTT); writeFileSync(path.join(dirOf(ORIG_EMPTY), "transcript.en.vtt"), ROLLING); writeFileSync(path.join(dirOf(EN_EMPTY), "transcript.en.vtt"), EMPTY_VTT); writeFileSync(path.join(dirOf(EN_EMPTY), "transcript.en-US.vtt"), CUE_BLOCKS); await runIndex(); assert.equal(cuesOf(ORIG_EMPTY)?.length, 3); assert.equal(cuesOf(EN_EMPTY)?.length, 7); assert.equal(cuesOf(EN_EMPTY)?.[0].text.startsWith("welcome back everyone"), true); });