// Integration: ALTERNATE TRACKS (lib/captionTracks.ts) through the REAL // buildIndex over a temp corpus. A record whose served `en` says something its // en-orig does not carries that track on its transcript page — so a word only // in `en` is findable, with the track named — and a record whose tracks are // identical carries nothing extra; no English VTT is shipped again as a // subtitle track; and the one-shot pass (ALT_TRACKS_VERSION) re-reads exactly // the records that can hold an alternate. // // Run with: node_modules/.bin/tsx --test common/controller/buildIndexAltTracks.test.ts import { after, test } from "node:test"; import assert from "node:assert/strict"; import { existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-alts-")); const PINNED: Record = { TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), SITES_DIR: path.join(ROOT, "transcripts", "sites"), SETTINGS_FILE: path.join(ROOT, "settings.json"), EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"), EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"), CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"), SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"), CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"), ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"), }; Object.assign(process.env, PINNED); delete process.env.ARCHILYZER_INDEX_ALLOW_HELD; after(() => rmSync(ROOT, { recursive: true, force: true })); const { getPaths } = await import("../lib/paths"); const { buildIndex } = await import("./buildIndex"); const { open } = await import("lmdb"); const { hitsAcrossTracks } = await import("../lib/captionTracks"); const paths = getPaths(); const CHANNEL = "example-channel"; const SITE = "testsite"; const DIFFER = "DifferTrack1"; // en-orig + an en that says other words const SAME = "SameTracks01"; // en-orig + an identical en const WHISPER = "Transcribed1"; // transcript.json + captions that differ const SPANISH = "SpanishSub01"; // en-orig + a Spanish subtitle track const fixture = (name: string) => readFileSync(path.join(import.meta.dirname, "..", "lib", "__fixtures__", name), "utf8"); const ROLLING = fixture("vtt-rolling.vtt"); const vtt = (...lines: [string, string, string][]) => "WEBVTT\nKind: captions\nLanguage: en\n\n" + lines.map(([a, b, t]) => `${a} --> ${b}\n${t}\n`).join("\n"); // What the served `en` says: the same opening, then a word en-orig never has, // far from anything en-orig matches. const SERVED_EN = vtt( ["00:00:03.080", "00:00:05.670", "are talking about the harbor"], ["00:01:40.000", "00:01:43.000", "the zeppelin landed in nineteen thirty"], ); const WHISPER_JSON = JSON.stringify({ transcription: [ { offsets: { from: 0, to: 2000 }, text: " a local transcription says hello" }, { offsets: { from: 2000, to: 4000 }, text: " and nothing else" }, ], }); const writeJson = (file: string, value: unknown) => { mkdirSync(path.dirname(file), { recursive: true }); writeFileSync(file, JSON.stringify(value, null, 2)); }; const dirOf = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id); function seed(): void { rmSync(paths.transcriptsDir, { recursive: true, force: true }); rmSync(PINNED.EXPORT_INDEX_DIR, { recursive: true, force: true }); writeFileSync(paths.settingsFile, "{}"); writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), { handling: "youtube", name: CHANNEL, }); writeJson(path.join(paths.sitesDir, SITE, "site.json"), { siteId: SITE, siteTitle: "Test Site", siteDescription: "fixture", headerTitle: "Test Site", homeTagline: "", socialLinks: [], groups: [{ id: "default", name: "All channels", selectedByDefault: true }], defaultGroupId: "default", channels: [{ slug: CHANNEL, groupId: "default" }], }); for (const [i, id] of [DIFFER, SAME, WHISPER, SPANISH].entries()) { writeJson(path.join(dirOf(id), "metadata.info.json"), { id, title: `Video ${id}`, upload_date: `2026060${i + 1}`, duration: 200, webpage_url: `https://www.youtube.com/watch?v=${id}`, extractor_key: "Youtube", }); } writeFileSync(path.join(dirOf(DIFFER), "transcript.en-orig.vtt"), ROLLING); writeFileSync(path.join(dirOf(DIFFER), "transcript.en.vtt"), SERVED_EN); writeFileSync(path.join(dirOf(SAME), "transcript.en-orig.vtt"), ROLLING); writeFileSync(path.join(dirOf(SAME), "transcript.en.vtt"), ROLLING); writeFileSync(path.join(dirOf(WHISPER), "transcript.json"), WHISPER_JSON); writeFileSync(path.join(dirOf(WHISPER), "transcript.en-orig.vtt"), ROLLING); writeFileSync(path.join(dirOf(SPANISH), "transcript.en-orig.vtt"), ROLLING); writeFileSync( path.join(dirOf(SPANISH), "transcript.es.vtt"), vtt(["00:00:01.000", "00:00:02.000", "hola a todos"]), ); } type Cue = { start: number; end: number; text: string }; type Rec = { id: string; cues?: Cue[]; track?: string; altTracks?: { track: string; cues: Cue[] }[]; }; // Every record of the channel's shared transcript pages, by id. function pageRecords(): Map { const dir = path.join(paths.exportSharedTranscriptsDir, CHANNEL); const out = new Map(); for (const name of readdirSync(dir)) { if (!/^page-\d+\.json$/.test(name)) continue; for (const r of JSON.parse(readFileSync(path.join(dir, name), "utf8")) as Rec[]) { out.set(r.id, r); } } return out; } function subsTracks(): Record { const dir = path.join(paths.exportSharedSubsDir, CHANNEL); const out: Record = {}; if (!existsSync(dir)) return out; for (const name of readdirSync(dir)) { if (!/^page-\d+\.json$/.test(name)) continue; for (const r of JSON.parse(readFileSync(path.join(dir, name), "utf8")) as { id: string; tracks: Record; }[]) { out[r.id] = Object.keys(r.tracks).sort(); } } return out; } async function runIndex(): Promise { const log: string[] = []; await buildIndex({ paths, onLog: (s) => log.push(s) }); return log; } test("a differing served en rides on the page as an alternate; identical tracks add nothing", async () => { seed(); await runIndex(); const recs = pageRecords(); const differ = recs.get(DIFFER)!; assert.equal(differ.track, "en-orig"); assert.deepEqual(differ.altTracks?.map((t) => t.track), ["en"]); // The word only `en` has is found — in `en`, named — and the opening both // say is found once, in the primary. const find = (q: string) => hitsAcrossTracks(differ, (cues) => cues.filter((c) => c.text.includes(q))); assert.deepEqual( find("zeppelin").map((h) => [h.track, Math.round(h.start)]), [["en", 100]], ); assert.deepEqual(find("harbor").map((h) => h.track), [undefined]); // Identical tracks: no fields at all, so the record is what it always was. const same = recs.get(SAME)!; assert.equal("track" in same, false); assert.equal("altTracks" in same, false); // A transcription is the primary; the captions it replaced are an alternate. const whisper = recs.get(WHISPER)!; assert.equal(whisper.track, "transcription"); assert.equal(whisper.cues?.[0].text, "a local transcription says hello"); assert.deepEqual(whisper.altTracks?.map((t) => t.track), ["en-orig"]); // No English VTT is a subtitle track any more; a Spanish one still is. assert.deepEqual(subsTracks(), { [SPANISH]: ["es"] }); }); test("the one-shot pass re-reads exactly the records that can hold an alternate, once", async () => { // The index as a build before alternate tracks left it: no `alts` records, // no version. const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true }); try { await root.openDB({ name: "alts", encoding: "msgpack" }).clearAsync(); await root.openDB({ name: "meta", encoding: "msgpack" }).remove("altTracks"); } finally { await root.close(); } const log = await runIndex(); // DIFFER, SAME (two English VTTs) and WHISPER (a transcription beside // captions); not SPANISH. assert.ok(log.includes("Alternate tracks v1: 3 record(s) re-read."), log.join("\n")); assert.deepEqual(pageRecords().get(DIFFER)?.altTracks?.map((t) => t.track), ["en"]); const again = await runIndex(); assert.equal(again.some((l) => /Alternate tracks v1/.test(l)), false, again.join("\n")); });