// Tests for cues.mjs — the local-or-published cue resolver. // // The fetches are stubbed against a miniature of the real published shape, so // these run offline and in CI. One separate, opt-in test hits a live archive to // prove the miniature has not drifted from reality; see LIVE below. // // Run with: pnpm test:scripts import assert from "node:assert/strict"; import test from "node:test"; import { mkdtemp, mkdir, symlink, writeFile, rm } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import { CAPTION_TRACK_RULE_VERSION, createCueSource, pageFileName, pageUrlFrom, siteOriginFromManifest, } from "./cues.mjs"; const ORIGIN = "https://example.pages.dev"; // A published record. Deliberately the SAME field set a local transcript.cues.json // carries — that parity is the whole reason one resolver can serve both. const RECORD = { slug: "chan/vid1", id: "vid1", channelSlug: "chan", title: "A Title", uploadDate: "20260101", duration: 1200, webpageUrl: "https://rumble.com/localslug-a-title.html", cues: [ { start: 0, end: 2.5, text: "first line" }, { start: 2.5, end: 5, text: "second line." }, ], }; function stubFetch(routes, seen = []) { return async (url) => { seen.push(url); if (!(url in routes)) return { ok: false, status: 404 }; return { ok: true, status: 200, json: async () => routes[url] }; }; } const ROUTES = { [`${ORIGIN}/corpus.json`]: { channels: [ { slug: "chan", manifests: { transcripts: `${ORIGIN}/transcripts/chan/manifest.json` } }, ], }, [`${ORIGIN}/transcripts/chan/manifest.json`]: { pageCount: 2, slugToPage: { vid1: 0, other: 1 }, }, [`${ORIGIN}/transcripts/chan/page-0000.json`]: [RECORD], [`${ORIGIN}/transcripts/chan/page-0001.json`]: [ { ...RECORD, id: "other", slug: "chan/other", webpageUrl: "https://rumble.com/v94hyv-x.html" }, ], }; // cacheDir:null keeps every test off the real disk cache, so one test cannot // poison another (or a developer's home directory). function source(extra = {}) { return createCueSource({ channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"), siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES), ...extra, }); } // --- the shard walk --------------------------------------------------------- test("page numbers are zero-padded to four digits", () => { assert.equal(pageFileName(0), "page-0000.json"); assert.equal(pageFileName(7), "page-0007.json"); assert.equal(pageFileName(1234), "page-1234.json"); // Not a cosmetic detail: page-0.json is a 404 on a real archive. assert.notEqual(pageFileName(0), "page-0.json"); }); test("a page URL replaces only the manifest's last segment", () => { assert.equal( pageUrlFrom("https://x.dev/transcripts/some-chan/manifest.json", 3), "https://x.dev/transcripts/some-chan/page-0003.json", ); }); test("resolves a video over HTTP with cues intact", async () => { const got = await source().load("chan", "vid1"); assert.equal(got.from, "http"); assert.equal(got.title, "A Title"); assert.equal(got.duration, 1200); assert.equal(got.cues.length, 2); assert.deepEqual(got.cues[0], { start: 0, end: 2.5, text: "first line" }); // The end times are the entire point: they are what widens a clip to a // whole sentence, and nothing else in the pipeline carries them. assert.ok(got.cues.every((c) => typeof c.end === "number")); }); test("fetches each URL once, however many videos are read", async () => { const seen = []; const src = createCueSource({ channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"), siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES, seen), }); await src.load("chan", "vid1"); await src.load("chan", "vid1"); await src.load("chan", "other"); // corpus + manifest once each, then one shard per distinct page. assert.deepEqual(seen, [ `${ORIGIN}/corpus.json`, `${ORIGIN}/transcripts/chan/manifest.json`, `${ORIGIN}/transcripts/chan/page-0000.json`, `${ORIGIN}/transcripts/chan/page-0001.json`, ]); }); // --- local wins ------------------------------------------------------------- test("a local corpus is preferred over the network", async () => { const dir = await mkdtemp(path.join(tmpdir(), "cues-local-")); try { const vdir = path.join(dir, "chan", "data", "vid1"); await mkdir(vdir, { recursive: true }); await writeFile( path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, title: "LOCAL COPY" }), ); const seen = []; const src = createCueSource({ channelsDir: dir, siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES, seen), }); const got = await src.load("chan", "vid1"); assert.equal(got.from, "local"); assert.equal(got.title, "LOCAL COPY"); assert.deepEqual(seen, [], "must not touch the network when local data exists"); } finally { await rm(dir, { recursive: true, force: true }); } }); // --- a caption record normalized under an older caption-track rule --------- test("a local caption record from before the caption-track rule is refused where the rule could read other words", async () => { const dir = await mkdtemp(path.join(tmpdir(), "cues-rule-")); try { const vdir = path.join(dir, "chan", "data", "vid1"); await mkdir(vdir, { recursive: true }); await writeFile(path.join(vdir, "transcript.en.vtt"), "WEBVTT\n\nserved words\n"); await writeFile(path.join(vdir, "transcript.en-orig.vtt"), "WEBVTT\n\nwhat was said\n"); await writeFile(path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, source: "vtt" })); const seen = []; const src = createCueSource({ channelsDir: dir, siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES, seen) }); await assert.rejects(src.load("chan", "vid1"), (err) => { assert.equal(err.name, "CueLookupError"); assert.match(err.message, /older caption-track rule.*Run Normalize for channel chan/s); return true; }); assert.deepEqual(seen, [], "never answered from the archive instead"); // Normalized under the current rule: read as usual. await writeFile( path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, source: "vtt", captionTrackRule: CAPTION_TRACK_RULE_VERSION }), ); assert.equal((await src.load("chan", "vid1")).from, "local"); // An unversioned record beside a byte-identical pair reads the same either way. await writeFile(path.join(vdir, "transcript.en.vtt"), "WEBVTT\n\nwhat was said\n"); await writeFile(path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, source: "vtt" })); assert.equal((await src.load("chan", "vid1")).from, "local"); } finally { await rm(dir, { recursive: true, force: true }); } }); test("a lone-track caption record with cues needs no rule to be read", async () => { const dir = await mkdtemp(path.join(tmpdir(), "cues-rule-")); try { const vdir = path.join(dir, "chan", "data", "vid1"); await mkdir(vdir, { recursive: true }); await writeFile(path.join(vdir, "transcript.en.vtt"), "WEBVTT\n"); await writeFile(path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, source: "vtt" })); const src = createCueSource({ channelsDir: dir, siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES) }); assert.equal((await src.load("chan", "vid1")).from, "local"); // …but one with no cues is refused: a cue-block track used to parse to none. await writeFile(path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, source: "vtt", cues: [] })); await assert.rejects(src.load("chan", "vid1"), /older caption-track rule/); } finally { await rm(dir, { recursive: true, force: true }); } }); // --- a channel the corpus holds but whose text it cannot read --------------- // // The bug these cover: on the RETIRED layout `data/` is a symlink to another // drive, with the target recorded in the channel's `config.json` as `dataDir`. // An unmounted drive reads as a plain ENOENT, which used to fall through to the // archive — so a relocated channel silently cut from a snapshot's cues, which // can differ from the corpus's by seconds. Since release 17 such a channel is // `legacy` and refused mounted or not (its way out is `migrate-tier`); a // channel whose MEDIA alone is relocated (`media/` a link, `mediaDir`) keeps // its text on the corpus disk and is read as usual, drive or no drive. const SCRATCH = process.env.CUES_TEST_DIR ?? tmpdir(); // A mirrored channel: /channels// with a config.json, and `data/` // (and `media`) however the caller wants it. async function corpusWith(slug, { dataDir, data, mediaDir } = {}) { const root = await mkdtemp(path.join(SCRATCH, "cues-reach-")); const channelDir = path.join(root, slug); await mkdir(channelDir, { recursive: true }); await writeFile( path.join(channelDir, "config.json"), JSON.stringify({ url: "https://x/", ...(dataDir ? { dataDir } : {}), ...(mediaDir ? { mediaDir } : {}), }), ); if (data === "symlink") await symlink(dataDir, path.join(channelDir, "data")); if (data === "dir") await mkdir(path.join(channelDir, "data"), { recursive: true }); if (mediaDir) await symlink(mediaDir, path.join(channelDir, "media")); return root; } async function writeCues(dir, videoId, record) { const vdir = path.join(dir, videoId); await mkdir(vdir, { recursive: true }); await writeFile(path.join(vdir, "transcript.cues.json"), JSON.stringify(record)); } function sourceOver(root, seen) { return createCueSource({ channelsDir: root, siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES, seen), }); } test("a legacy channel whose drive is not mounted throws, names the way out, and never fetches", async () => { const missing = path.join(SCRATCH, "cues-not-mounted-" + process.pid, "chan", "data"); const root = await corpusWith("chan", { dataDir: missing, data: "symlink" }); const seen = []; try { await assert.rejects(() => sourceOver(root, seen).load("chan", "vid1"), (err) => { assert.equal(err.name, "CueLookupError"); assert.match(err.message, /chan/); assert.match(err.message, /retired whole-directory one/); assert.match(err.message, /archilyzer storage migrate-tier chan/); assert.ok(err.message.includes(missing), "the error names the retired target"); return true; }); assert.deepEqual(seen, [], "a legacy channel must not fall through to the archive"); } finally { await rm(root, { recursive: true, force: true }); } }); test("a legacy channel is refused even with its drive mounted — never followed", async () => { const elsewhere = await mkdtemp(path.join(SCRATCH, "cues-drive-")); const target = path.join(elsewhere, "chan", "data"); await mkdir(target, { recursive: true }); await writeCues(target, "vid1", { ...RECORD, title: "RELOCATED COPY" }); const root = await corpusWith("chan", { dataDir: target, data: "symlink" }); const seen = []; try { await assert.rejects(() => sourceOver(root, seen).load("chan", "vid1"), /migrate-tier chan/); assert.deepEqual(seen, []); } finally { await rm(root, { recursive: true, force: true }); await rm(elsewhere, { recursive: true, force: true }); } }); test("a recorded dataDir with a real data/ is legacy too", async () => { const root = await corpusWith("chan", { dataDir: path.join(SCRATCH, "elsewhere"), data: "dir" }); const seen = []; try { await assert.rejects(() => sourceOver(root, seen).load("chan", "vid1"), /retired whole-directory/); assert.deepEqual(seen, []); } finally { await rm(root, { recursive: true, force: true }); } }); test("a tier migration in flight is refused rather than half-read", async () => { const root = await corpusWith("chan", { data: "dir" }); await writeFile( path.join(root, "chan", ".relocating.json"), JSON.stringify({ target: "/mnt/big/chan/media", direction: "out", phase: "copy", scope: "tier-migration" }), ); const seen = []; try { await assert.rejects(() => sourceOver(root, seen).load("chan", "vid1"), (err) => { assert.match(err.message, /being migrated \(phase "copy"\)/); return true; }); assert.deepEqual(seen, []); } finally { await rm(root, { recursive: true, force: true }); } }); test("a media move in flight is not refused: the text stays where it is", async () => { const root = await corpusWith("chan", { data: "dir" }); await writeCues(path.join(root, "chan", "data"), "vid1", { ...RECORD, title: "LOCAL DURING MOVE" }); await writeFile( path.join(root, "chan", ".relocating.json"), JSON.stringify({ target: "/mnt/big/chan/media", direction: "out", phase: "copy", scope: "media" }), ); const seen = []; try { const got = await sourceOver(root, seen).load("chan", "vid1"); assert.equal(got.from, "local"); assert.equal(got.title, "LOCAL DURING MOVE"); assert.deepEqual(seen, []); } finally { await rm(root, { recursive: true, force: true }); } }); test("relocated MEDIA on an unmounted drive: the text is read locally as usual", async () => { const missing = path.join(SCRATCH, "cues-media-not-mounted-" + process.pid, "chan", "media"); const root = await corpusWith("chan", { data: "dir", mediaDir: missing }); await writeCues(path.join(root, "chan", "data"), "vid1", { ...RECORD, title: "TEXT ON THE SSD" }); const seen = []; try { const got = await sourceOver(root, seen).load("chan", "vid1"); assert.equal(got.from, "local"); assert.equal(got.title, "TEXT ON THE SSD"); assert.deepEqual(seen, [], "a local read, whatever the media drive is doing"); } finally { await rm(root, { recursive: true, force: true }); } }); test("a reachable channel that simply lacks this video still falls back to HTTP", async () => { // The narrow scope of the guard: present-but-empty is not unreachable. const root = await corpusWith("chan", { data: "dir" }); try { const src = createCueSource({ channelsDir: root, siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES), }); const got = await src.load("chan", "vid1"); assert.equal(got.from, "http"); } finally { await rm(root, { recursive: true, force: true }); } }); test("no local corpus at all is the archive-only case, not an unreachable channel", async () => { // A clone with no `transcripts/channels` is a first-class way to use this // repo: the cues come over HTTP and nothing about that is an error. const seen = []; const src = createCueSource({ channelsDir: path.join(SCRATCH, "definitely-no-corpus-here"), siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES, seen), }); const got = await src.load("chan", "vid1"); assert.equal(got.from, "http"); assert.deepEqual(seen, [ `${ORIGIN}/corpus.json`, `${ORIGIN}/transcripts/chan/manifest.json`, `${ORIGIN}/transcripts/chan/page-0000.json`, ]); }); test("--cue-source local still refuses to fall back, unreachable or not", async () => { const root = await corpusWith("chan", { data: "dir" }); try { const src = createCueSource({ channelsDir: root, siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES), prefer: "local", }); await assert.rejects(() => src.load("chan", "vid1"), (err) => { assert.match(err.message, /forbids falling back/); return true; }); } finally { await rm(root, { recursive: true, force: true }); } }); // --- the Rumble two-id trap ------------------------------------------------- test("a published-id miss fails loudly, naming the two-id trap", async () => { // A Rumble video has two ids: the archive keys it by the EMBED id, while a // local cue directory is named for the URL SLUG. A manifest authored against // local dirs therefore carries an id the archive has never heard of. await assert.rejects( () => source().load("chan", "localslug"), (err) => { assert.equal(err.name, "CueLookupError"); assert.match(err.message, /EMBED id/); assert.match(err.message, /siteVideo/); assert.match(err.message, /--resolve-site-ids/); // It must also warn off the tempting wrong fix. assert.match(err.message, /citeUrl is NOT usable/); return true; }, ); }); test("an explicit siteVideo hint resolves the mismatch", async () => { const got = await source().load("chan", "localslug", { siteVideo: "vid1" }); assert.equal(got.id, "vid1"); assert.equal(got.cues.length, 2); }); test("--resolve-site-ids finds the record by scanning shards", async () => { // `other`'s webpageUrl embeds v94hyv, mirroring how a Rumble URL carries the // slug while the archive is keyed by the embed id. const got = await source({ resolveSiteIds: true }).load("chan", "v94hyv"); assert.equal(got.id, "other"); assert.equal(got.from, "http"); }); test("scanning is off by default, because a shard is up to 8 MB", async () => { await assert.rejects(() => source().load("chan", "v94hyv"), { name: "CueLookupError" }); }); // --- prefer: the two sources can genuinely disagree -------------------------- test("prefer:http ignores a local copy entirely", async () => { // Not a micro-optimisation. A published archive is a snapshot and a corpus // keeps moving: measured on the real corpus, one video of four had 65 of its // 84 cue texts rewritten and timings shifted by up to 2.24s between a // 2026-08-07 publish and the local copy six days later. Which source answered // decides where a clip gets cut. const dir = await mkdtemp(path.join(tmpdir(), "cues-prefer-")); try { const vdir = path.join(dir, "chan", "data", "vid1"); await mkdir(vdir, { recursive: true }); await writeFile( path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, title: "LOCAL COPY" }), ); const src = createCueSource({ channelsDir: dir, siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES), prefer: "http", }); const got = await src.load("chan", "vid1"); assert.equal(got.from, "http"); assert.equal(got.title, "A Title", "must be the archive's copy, not the local one"); } finally { await rm(dir, { recursive: true, force: true }); } }); test("prefer:local refuses to fall back rather than cut from other cues", async () => { const src = createCueSource({ channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"), siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES), prefer: "local", }); await assert.rejects(() => src.load("chan", "vid1"), (err) => { assert.equal(err.name, "CueLookupError"); assert.match(err.message, /forbids falling back/); return true; }); }); // --- origin discovery ------------------------------------------------------- test("the archive origin comes from the manifest, in priority order", () => { assert.equal( siteOriginFromManifest({ provenance: { siteOrigin: "https://a.dev/" } }), "https://a.dev", "trailing slash trimmed", ); assert.equal( siteOriginFromManifest({ provenance: { corpus: "remote:https://b.dev" } }), "https://b.dev", ); assert.equal( siteOriginFromManifest({ provenance: { shareLink: "https://c.dev/?v=x&t=1" } }), "https://c.dev", ); // A local-corpus manifest names no remote origin, and must not invent one. assert.equal(siteOriginFromManifest({ provenance: { corpus: "local:/srv/x" } }), null); assert.equal(siteOriginFromManifest({}), null); }); test("no origin and no local copy is a clear error, not a crash", async () => { const src = createCueSource({ channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"), siteOrigin: null, cacheDir: null, fetchImpl: stubFetch({}), }); await assert.rejects(() => src.load("chan", "vid1"), (err) => { assert.equal(err.name, "CueLookupError"); assert.match(err.message, /no archive origin/); return true; }); }); test("a channel absent from corpus.json is reported as such", async () => { await assert.rejects(() => source().load("nosuch", "vid1"), (err) => { assert.match(err.message, /not in .*corpus\.json/); return true; }); }); test("a cited report site is named as such: it has no cues to read", async () => { const cited = { [`${ORIGIN}/corpus.json`]: { spec: 5, site: { scope: "cited" }, channels: [] }, }; await assert.rejects(() => source({ fetchImpl: stubFetch(cited) }).load("chan", "vid1"), (err) => { assert.match(err.message, /cited report site/); assert.match(err.message, /publishes no transcripts/); return true; }); }); // --- reality check ---------------------------------------------------------- // Opt-in: `LIVE=1 pnpm test:scripts`. The stubs above encode assumptions about a // published archive's shape; this is the only thing that can catch them going // stale. Skipped by default so the suite stays offline and deterministic. test("LIVE: a real archive still matches the shape these stubs assume", { skip: process.env.LIVE === "1" ? false : "set LIVE=1 to hit the network", }, async () => { const src = createCueSource({ channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"), siteOrigin: "https://jeralyzer.pages.dev", cacheDir: null, }); const got = await src.load("chrissie-mayr", "2Pn_rMrHmEs"); assert.equal(got.from, "http"); assert.ok(got.cues.length > 0); assert.ok(got.cues.every((c) => typeof c.start === "number" && typeof c.end === "number")); for (const field of ["title", "uploadDate", "duration", "webpageUrl"]) { assert.ok(got[field] !== undefined, `published record should carry ${field}`); } });