// Integration: the stats cache, through the REAL buildIndex and buildStats, over // a temp corpus. // // The stats cache (`statsByPath`) used to be keyed on metadata.info.json's // mtime alone, while hasTranscript / cueCount / coverage / transcribedDate come // from the index and the transcript files. So a transcript that arrived after a // video was first seen never reached its stat, and a caption video (a VTT, no // transcript.json, no outcome sidecar) could never be dated at all. Measured on // a real corpus: one site served 1,889 videos and the homepage said 0 // transcripts, 0 channels, 0 hours. These cases pin the fix: the key also holds // the index's own record for the video, and a transcript always has a date. // // Run with: node_modules/.bin/tsx --test common/controller/buildStats.test.ts import { after, test } from "node:test"; import assert from "node:assert/strict"; import { spawnSync } from "node:child_process"; import { createRequire, syncBuiltinESMExports } from "node:module"; import { mkdirSync, mkdtempSync, readFileSync, renameSync, rmSync, symlinkSync, utimesSync, writeFileSync, } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; import { fileURLToPath } from "node:url"; // EVERY PATH getPaths() CAN RESOLVE TO A PLACE THIS FILE'S CODE MAY WRITE IS // PINNED UNDER ROOT, before anything calls it (it is lazy and cached — the // maybeMissingBuild.test.ts pattern). From common/lib/paths.ts: TRANSCRIPTS_DIR // (the LMDB, channels, jobs), SAVED_VIDEOS_DIR, SITES_DIR, SETTINGS_FILE, // EXPORT_PUBLIC_DIR, EXPORT_INDEX_DIR, EXPORT_BUILDS_DIR, EDITOR_CHANGELOG_FILE, // EXPORT_CHANGELOG_FILE, CHARTS_CONFIG_FILE, SEARCH_ALIASES_FILE, // CURATED_TAGS_FILE, ARCHILYZER_CONFIG_DIR, ARCHILYZER_SOURCE_SCRATCH. The rest // of its variables name binaries and a URL, which nothing here runs. The last // test proves no write this file caused landed outside ROOT. const ROOT = mkdtempSync(path.join(tmpdir(), "build-stats-")); const PINNED: Record = { TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), SITES_DIR: path.join(ROOT, "transcripts", "sites"), SETTINGS_FILE: path.join(ROOT, "settings.json"), EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"), EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"), CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"), SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"), CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"), ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"), }; Object.assign(process.env, PINNED); delete process.env.ARCHILYZER_STATS_ALLOW_DOWNGRADE; after(() => rmSync(ROOT, { recursive: true, force: true })); const { getPaths } = await import("../lib/paths"); const { buildIndex } = await import("./buildIndex"); const { buildStats, STATS_DOWNGRADE_ENV } = await import("./buildStats"); const { normalizeTranscript } = await import("./normalizeTranscript"); const { readStatsPages } = await import("./poolSummary"); const { siteStatsDir } = await import("../lib/site"); const { STATS_SCHEMA_VERSION } = await import("../lib/stats"); const { open } = await import("lmdb"); const paths = getPaths(); const CHANNEL = "test-channel"; const DRIVE_CHANNEL = "drive-channel"; const SITE = "testsite"; const POOL = path.join(ROOT, "pool-stats"); const COMMON = fileURLToPath(new URL("..", import.meta.url)); // ── an fs spy over the whole file ─────────────────────────────────────────── // Wraps the node:fs and node:fs/promises functions on their CJS exports objects // and syncs them into the named ESM imports the code under test holds. Records // every call's path(s), and whether it writes. Case (e) reads the reads; the // last case reads the writes. type FsCall = { fn: string; p: string; write: boolean }; const fsCalls: FsCall[] = []; { const req = createRequire(import.meta.url); const fsCjs = req("node:fs") as Record; const fspCjs = req("node:fs/promises") as Record; const READS = ["readFile", "stat", "lstat", "readdir", "readlink"]; const WRITES = ["writeFile", "appendFile", "rename", "mkdir", "rm", "rmdir", "unlink", "copyFile", "cp", "symlink", "link", "utimes", "truncate", "mkdtemp"]; const TWO_PATHS = new Set(["rename", "copyFile", "cp", "symlink", "link"]); const opensForWrite = (flags: unknown) => (typeof flags === "string" && /[wa+]/.test(flags)) || (typeof flags === "number" && (flags & 3) !== 0); const asPath = (v: unknown) => typeof v === "string" ? v : v instanceof URL ? fileURLToPath(v) : Buffer.isBuffer(v) ? v.toString() : null; const wrap = (mod: Record, name: string, write: boolean | "open") => { const fn = mod[name]; if (typeof fn !== "function") return; mod[name] = function (this: unknown, ...args: unknown[]) { const isWrite = write === "open" ? opensForWrite(args[1]) : write; const paths = TWO_PATHS.has(name.replace(/Sync$/, "")) ? [args[0], args[1]] : [args[0]]; for (const a of paths) { const p = asPath(a); if (p !== null) fsCalls.push({ fn: name, p: path.resolve(p), write: isWrite }); } return (fn as (...a: unknown[]) => unknown).apply(this, args); }; }; for (const n of READS) wrap(fspCjs, n, false); for (const n of WRITES) { wrap(fspCjs, n, true); wrap(fsCjs, n, true); wrap(fsCjs, `${n}Sync`, true); } wrap(fspCjs, "open", "open"); wrap(fsCjs, "open", "open"); wrap(fsCjs, "openSync", "open"); wrap(fsCjs, "createWriteStream", true); syncBuiltinESMExports(); } const at = (iso: string) => new Date(iso); const writeJson = (file: string, value: unknown) => { mkdirSync(path.dirname(file), { recursive: true }); writeFileSync(file, JSON.stringify(value, null, 2)); }; const videoDir = (id: string, channel = CHANNEL) => path.join(paths.channelsDir, channel, "data", id); const touch = (file: string, iso: string) => utimesSync(file, at(iso), at(iso)); // A fresh corpus (and a fresh LMDB) per test: every count below is exact. function resetCorpus(): void { for (const p of [paths.transcriptsDir, PINNED.EXPORT_INDEX_DIR, POOL, path.join(ROOT, "media")]) { rmSync(p, { recursive: true, force: true }); } mkdirSync(paths.transcriptsDir, { recursive: true }); writeFileSync(paths.settingsFile, JSON.stringify({})); writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), { handling: "youtube", name: "Test Channel", url: "https://www.youtube.com/@example/videos", }); writeJson(path.join(paths.sitesDir, SITE, "site.json"), { siteId: SITE, siteTitle: "Test Site", siteDescription: "fixture", headerTitle: "Test Site", homeTagline: "", socialLinks: [], groups: [{ id: "default", name: "All channels", selectedByDefault: true }], defaultGroupId: "default", channels: [{ slug: CHANNEL, groupId: "default" }], }); } // Metadata only — what a download leaves before any transcript exists. With // `metaIso` null the file keeps its real mtime (now): "downloaded just now". function seedVideo( id: string, metaIso: string | null = "2026-07-11T11:00:00Z", extra: Record = {}, channel = CHANNEL, ): void { const file = path.join(videoDir(id, channel), "metadata.info.json"); writeJson(file, { id, title: `Video ${id}`, channel: "Test Channel", upload_date: "20260601", duration: 120, description: "fixture", webpage_url: `https://www.youtube.com/watch?v=${id}`, extractor_key: "Youtube", ...extra, }); if (metaIso) touch(file, metaIso); } // YouTube's own captions, as a youtube-handled download writes them. parseVtt // keeps only lines carrying inline timing tags (YouTube's rolling-caption // shape), so the fixture has them. function addCaptions(id: string, iso: string, channel = CHANNEL): void { const file = path.join(videoDir(id, channel), "transcript.en.vtt"); writeFileSync( file, "WEBVTT\nKind: captions\nLanguage: en\n\n" + "00:00:00.000 --> 00:00:05.000 align:start position:0%\n" + "First<00:00:01.000> caption<00:00:02.000> line.\n\n" + "00:01:00.000 --> 00:01:50.000 align:start position:0%\n" + "Second<00:01:10.000> caption<00:01:20.000> line.\n", ); touch(file, iso); } // A Whisper-family run: transcript.json, and (unless `outcome` is false) the // transcribe-outcome.json sidecar every run writes beside it. function addWhisper(id: string, iso: string, outcome = true): void { const file = path.join(videoDir(id), "transcript.json"); writeJson(file, { duration_seconds: 120, chunks: 2, text: "one two", chunk_data: [ { start_time: 0, end_time: 5, text: "Spoken line one." }, { start_time: 60, end_time: 110, text: "Spoken line two." }, ], }); touch(file, iso); if (outcome) { writeJson(path.join(videoDir(id), "transcribe-outcome.json"), { videoId: id, transcribedAt: iso, }); } } type Stat = Awaited>[number]; async function runStats(log: string[] = []) { const res = await buildStats({ paths, onLog: (s) => log.push(s), wholePoolStatsDir: POOL, }); const byId = new Map( (await readStatsPages(POOL)).map((s) => [s.id, s]), ); return { res, byId, log }; } const runIndex = () => buildIndex({ paths, onLog: () => {} }); function statOf(byId: Map, id: string): Stat { const s = byId.get(id); assert.ok(s, `${id} has a stat`); return s; } test("(a) a transcript that arrives after the stat was cached reaches it on the next run", async () => { resetCorpus(); seedVideo("late"); await runIndex(); const first = await runStats(); assert.equal(statOf(first.byId, "late").hasTranscript, false); // Whisper runs days later. metadata.info.json — the old key — is untouched. addWhisper("late", "2026-07-17T05:11:14Z"); await runIndex(); const second = await runStats(); assert.equal(second.res.changed, 1, "the index's record moved, so the stat is redone"); const s = statOf(second.byId, "late"); assert.equal(s.hasTranscript, true); assert.equal(s.cueCount, 2); assert.equal(s.transcribedDate, "20260717"); // The MCP's "covers only N% — truncated" note reads this field. A stale // record said 0 for a complete transcript. assert.ok(s.coverage != null && s.coverage > 0.9, `coverage ${s.coverage}`); }); test("(b) stats built before the index had the video heal after the index build", async () => { resetCorpus(); seedVideo("early"); addCaptions("early", "2026-07-11T12:00:00Z"); // The pool composers run buildStats against the index as it stands; this // video was downloaded after the last index build. const first = await runStats(); assert.equal(statOf(first.byId, "early").hasTranscript, false); await runIndex(); const second = await runStats(); assert.equal(second.res.changed, 1, "the key moved off NOT_INDEXED, so the stat is redone"); const s = statOf(second.byId, "early"); assert.equal(s.hasTranscript, true); assert.equal(s.transcribedDate, "20260711"); // And the lag is said, not silent: counted in the result and logged. (No // index build has completed yet, so it is "not indexed yet".) assert.equal(first.res.notIndexedYet, 1); assert.equal(first.res.notIndexable, 0); assert.ok( first.log.some((l) => l.startsWith("1 video(s) were downloaded after the last index build")), first.log.join("\n"), ); assert.equal(second.res.notIndexedYet, 0); }); test("(c) a caption-only video is dated by its captions' arrival, not by a later Normalize", async () => { resetCorpus(); // Captions only: no transcript.json, no outcome sidecar, never normalized. seedVideo("vtt-only"); addCaptions("vtt-only", "2026-07-11T12:00:00Z"); // Captions that arrived the same day, normalized a month later. seedVideo("normalized"); addCaptions("normalized", "2026-07-11T12:00:00Z"); const norm = await normalizeTranscript({ videoDir: videoDir("normalized"), channelSlug: CHANNEL, }); assert.equal(norm.status, "wrote"); touch(path.join(videoDir("normalized"), "transcript.cues.json"), "2026-08-10T13:44:00Z"); await runIndex(); const { byId } = await runStats(); for (const id of ["vtt-only", "normalized"]) { const s = statOf(byId, id); assert.equal(s.hasTranscript, true, id); assert.equal(s.transcribedDate, "20260711", id); } }); test("(d) a Whisper video resolves exactly as before: the outcome sidecar, else transcript.json", async () => { resetCorpus(); // Transcribed before the stats first saw it, with and without the sidecar. seedVideo("with-outcome"); addWhisper("with-outcome", "2026-06-01T10:00:00Z"); seedVideo("no-outcome"); addWhisper("no-outcome", "2026-06-02T10:00:00Z", false); // The sidecar wins over every mtime, even a caption file's. seedVideo("hybrid"); addCaptions("hybrid", "2026-05-01T10:00:00Z"); addWhisper("hybrid", "2026-06-04T10:00:00Z"); // Transcribed AFTER the stats first saw it. seedVideo("after"); await runIndex(); await runStats(); addWhisper("after", "2026-06-03T10:00:00Z"); await runIndex(); const { byId } = await runStats(); const dates = Object.fromEntries( ["with-outcome", "no-outcome", "hybrid", "after"].map((id) => [ id, statOf(byId, id).transcribedDate, ]), ); assert.deepEqual(dates, { "with-outcome": "20260601", "no-outcome": "20260602", hybrid: "20260604", after: "20260603", }); }); test("(e) the key does not churn: a heal redoes one stat, then an unchanged run reads nothing per video", async () => { resetCorpus(); seedVideo("w"); addWhisper("w", "2026-06-01T10:00:00Z"); seedVideo("c"); addCaptions("c", "2026-06-01T10:00:00Z"); seedVideo("m"); await runIndex(); const first = await runStats(); assert.equal(first.res.added, 3); addWhisper("m", "2026-06-05T10:00:00Z"); await runIndex(); const heal = await runStats(); assert.equal(heal.res.added, 0); assert.equal(heal.res.changed, 1, "only the video whose transcript arrived"); // An index build over an unchanged corpus rewrites nothing the key reads. await runIndex(); const dataDir = path.join(paths.channelsDir, CHANNEL, "data"); const from = fsCalls.length; const steady = await runStats(); assert.equal(steady.res.added, 0); assert.equal(steady.res.changed, 0); assert.equal(steady.res.notIndexedYet + steady.res.notIndexable, 0); // The spy sees node:fs and node:fs/promises, async and sync. const perVideo = fsCalls.slice(from).filter((c) => c.p.startsWith(dataDir + path.sep)); assert.deepEqual( perVideo.map((c) => `${c.fn} ${path.relative(dataDir, c.p)}`).sort(), [ `stat ${path.join("c", "metadata.info.json")}`, `stat ${path.join("m", "metadata.info.json")}`, `stat ${path.join("w", "metadata.info.json")}`, ], "the unchanged path stats each metadata file and touches nothing else in a video dir", ); }); test("(f) cues are read under the index's own key, even when the metadata's upload date moved", async () => { resetCorpus(); seedVideo("moved", "2026-07-11T11:00:00Z"); addCaptions("moved", "2026-07-11T12:00:00Z"); const norm = await normalizeTranscript({ videoDir: videoDir("moved"), channelSlug: CHANNEL }); assert.equal(norm.status, "wrote"); touch(path.join(videoDir("moved"), "transcript.cues.json"), "2026-08-10T13:44:00Z"); // The metadata is rewritten with another upload date (a stream's date // settled) but stays older than the normalized cues, so buildIndex still // trusts transcript.cues.json — and keys the cues by ITS upload date. seedVideo("moved", "2026-07-12T11:00:00Z", { upload_date: "20260602" }); await runIndex(); const { byId } = await runStats(); const s = statOf(byId, "moved"); assert.equal(s.uploadDate, "20260602"); assert.equal(s.hasTranscript, true, "the cues are found under the key the index used"); assert.equal(s.cueCount, 2); }); test("(g) a re-index for another reason that changes the cues redoes the stat", async () => { resetCorpus(); seedVideo("drift"); addCaptions("drift", "2026-07-11T12:00:00Z"); await runIndex(); const first = await runStats(); assert.equal(statOf(first.byId, "drift").cueCount, 2); // A Normalize run writes transcript.cues.json (here with one cue fewer than // the raw parse). buildIndex does not re-index for that alone... await normalizeTranscript({ videoDir: videoDir("drift"), channelSlug: CHANNEL }); const cuesPath = path.join(videoDir("drift"), "transcript.cues.json"); const doc = JSON.parse(readFileSync(cuesPath, "utf8")) as { cues: unknown[] }; doc.cues = doc.cues.slice(0, 1); writeFileSync(cuesPath, JSON.stringify(doc)); // ...but an availability recheck makes it re-process the video, and then it // reads the fresher cues.json. The transcript mtime did not move. writeJson(path.join(videoDir("drift"), "availability.json"), { checkedAt: "2026-08-20T00:00:00.000Z", availability: "public", }); await runIndex(); const second = await runStats(); assert.equal(second.res.changed, 1); assert.equal(statOf(second.byId, "drift").cueCount, 1); }); test("(h) a video the index skipped is not announced as pending on every run", async () => { resetCorpus(); seedVideo("fine"); // No upload_date: buildIndex skips it (it needs one for the index key). seedVideo("undated", "2026-07-11T11:00:00Z", { upload_date: undefined }); await runIndex(); // Downloaded after that index build. seedVideo("fresh", null); for (let run = 0; run < 2; run++) { const { res, log } = await runStats(); assert.equal(res.notIndexedYet, 1, `run ${run}`); assert.equal(res.notIndexable, 1, `run ${run}`); assert.ok(log.some((l) => l.startsWith("1 video(s) were downloaded after the last index build")), log.join("\n")); assert.ok( log.some( (l) => l.startsWith("1 video(s) are not in the index although they are older than its last build") && l.includes("or its channel's media was unreachable during that build"), ), log.join("\n"), ); } await runIndex(); const { res } = await runStats(); assert.equal(res.notIndexedYet, 0, "the next index build takes the fresh one"); assert.equal(res.notIndexable, 1, "the undated one stays, and is said as such"); }); // A second channel whose MEDIA is relocated (release 17): its text in a real // data/ on the corpus disk, channels//media -> //media, with // config.mediaDir recording the target. function driveConfig(extra: Record): void { writeJson(path.join(paths.channelsDir, DRIVE_CHANNEL, "config.json"), { handling: "youtube", name: "Drive Channel", url: "https://www.youtube.com/@drive/videos", ...extra, }); } function seedDriveChannel(): { target: string } { const target = path.join(ROOT, "media", DRIVE_CHANNEL, "media"); mkdirSync(target, { recursive: true }); driveConfig({ mediaDir: target }); symlinkSync(target, path.join(paths.channelsDir, DRIVE_CHANNEL, "media")); for (const id of ["d1", "d2"]) { seedVideo(id, "2026-07-11T11:00:00Z", {}, DRIVE_CHANNEL); addCaptions(id, "2026-07-11T12:00:00Z", DRIVE_CHANNEL); } return { target }; } // The text goes away: the channel put on the retired whole-directory layout // (`data` an absolute link to //data, `dataDir` recorded) with its // drive not mounted — the text moved aside, the link dangling. And back. function retireAndUnmount(): void { const data = path.join(paths.channelsDir, DRIVE_CHANNEL, "data"); const away = path.join(ROOT, "media-away", DRIVE_CHANNEL, "data"); mkdirSync(path.dirname(away), { recursive: true }); renameSync(data, away); const target = path.join(ROOT, "media", DRIVE_CHANNEL, "data"); symlinkSync(target, data); driveConfig({ dataDir: target }); } function migrateHome(): void { const data = path.join(paths.channelsDir, DRIVE_CHANNEL, "data"); rmSync(data); renameSync(path.join(ROOT, "media-away", DRIVE_CHANNEL, "data"), data); driveConfig({ mediaDir: path.join(ROOT, "media", DRIVE_CHANNEL, "media") }); } test("(i2) release 17: an unmounted MEDIA drive does not hold the stats build", async () => { resetCorpus(); seedVideo("local"); seedDriveChannel(); await runIndex(); await runStats(); const media = path.join(ROOT, "media"); renameSync(media, `${media}-unmounted`); try { const log: string[] = []; const away = await runStats(log); assert.deepEqual(away.res.heldChannels, [], log.join("\n")); assert.equal(away.res.removed, 0); assert.equal(statOf(away.byId, "d1").hasTranscript, true); } finally { renameSync(`${media}-unmounted`, media); } }); test("(i) a channel whose text cannot be read (the retired layout, drive away) keeps its stats; a cache clear refuses", async () => { resetCorpus(); seedVideo("local"); seedDriveChannel(); await runIndex(); const mounted = await runStats(); assert.equal(statOf(mounted.byId, "d1").hasTranscript, true); // The drive is a storage location, as /storage records it; the refusal names // it by its label, never by a path. const media = path.join(ROOT, "media"); writeFileSync( paths.settingsFile, JSON.stringify({ storage: { locations: [{ id: "usb", label: "USB drive", root: media, autoRepoint: false }] } }), ); retireAndUnmount(); const log: string[] = []; const away = await runStats(log); assert.equal(away.res.removed, 0, "not read as a channel with no videos"); for (const id of ["d1", "d2", "local"]) assert.ok(away.byId.has(id), `${id} still published`); assert.deepEqual(away.res.heldChannels, [DRIVE_CHANNEL]); assert.equal(statOf(away.byId, "d1").hasTranscript, true); assert.ok( log.some((l) => l.startsWith(`Channel ${DRIVE_CHANNEL}: its media layout is the retired`) && l.includes("its 2 cached stat(s) are kept")), log.join("\n"), ); // A schema change needs the whole cache rebuilt, which cannot include a // channel it cannot read: refuse, and leave the cache as it is. setStoredSchema(STATS_SCHEMA_VERSION - 1); await assert.rejects(runStats(), (err: Error) => { assert.match( err.message, /must be rebuilt .* cannot be read: drive-channel \(its media layout is the retired whole-directory one \(run archilyzer storage migrate-tier\), on location "USB drive"\)/, ); // The ways out, mounting first, and no path in the message. assert.match( err.message, /For each: mount its media and run this again; or repair or re-point its location on \/storage; or finish or clear its move .*; or, if it is gone for good, delete the channel or set excludeFromBuild/, ); assert.ok(!err.message.includes(ROOT), err.message); return true; }); assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION - 1); assert.equal(countStats(), 3); migrateHome(); const back = await runStats(); assert.deepEqual(back.res.heldChannels, []); assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION); assert.equal(back.byId.size, 3); }); // Direct access to the temp LMDB's stats cache, for the schema cases. function withDb(fn: (dbs: { meta: ReturnType["openDB"]>; stats: ReturnType["openDB"]> }) => T): T { const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true }); try { return fn({ meta: root.openDB({ name: "statsMeta", encoding: "msgpack" }), stats: root.openDB({ name: "statsByPath", encoding: "msgpack" }), }); } finally { root.close(); } } const setStoredSchema = (v: number) => withDb(({ meta }) => meta.putSync("schema", v)); const readStoredSchema = () => withDb(({ meta }) => meta.get("schema")); const countStats = () => withDb(({ stats }) => [...stats.getKeys()].length); test("(j) the schema guard: an older cache is cleared, a newer one is refused unless overridden", async () => { resetCorpus(); seedVideo("v1"); addCaptions("v1", "2026-07-11T12:00:00Z"); await runIndex(); await runStats(); assert.equal(countStats(), 1); // Older: cleared and rebuilt, as every schema bump has always done. setStoredSchema(STATS_SCHEMA_VERSION - 1); const log: string[] = []; const older = await runStats(log); assert.ok(log.some((l) => l.includes("clearing stats cache")), log.join("\n")); assert.equal(older.res.added, 1); assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION); // Newer: refused, naming both versions and the override; nothing touched. setStoredSchema(STATS_SCHEMA_VERSION + 1); await assert.rejects( runStats(), new RegExp( `newer build \\(stats schema ${STATS_SCHEMA_VERSION + 1}; this build's is ${STATS_SCHEMA_VERSION}\\).*${STATS_DOWNGRADE_ENV}=1`, ), ); assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION + 1); assert.equal(countStats(), 1); // The CLI exits non-zero on it. const cli = spawnSync( path.join(COMMON, "node_modules", ".bin", "tsx"), ["bin/archilyzer.ts", "build", "stats"], { cwd: COMMON, env: { ...process.env }, encoding: "utf8" }, ); assert.notEqual(cli.status, 0, cli.stdout + cli.stderr); assert.match(cli.stderr, /written by a newer build/); assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION + 1); assert.equal(countStats(), 1); // A deliberate rollback, overridden: cleared and rebuilt at this version. process.env[STATS_DOWNGRADE_ENV] = "1"; try { const rolled = await runStats(); assert.equal(rolled.res.added, 1); assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION); } finally { delete process.env[STATS_DOWNGRADE_ENV]; } }); // Release 14 slice HS: the whole-pool bundle is published as the homepage's // `stats/`, and an unlisted site's content is in no public total. test("(k) the whole-pool bundle leaves out a channel only an unlisted site exposes; the site's own bundle keeps it", async () => { resetCorpus(); const seedChannel = (slug: string, ids: string[]) => { writeJson(path.join(paths.channelsDir, slug, "config.json"), { handling: "youtube", name: slug, url: `https://www.youtube.com/@${slug}/videos`, }); for (const id of ids) { seedVideo(id, "2026-07-11T11:00:00Z", {}, slug); addCaptions(id, "2026-07-11T12:00:00Z", slug); } }; seedVideo("listed-1"); seedChannel("unlisted-channel", ["u1", "u2"]); seedChannel("shared-channel", ["s1"]); seedChannel("pool-channel", ["p1"]); // The listed site also exposes the shared channel; the unlisted site exposes // its own channel and the shared one. The pool channel is on no site. const siteFile = path.join(paths.sitesDir, SITE, "site.json"); const listed = JSON.parse(readFileSync(siteFile, "utf8")); listed.channels.push({ slug: "shared-channel", groupId: "default" }); writeJson(siteFile, listed); writeJson(path.join(paths.sitesDir, "fixture-unlisted", "site.json"), { ...listed, siteId: "fixture-unlisted", siteTitle: "Unlisted", headerTitle: "Unlisted", siteUrl: "https://unlisted.example", listed: false, channels: [ { slug: "unlisted-channel", groupId: "default" }, { slug: "shared-channel", groupId: "default" }, ], }); await runIndex(); const log: string[] = []; const { byId } = await runStats(log); assert.deepEqual([...byId.keys()].sort(), ["listed-1", "p1", "s1"]); const manifest = JSON.parse(readFileSync(path.join(POOL, "manifest.json"), "utf8")); assert.equal(manifest.totalCount, 3); assert.deepEqual( manifest.channels.map((c: { slug: string }) => c.slug), ["pool-channel", "shared-channel", CHANNEL], ); assert.ok( log.includes("Stats whole-pool: 3 videos, 1 page(s); 1 channel(s) only unlisted sites expose left out."), log.join("\n"), ); // The unlisted site still builds as before: its own bundle has its videos. const own = await readStatsPages(siteStatsDir(paths, "fixture-unlisted")); assert.deepEqual(own.map((s) => s.id).sort(), ["s1", "u1", "u2"]); }); test("(z) no write this file caused landed outside its temp root", () => { // LMDB writes natively, past the spy: its file must be under the root too. assert.ok(paths.lmdbPath.startsWith(ROOT + path.sep), paths.lmdbPath); const outside = fsCalls.filter( (c) => c.write && c.p !== ROOT && !c.p.startsWith(ROOT + path.sep), ); assert.deepEqual(outside, []); assert.ok(fsCalls.some((c) => c.write), "the spy saw the writes"); });