commit 58d877e65bb26e6325232da5a04e920da35cfcc1
parent cc70080585a9fe8c0810d1eb85052edbc57a3cd9
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 28 Sep 2026 18:42:32 -0400
common: buildStats refuses a newer cache, keys on the whole index record, keeps an unmounted drive's stats
Review of the stats cache key fix:
- A build never clears a stats cache a NEWER schema wrote: it throws, naming
both versions and ARCHILYZER_STATS_ALLOW_DOWNGRADE (declared in envVars.ts,
ENVIRONMENT.md regenerated) for a deliberate rollback.
- The key is the whole `mtimes` record (every input that makes buildIndex
re-process the video, and its indexKey), and the cues are read under that
indexKey: closes the cueCount drift and the uploadDate key split.
- "Not indexed yet" (metadata newer than the last index scan) and "not
indexable" (the scan saw it and skipped it) are counted and logged apart, so
a video with no upload_date is not announced as pending on every run.
- A channel whose media is not reachable (inspectChannelMedia) is not
rescanned and its cached stats are kept; a cache clear with one refuses.
The job registry's needsMedia guard is per channel and never covered this
pool-wide build, from the editor or the CLI.
- buildStats.test.ts: cases (f)-(j) for the above, every getPaths() path
pinned under the temp root, the root removed after, and an fs spy (node:fs
and node:fs/promises, async and sync) asserting no write lands outside it.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
4 files changed, 490 insertions(+), 111 deletions(-)
diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md
@@ -75,6 +75,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not
| `TRANSCRIPT_PLATFORM_LINKS` | off | `1` cites platform watch pages instead of the archive's own pages. | common/lib/archive/reader-fs.ts |
| `AUDIO_CHECK_RESUME_DURING_PROBE` | the channel's `audioCheck.resumeDuringProbe` | `1` or `true` resumes yt-dlp during the audio check's probe, anything else holds it, for a one-off comparison run; unset = the channel's setting. | common/ytdlp/audioCheckedDownload.ts |
| `AUDIO_CHECK_BACKOFF_FACTOR` | the built-in factor | The audio check's interval backoff factor, in (0, 1], for a one-off run. | common/ytdlp/audioCheckedDownload.ts |
+| `ARCHILYZER_STATS_ALLOW_DOWNGRADE` | off | `1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions. | common/controller/buildStats.ts |
| `MCP_IO_STATS` | off | `1` turns on per-call I/O accounting, for `mcp/bench`. | common/lib/archive/io-stats.ts |
| `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's `fetch_clip` talk to. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool |
| `ARCHILYZER_AGENT` | `cli` | Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`. | scripts/archilyzer-ops.mjs |
diff --git a/common/controller/buildStats.test.ts b/common/controller/buildStats.test.ts
@@ -8,56 +8,132 @@
// transcript.json, no outcome sidecar) could never be dated at all. Measured on
// a real corpus: one site served 1,889 videos and the homepage said 0
// transcripts, 0 channels, 0 hours. These cases pin the fix: the key also holds
-// the index's transcript record, and a transcript always has a date.
+// the index's own record for the video, and a transcript always has a date.
//
// Run with: node_modules/.bin/tsx --test common/controller/buildStats.test.ts
-import { test } from "node:test";
+import { after, test } from "node:test";
import assert from "node:assert/strict";
+import { spawnSync } from "node:child_process";
import { createRequire, syncBuiltinESMExports } from "node:module";
import {
mkdirSync,
mkdtempSync,
+ readFileSync,
+ renameSync,
rmSync,
+ symlinkSync,
utimesSync,
writeFileSync,
} from "node:fs";
import { tmpdir } from "node:os";
import path from "node:path";
+import { fileURLToPath } from "node:url";
-// getPaths() is lazy and cached, and nothing above calls it at import time, so
-// pointing the whole path graph at a temp root here isolates this file's
-// process from the real corpus (the maybeMissingBuild.test.ts pattern).
+// EVERY PATH getPaths() CAN RESOLVE TO A PLACE THIS FILE'S CODE MAY WRITE IS
+// PINNED UNDER ROOT, before anything calls it (it is lazy and cached — the
+// maybeMissingBuild.test.ts pattern). From common/lib/paths.ts: TRANSCRIPTS_DIR
+// (the LMDB, channels, jobs), SAVED_VIDEOS_DIR, SITES_DIR, SETTINGS_FILE,
+// EXPORT_PUBLIC_DIR, EXPORT_INDEX_DIR, EXPORT_BUILDS_DIR, EDITOR_CHANGELOG_FILE,
+// EXPORT_CHANGELOG_FILE, CHARTS_CONFIG_FILE, SEARCH_ALIASES_FILE,
+// CURATED_TAGS_FILE, ARCHILYZER_CONFIG_DIR, ARCHILYZER_SOURCE_SCRATCH. The rest
+// of its variables name binaries and a URL, which nothing here runs. The last
+// test proves no write this file caused landed outside ROOT.
const ROOT = mkdtempSync(path.join(tmpdir(), "build-stats-"));
-process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts");
-process.env.EXPORT_PUBLIC_DIR = path.join(ROOT, "public");
-process.env.EXPORT_INDEX_DIR = path.join(ROOT, ".export-index");
-process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+const PINNED: Record<string, string> = {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"),
+ EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"),
+ CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"),
+ SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"),
+ CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+ ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"),
+};
+Object.assign(process.env, PINNED);
+delete process.env.ARCHILYZER_STATS_ALLOW_DOWNGRADE;
+after(() => rmSync(ROOT, { recursive: true, force: true }));
const { getPaths } = await import("../lib/paths");
const { buildIndex } = await import("./buildIndex");
-const { buildStats } = await import("./buildStats");
+const { buildStats, STATS_DOWNGRADE_ENV } = await import("./buildStats");
const { normalizeTranscript } = await import("./normalizeTranscript");
const { readStatsPages } = await import("./poolSummary");
+const { STATS_SCHEMA_VERSION } = await import("../lib/stats");
+const { open } = await import("lmdb");
const paths = getPaths();
const CHANNEL = "test-channel";
+const DRIVE_CHANNEL = "drive-channel";
const SITE = "testsite";
const POOL = path.join(ROOT, "pool-stats");
+const COMMON = fileURLToPath(new URL("..", import.meta.url));
+
+// ── an fs spy over the whole file ───────────────────────────────────────────
+// Wraps the node:fs and node:fs/promises functions on their CJS exports objects
+// and syncs them into the named ESM imports the code under test holds. Records
+// every call's path(s), and whether it writes. Case (e) reads the reads; the
+// last case reads the writes.
+type FsCall = { fn: string; p: string; write: boolean };
+const fsCalls: FsCall[] = [];
+{
+ const req = createRequire(import.meta.url);
+ const fsCjs = req("node:fs") as Record<string, unknown>;
+ const fspCjs = req("node:fs/promises") as Record<string, unknown>;
+ const READS = ["readFile", "stat", "lstat", "readdir", "readlink"];
+ const WRITES = ["writeFile", "appendFile", "rename", "mkdir", "rm", "rmdir", "unlink", "copyFile", "cp", "symlink", "link", "utimes", "truncate", "mkdtemp"];
+ const TWO_PATHS = new Set(["rename", "copyFile", "cp", "symlink", "link"]);
+ const opensForWrite = (flags: unknown) =>
+ (typeof flags === "string" && /[wa+]/.test(flags)) ||
+ (typeof flags === "number" && (flags & 3) !== 0);
+ const asPath = (v: unknown) =>
+ typeof v === "string" ? v : v instanceof URL ? fileURLToPath(v) : Buffer.isBuffer(v) ? v.toString() : null;
+ const wrap = (mod: Record<string, unknown>, name: string, write: boolean | "open") => {
+ const fn = mod[name];
+ if (typeof fn !== "function") return;
+ mod[name] = function (this: unknown, ...args: unknown[]) {
+ const isWrite = write === "open" ? opensForWrite(args[1]) : write;
+ const paths = TWO_PATHS.has(name.replace(/Sync$/, "")) ? [args[0], args[1]] : [args[0]];
+ for (const a of paths) {
+ const p = asPath(a);
+ if (p !== null) fsCalls.push({ fn: name, p: path.resolve(p), write: isWrite });
+ }
+ return (fn as (...a: unknown[]) => unknown).apply(this, args);
+ };
+ };
+ for (const n of READS) wrap(fspCjs, n, false);
+ for (const n of WRITES) {
+ wrap(fspCjs, n, true);
+ wrap(fsCjs, n, true);
+ wrap(fsCjs, `${n}Sync`, true);
+ }
+ wrap(fspCjs, "open", "open");
+ wrap(fsCjs, "open", "open");
+ wrap(fsCjs, "openSync", "open");
+ wrap(fsCjs, "createWriteStream", true);
+ syncBuiltinESMExports();
+}
const at = (iso: string) => new Date(iso);
const writeJson = (file: string, value: unknown) => {
mkdirSync(path.dirname(file), { recursive: true });
writeFileSync(file, JSON.stringify(value, null, 2));
};
-const videoDir = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id);
+const videoDir = (id: string, channel = CHANNEL) =>
+ path.join(paths.channelsDir, channel, "data", id);
const touch = (file: string, iso: string) => utimesSync(file, at(iso), at(iso));
// A fresh corpus (and a fresh LMDB) per test: every count below is exact.
function resetCorpus(): void {
- rmSync(paths.transcriptsDir, { recursive: true, force: true });
- rmSync(path.join(ROOT, ".export-index"), { recursive: true, force: true });
- rmSync(POOL, { recursive: true, force: true });
+ for (const p of [paths.transcriptsDir, PINNED.EXPORT_INDEX_DIR, POOL, path.join(ROOT, "media")]) {
+ rmSync(p, { recursive: true, force: true });
+ }
mkdirSync(paths.transcriptsDir, { recursive: true });
writeFileSync(paths.settingsFile, JSON.stringify({}));
writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), {
@@ -78,9 +154,15 @@ function resetCorpus(): void {
});
}
-// Metadata only — what a download leaves before any transcript exists.
-function seedVideo(id: string, metaIso = "2026-07-11T11:00:00Z"): void {
- const file = path.join(videoDir(id), "metadata.info.json");
+// Metadata only — what a download leaves before any transcript exists. With
+// `metaIso` null the file keeps its real mtime (now): "downloaded just now".
+function seedVideo(
+ id: string,
+ metaIso: string | null = "2026-07-11T11:00:00Z",
+ extra: Record<string, unknown> = {},
+ channel = CHANNEL,
+): void {
+ const file = path.join(videoDir(id, channel), "metadata.info.json");
writeJson(file, {
id,
title: `Video ${id}`,
@@ -90,15 +172,16 @@ function seedVideo(id: string, metaIso = "2026-07-11T11:00:00Z"): void {
description: "fixture",
webpage_url: `https://www.youtube.com/watch?v=${id}`,
extractor_key: "Youtube",
+ ...extra,
});
- touch(file, metaIso);
+ if (metaIso) touch(file, metaIso);
}
// YouTube's own captions, as a youtube-handled download writes them. parseVtt
// keeps only lines carrying inline timing tags (YouTube's rolling-caption
// shape), so the fixture has them.
-function addCaptions(id: string, iso: string): void {
- const file = path.join(videoDir(id), "transcript.en.vtt");
+function addCaptions(id: string, iso: string, channel = CHANNEL): void {
+ const file = path.join(videoDir(id, channel), "transcript.en.vtt");
writeFileSync(
file,
"WEBVTT\nKind: captions\nLanguage: en\n\n" +
@@ -165,7 +248,7 @@ test("(a) a transcript that arrives after the stat was cached reaches it on the
addWhisper("late", "2026-07-17T05:11:14Z");
await runIndex();
const second = await runStats();
- assert.equal(second.res.changed, 1, "the index's transcript record moved, so the stat is redone");
+ assert.equal(second.res.changed, 1, "the index's record moved, so the stat is redone");
const s = statOf(second.byId, "late");
assert.equal(s.hasTranscript, true);
assert.equal(s.cueCount, 2);
@@ -191,13 +274,15 @@ test("(b) stats built before the index had the video heal after the index build"
assert.equal(s.hasTranscript, true);
assert.equal(s.transcribedDate, "20260711");
- // And the lag is said, not silent: counted in the result and logged.
- assert.equal(first.res.unindexed, 1);
+ // And the lag is said, not silent: counted in the result and logged. (No
+ // index build has completed yet, so it is "not indexed yet".)
+ assert.equal(first.res.notIndexedYet, 1);
+ assert.equal(first.res.notIndexable, 0);
assert.ok(
- first.log.some((l) => l.startsWith("1 video(s) are not in the index yet")),
+ first.log.some((l) => l.startsWith("1 video(s) were downloaded after the last index build")),
first.log.join("\n"),
);
- assert.equal(second.res.unindexed, 0);
+ assert.equal(second.res.notIndexedYet, 0);
});
test("(c) a caption-only video is dated by its captions' arrival, not by a later Normalize", async () => {
@@ -257,33 +342,6 @@ test("(d) a Whisper video resolves exactly as before: the outcome sidecar, else
});
});
-// Every fs/promises call inside a video dir, by function name.
-function spyFs(names: ("readFile" | "stat" | "readdir" | "open")[]) {
- const calls: { fn: string; p: string }[] = [];
- // The CJS exports object: patching it and syncing is what reaches the named
- // ESM imports buildStats and its helpers hold.
- const mod = createRequire(import.meta.url)("node:fs/promises") as Record<
- string,
- (...a: unknown[]) => unknown
- >;
- const orig = new Map(names.map((n) => [n, mod[n]]));
- for (const n of names) {
- const fn = orig.get(n)!;
- mod[n] = (p: unknown, ...rest: unknown[]) => {
- calls.push({ fn: n, p: String(p) });
- return fn(p, ...rest);
- };
- }
- syncBuiltinESMExports();
- return {
- calls,
- restore() {
- for (const [n, fn] of orig) mod[n] = fn;
- syncBuiltinESMExports();
- },
- };
-}
-
test("(e) the key does not churn: a heal redoes one stat, then an unchanged run reads nothing per video", async () => {
resetCorpus();
seedVideo("w");
@@ -301,18 +359,16 @@ test("(e) the key does not churn: a heal redoes one stat, then an unchanged run
assert.equal(heal.res.added, 0);
assert.equal(heal.res.changed, 1, "only the video whose transcript arrived");
+ // An index build over an unchanged corpus rewrites nothing the key reads.
+ await runIndex();
const dataDir = path.join(paths.channelsDir, CHANNEL, "data");
- const spy = spyFs(["readFile", "stat", "readdir", "open"]);
- let steady;
- try {
- steady = await runStats();
- } finally {
- spy.restore();
- }
+ const from = fsCalls.length;
+ const steady = await runStats();
assert.equal(steady.res.added, 0);
assert.equal(steady.res.changed, 0);
- assert.equal(steady.res.unindexed, 0);
- const perVideo = spy.calls.filter((c) => c.p.startsWith(dataDir + path.sep));
+ assert.equal(steady.res.notIndexedYet + steady.res.notIndexable, 0);
+ // The spy sees node:fs and node:fs/promises, async and sync.
+ const perVideo = fsCalls.slice(from).filter((c) => c.p.startsWith(dataDir + path.sep));
assert.deepEqual(
perVideo.map((c) => `${c.fn} ${path.relative(dataDir, c.p)}`).sort(),
[
@@ -323,3 +379,200 @@ test("(e) the key does not churn: a heal redoes one stat, then an unchanged run
"the unchanged path stats each metadata file and touches nothing else in a video dir",
);
});
+
+test("(f) cues are read under the index's own key, even when the metadata's upload date moved", async () => {
+ resetCorpus();
+ seedVideo("moved", "2026-07-11T11:00:00Z");
+ addCaptions("moved", "2026-07-11T12:00:00Z");
+ const norm = await normalizeTranscript({ videoDir: videoDir("moved"), channelSlug: CHANNEL });
+ assert.equal(norm.status, "wrote");
+ touch(path.join(videoDir("moved"), "transcript.cues.json"), "2026-08-10T13:44:00Z");
+ // The metadata is rewritten with another upload date (a stream's date
+ // settled) but stays older than the normalized cues, so buildIndex still
+ // trusts transcript.cues.json — and keys the cues by ITS upload date.
+ seedVideo("moved", "2026-07-12T11:00:00Z", { upload_date: "20260602" });
+ await runIndex();
+ const { byId } = await runStats();
+ const s = statOf(byId, "moved");
+ assert.equal(s.uploadDate, "20260602");
+ assert.equal(s.hasTranscript, true, "the cues are found under the key the index used");
+ assert.equal(s.cueCount, 2);
+});
+
+test("(g) a re-index for another reason that changes the cues redoes the stat", async () => {
+ resetCorpus();
+ seedVideo("drift");
+ addCaptions("drift", "2026-07-11T12:00:00Z");
+ await runIndex();
+ const first = await runStats();
+ assert.equal(statOf(first.byId, "drift").cueCount, 2);
+
+ // A Normalize run writes transcript.cues.json (here with one cue fewer than
+ // the raw parse). buildIndex does not re-index for that alone...
+ await normalizeTranscript({ videoDir: videoDir("drift"), channelSlug: CHANNEL });
+ const cuesPath = path.join(videoDir("drift"), "transcript.cues.json");
+ const doc = JSON.parse(readFileSync(cuesPath, "utf8")) as { cues: unknown[] };
+ doc.cues = doc.cues.slice(0, 1);
+ writeFileSync(cuesPath, JSON.stringify(doc));
+ // ...but an availability recheck makes it re-process the video, and then it
+ // reads the fresher cues.json. The transcript mtime did not move.
+ writeJson(path.join(videoDir("drift"), "availability.json"), {
+ checkedAt: "2026-08-20T00:00:00.000Z",
+ availability: "public",
+ });
+ await runIndex();
+ const second = await runStats();
+ assert.equal(second.res.changed, 1);
+ assert.equal(statOf(second.byId, "drift").cueCount, 1);
+});
+
+test("(h) a video the index skipped is not announced as pending on every run", async () => {
+ resetCorpus();
+ seedVideo("fine");
+ // No upload_date: buildIndex skips it (it needs one for the index key).
+ seedVideo("undated", "2026-07-11T11:00:00Z", { upload_date: undefined });
+ await runIndex();
+ // Downloaded after that index build.
+ seedVideo("fresh", null);
+ for (let run = 0; run < 2; run++) {
+ const { res, log } = await runStats();
+ assert.equal(res.notIndexedYet, 1, `run ${run}`);
+ assert.equal(res.notIndexable, 1, `run ${run}`);
+ assert.ok(log.some((l) => l.startsWith("1 video(s) were downloaded after the last index build")), log.join("\n"));
+ assert.ok(log.some((l) => l.startsWith("1 video(s) are not in the index although its last build saw them")), log.join("\n"));
+ }
+ await runIndex();
+ const { res } = await runStats();
+ assert.equal(res.notIndexedYet, 0, "the next index build takes the fresh one");
+ assert.equal(res.notIndexable, 1, "the undated one stays, and is said as such");
+});
+
+// A second channel whose data/ is a relocated symlink, the way the editor's
+// Storage panel leaves it: channels/<slug>/data -> <root>/<slug>/data, with
+// config.dataDir recording the target.
+function seedDriveChannel(): { target: string } {
+ const target = path.join(ROOT, "media", DRIVE_CHANNEL, "data");
+ mkdirSync(target, { recursive: true });
+ writeJson(path.join(paths.channelsDir, DRIVE_CHANNEL, "config.json"), {
+ handling: "youtube",
+ name: "Drive Channel",
+ url: "https://www.youtube.com/@drive/videos",
+ dataDir: target,
+ });
+ symlinkSync(target, path.join(paths.channelsDir, DRIVE_CHANNEL, "data"));
+ for (const id of ["d1", "d2"]) {
+ seedVideo(id, "2026-07-11T11:00:00Z", {}, DRIVE_CHANNEL);
+ addCaptions(id, "2026-07-11T12:00:00Z", DRIVE_CHANNEL);
+ }
+ return { target };
+}
+
+test("(i) an unmounted media drive keeps its channel's stats; a cache clear refuses", async () => {
+ resetCorpus();
+ seedVideo("local");
+ seedDriveChannel();
+ await runIndex();
+ const mounted = await runStats();
+ assert.equal(statOf(mounted.byId, "d1").hasTranscript, true);
+
+ // Unmount: the link now dangles, exactly as an absent USB drive leaves it.
+ const media = path.join(ROOT, "media");
+ renameSync(media, `${media}-away`);
+ const log: string[] = [];
+ const away = await runStats(log);
+ assert.deepEqual(away.res.heldChannels, [DRIVE_CHANNEL]);
+ assert.equal(away.res.removed, 0, "not read as a channel with no videos");
+ for (const id of ["d1", "d2", "local"]) assert.ok(away.byId.has(id), `${id} still published`);
+ assert.equal(statOf(away.byId, "d1").hasTranscript, true);
+ assert.ok(
+ log.some((l) => l.startsWith(`Channel ${DRIVE_CHANNEL}: media not reachable`) && l.includes("its 2 cached stat(s) are kept")),
+ log.join("\n"),
+ );
+
+ // A schema change needs the whole cache rebuilt, which cannot include a
+ // channel it cannot read: refuse, and leave the cache as it is.
+ setStoredSchema(STATS_SCHEMA_VERSION - 1);
+ await assert.rejects(runStats(), /must be rebuilt .* cannot be read: drive-channel/);
+ assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION - 1);
+ assert.equal(countStats(), 3);
+
+ renameSync(`${media}-away`, media);
+ const back = await runStats();
+ assert.deepEqual(back.res.heldChannels, []);
+ assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION);
+ assert.equal(back.byId.size, 3);
+});
+
+// Direct access to the temp LMDB's stats cache, for the schema cases.
+function withDb<T>(fn: (dbs: { meta: ReturnType<ReturnType<typeof open>["openDB"]>; stats: ReturnType<ReturnType<typeof open>["openDB"]> }) => T): T {
+ const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true });
+ try {
+ return fn({
+ meta: root.openDB({ name: "statsMeta", encoding: "msgpack" }),
+ stats: root.openDB({ name: "statsByPath", encoding: "msgpack" }),
+ });
+ } finally {
+ root.close();
+ }
+}
+const setStoredSchema = (v: number) =>
+ withDb(({ meta }) => meta.putSync("schema", v));
+const readStoredSchema = () => withDb(({ meta }) => meta.get("schema"));
+const countStats = () => withDb(({ stats }) => [...stats.getKeys()].length);
+
+test("(j) the schema guard: an older cache is cleared, a newer one is refused unless overridden", async () => {
+ resetCorpus();
+ seedVideo("v1");
+ addCaptions("v1", "2026-07-11T12:00:00Z");
+ await runIndex();
+ await runStats();
+ assert.equal(countStats(), 1);
+
+ // Older: cleared and rebuilt, as every schema bump has always done.
+ setStoredSchema(STATS_SCHEMA_VERSION - 1);
+ const log: string[] = [];
+ const older = await runStats(log);
+ assert.ok(log.some((l) => l.includes("clearing stats cache")), log.join("\n"));
+ assert.equal(older.res.added, 1);
+ assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION);
+
+ // Newer: refused, naming both versions and the override; nothing touched.
+ setStoredSchema(STATS_SCHEMA_VERSION + 1);
+ await assert.rejects(
+ runStats(),
+ new RegExp(
+ `newer build \\(stats schema ${STATS_SCHEMA_VERSION + 1}; this build's is ${STATS_SCHEMA_VERSION}\\).*${STATS_DOWNGRADE_ENV}=1`,
+ ),
+ );
+ assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION + 1);
+ assert.equal(countStats(), 1);
+
+ // The CLI exits non-zero on it.
+ const cli = spawnSync(
+ path.join(COMMON, "node_modules", ".bin", "tsx"),
+ ["bin/archilyzer.ts", "build", "stats"],
+ { cwd: COMMON, env: { ...process.env }, encoding: "utf8" },
+ );
+ assert.notEqual(cli.status, 0, cli.stdout + cli.stderr);
+ assert.match(cli.stderr, /written by a newer build/);
+ assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION + 1);
+ assert.equal(countStats(), 1);
+
+ // A deliberate rollback, overridden: cleared and rebuilt at this version.
+ process.env[STATS_DOWNGRADE_ENV] = "1";
+ try {
+ const rolled = await runStats();
+ assert.equal(rolled.res.added, 1);
+ assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION);
+ } finally {
+ delete process.env[STATS_DOWNGRADE_ENV];
+ }
+});
+
+test("(z) no write this file caused landed outside its temp root", () => {
+ const outside = fsCalls.filter(
+ (c) => c.write && c.p !== ROOT && !c.p.startsWith(ROOT + path.sep),
+ );
+ assert.deepEqual(outside, []);
+ assert.ok(fsCalls.some((c) => c.write), "the spy saw the writes");
+});
diff --git a/common/controller/buildStats.ts b/common/controller/buildStats.ts
@@ -13,12 +13,23 @@
// stands, which the cache key below makes safe.
//
// THE CACHE KEY IS TWO THINGS: the metadata file's mtime AND the index's own
-// per-video transcript record (buildIndex's `mtimes.transcriptMs`, or
+// per-video record (buildIndex's `mtimes` entry, as indexSignature, or
// NOT_INDEXED). Until schema 6 it was the metadata mtime alone, while
// hasTranscript / cueCount / coverage / transcribedDate come from the index and
// the transcript files — so a transcript that arrived after a video was first
// seen (Whisper days later, or a stats run before build:index had the video)
// never reached its stat, and a whole channel could publish as untranscribed.
+//
+// A CHANNEL WHOSE MEDIA IS NOT REACHABLE (a relocated `data/` on an unmounted
+// drive, or one mid-relocation) is not rescanned: its cached stats are kept as
+// they are, rather than read as a channel with no videos and removed. This is
+// the stats build's own guard — the job registry's `needsMedia` check is per
+// channel, and this build is pool-wide — so the editor job and the CLI share it.
+//
+// ONE STATS BUILD AT A TIME. Nothing here takes a lock: two runs at once are
+// harmless unless one of them clears the cache (a schema change) after the
+// other scanned, when the other can publish truncated pages. The editor's build
+// jobs share the "build" queue by default; the CLI is outside every queue.
import path from "node:path";
import { createHash } from "node:crypto";
@@ -52,9 +63,11 @@ import {
import type { VideoStatus } from "../lib/stats";
import type { ChannelConfig } from "../lib/channelConfig";
import { readChannelConfigFile } from "./channels";
+import { inspectChannelMedia } from "../lib/channelMedia";
import type { Paths } from "../lib/paths";
import { listSites, siteStatsDir } from "../lib/site";
import {
+ INDEX_SCANNED_AT_KEY,
STATS_SCHEMA_VERSION,
STATS_MANIFEST_VERSION,
STATS_MAX_PAGE_BYTES,
@@ -67,15 +80,55 @@ import {
type IndexKey = [string, string, string];
type PathKey = [string, string];
-// `idxMs` is the index's per-video transcript record as this stat saw it — see
-// indexTranscriptMs. Optional only because a record written before schema 6
-// has none; the schema bump clears those, and a missing value compares
-// unequal to every real one, so such a record would be recomputed anyway.
-type StatsRecord = { metaMs: number; idxMs?: number | null; stat: VideoStat };
+// `idx` is the index's per-video record as this stat saw it (indexSignature).
+// Optional only because a record written before schema 6 has none; the schema
+// bump clears those, and a missing value compares unequal to every real one, so
+// such a record would be recomputed anyway.
+type StatsRecord = { metaMs: number; idx?: string; stat: VideoStat };
+
+// The part of buildIndex's `mtimes` record this cache reads: every input whose
+// change makes buildIndex re-process the video (and so rewrite its cues), and
+// the key it stored the cues under.
+type IndexRecord = {
+ metaMs: number;
+ transcriptMs: number | null;
+ subsMs?: number | null;
+ availabilityMs?: number | null;
+ digestMs?: number | null;
+ indexKey: IndexKey;
+};
+
+// buildIndex has no `mtimes` record for this video. See the two counts below.
+const NOT_INDEXED = "-";
+
+// The whole index record as one comparable string. Keying on all of it (not
+// the transcript mtime alone) means ANY re-index redoes the stat: a re-index
+// for another reason can read a fresher transcript.cues.json and change the cue
+// count, and a moved index key moves the cues. Numbers print as their shortest
+// round-trip form, so an unchanged record gives an identical string.
+function indexSignature(r: IndexRecord | undefined): string {
+ if (!r) return NOT_INDEXED;
+ const n = (v: number | null | undefined) => (v == null ? "" : String(v));
+ return [
+ n(r.metaMs),
+ n(r.transcriptMs),
+ n(r.subsMs),
+ n(r.availabilityMs),
+ n(r.digestMs),
+ ...r.indexKey,
+ ].join("\u0000");
+}
-// buildIndex has no `mtimes` record for this video yet: it was downloaded after
-// the last index build. Distinct from `null` (indexed, no transcript file).
-const NOT_INDEXED = -1;
+// Set to 1 (or true/yes/on) to let an OLDER build clear a stats cache a newer
+// one wrote — a deliberate rollback. Declared in lib/envVars.ts.
+export const STATS_DOWNGRADE_ENV = "ARCHILYZER_STATS_ALLOW_DOWNGRADE";
+const TRUTHY = new Set(["1", "true", "yes", "on"]);
+function allowsStatsDowngrade(
+ env: Record<string, string | undefined> = process.env,
+): boolean {
+ const raw = env.ARCHILYZER_STATS_ALLOW_DOWNGRADE;
+ return typeof raw === "string" && TRUTHY.has(raw.trim().toLowerCase());
+}
type ScanEntry = {
channelSlug: string;
@@ -92,10 +145,15 @@ export type BuildStatsResult = {
added: number;
changed: number;
removed: number;
- // Videos on disk that the index does not have yet. Their stats say "no
- // transcript" until the first run after the next index build, which
- // recomputes them (the key moves from NOT_INDEXED).
- unindexed: number;
+ // Videos on disk with no index record, in two kinds. `notIndexedYet`: their
+ // metadata is newer than the last completed index build's scan — downloaded
+ // since — and the first stats run after the next index build redoes them.
+ // `notIndexable`: the last index build saw them and did not index them (no
+ // upload_date, or it failed on them); nothing here will change that.
+ notIndexedYet: number;
+ notIndexable: number;
+ // Channels whose media was not reachable, so their cached stats were kept.
+ heldChannels: string[];
pagesWritten: number;
shortCircuited: boolean;
durationMs: number;
@@ -184,17 +242,24 @@ async function resolveAcquisitionDates(
return { downloadedDate, transcribedDate };
}
+// `held` maps each channel whose media is not reachable to the reason: it is
+// not scanned, and the caller keeps its cached stats (see the file header).
async function scanSource(
channelsDir: string,
log: (msg: string) => void,
-): Promise<{ entries: ScanEntry[]; channels: Map<string, ChannelConfig> }> {
+): Promise<{
+ entries: ScanEntry[];
+ channels: Map<string, ChannelConfig>;
+ held: Map<string, string>;
+}> {
const channels = new Map<string, ChannelConfig>();
const entries: ScanEntry[] = [];
+ const held = new Map<string, string>();
let channelEntries: Dirent[];
try {
channelEntries = await readdir(channelsDir, { withFileTypes: true });
} catch {
- return { entries, channels };
+ return { entries, channels, held };
}
for (const ch of channelEntries) {
if (!ch.isDirectory()) continue;
@@ -206,6 +271,13 @@ async function scanSource(
}
if (cfg.excludeFromBuild) continue;
channels.set(ch.name, cfg);
+ // An unmounted drive is not an empty channel (lib/channelMedia.ts): the
+ // readdir below would fail and every one of its stats would be removed.
+ const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg);
+ if (media.status !== "ok" && media.status !== "in-place") {
+ held.set(ch.name, media.detail ?? media.status);
+ continue;
+ }
const dataDir = path.join(channelDir, "data");
let videoEntries: Dirent[];
try {
@@ -231,7 +303,7 @@ async function scanSource(
});
}
}
- return { entries, channels };
+ return { entries, channels, held };
}
// Buffered byte-capped page writer. Stats records are small, so a page fits in
@@ -310,25 +382,51 @@ export async function buildStats({
encoding: "msgpack",
});
const meta = root.openDB<unknown, string>({ name: "statsMeta", encoding: "msgpack" });
- // Read-only view of buildIndex's per-video mtime record, keyed like
- // statsByPath. Only `transcriptMs` is read: the mtime of the transcript file
- // the index took this video's cues from (null when it had none). buildIndex
- // rewrites a video's `cues` when that number moves (buildIndex.ts, the
- // added/changed diff), so it is exactly the "has the transcript this stat was
- // computed from changed" signal — at the cost of one LMDB get per video, and
- // no file I/O, on the unchanged path.
- const indexMtimes = root.openDB<{ transcriptMs: number | null }, PathKey>({
+ // Read-only views of buildIndex's per-video `mtimes` record (keyed like
+ // statsByPath: the second half of this cache's key, and the key the video's
+ // cues were stored under) and of its `meta` (when its last scan began). One
+ // LMDB get per video, and no file I/O, on the unchanged path.
+ const indexMtimes = root.openDB<IndexRecord, PathKey>({
name: "mtimes",
encoding: "msgpack",
});
- const indexTranscriptMs = (k: PathKey): number | null => {
- const rec = indexMtimes.get(k);
- return rec ? (rec.transcriptMs ?? null) : NOT_INDEXED;
- };
+ const indexMeta = root.openDB<unknown, string>({ name: "meta", encoding: "msgpack" });
+ // NEVER CLEAR A CACHE A NEWER BUILD WROTE. An old build running beside a
+ // new one (an editor not yet restarted onto the new code) would otherwise
+ // clear it, refill it the old way, and the next new run would clear it back:
+ // a full pass each time, and old-logic numbers published in between.
const storedSchema = meta.get("schema") as number | undefined;
+ if (
+ typeof storedSchema === "number" &&
+ storedSchema > STATS_SCHEMA_VERSION &&
+ !allowsStatsDowngrade()
+ ) {
+ await root.close();
+ throw new Error(
+ `The stats cache was written by a newer build (stats schema ${storedSchema}; this build's is ${STATS_SCHEMA_VERSION}). ` +
+ `Refusing to clear it: rebuild and restart onto the current code. ` +
+ `For a deliberate rollback, set ${STATS_DOWNGRADE_ENV}=1.`,
+ );
+ }
+
+ const { entries, channels, held } = await scanSource(paths.channelsDir, log);
+ log(`Scanned ${entries.length} videos across ${channels.size} channels.`);
+
const schemaBumped = storedSchema !== STATS_SCHEMA_VERSION;
if (schemaBumped) {
+ // A clear with a channel's media unreachable would drop that channel's
+ // stats for good (it cannot be rescanned), and the pages built from this
+ // run would publish it as empty. Refuse instead.
+ if (held.size > 0) {
+ await root.close();
+ throw new Error(
+ `The stats cache must be rebuilt (stats schema ${storedSchema ?? "<none>"} -> ${STATS_SCHEMA_VERSION}), ` +
+ `but ${held.size} channel(s) cannot be read: ` +
+ [...held].map(([slug, why]) => `${slug} (${why})`).join("; ") +
+ `. Mount their media (see /storage) and run it again.`,
+ );
+ }
log(
`Stats schema change (${storedSchema ?? "<none>"} -> ${STATS_SCHEMA_VERSION}); clearing stats cache.`,
);
@@ -336,46 +434,65 @@ export async function buildStats({
await meta.put("schema", STATS_SCHEMA_VERSION);
}
- const { entries, channels } = await scanSource(paths.channelsDir, log);
- log(`Scanned ${entries.length} videos across ${channels.size} channels.`);
-
const liveIds = new Set(
entries.map((e) => pathKeyId([e.channelSlug, e.videoDir])),
);
- const toProcess: { e: ScanEntry; idxMs: number | null }[] = [];
+ const scannedAt = indexMeta.get(INDEX_SCANNED_AT_KEY) as number | undefined;
+ const toProcess: { e: ScanEntry; idx: string; indexKey?: IndexKey }[] = [];
let added = 0;
let changed = 0;
- let unindexed = 0;
+ let notIndexedYet = 0;
+ let notIndexable = 0;
for (const e of entries) {
const pk: PathKey = [e.channelSlug, e.videoDir];
- const idxMs = indexTranscriptMs(pk);
- if (idxMs === NOT_INDEXED) unindexed++;
+ const rec = indexMtimes.get(pk);
+ const idx = indexSignature(rec);
+ if (idx === NOT_INDEXED) {
+ if (typeof scannedAt !== "number" || e.metaMs > scannedAt) notIndexedYet++;
+ else notIndexable++;
+ }
const prev = statsByPath.get(pk);
if (!prev) {
added++;
- toProcess.push({ e, idxMs });
- } else if (prev.metaMs !== e.metaMs || prev.idxMs !== idxMs) {
+ toProcess.push({ e, idx, indexKey: rec?.indexKey });
+ } else if (prev.metaMs !== e.metaMs || prev.idx !== idx) {
// The second half is the fix for stats frozen at first sight: a
- // transcript that arrives later moves transcriptMs (and a video first
- // seen before the index had it moves off NOT_INDEXED), while the
- // metadata file — the old key's only input — is never touched.
+ // transcript that arrives later makes buildIndex re-process the video
+ // (and a video first seen before the index had it moves off
+ // NOT_INDEXED), while the metadata file — the old key's only input — is
+ // never touched.
changed++;
- toProcess.push({ e, idxMs });
+ toProcess.push({ e, idx, indexKey: rec?.indexKey });
}
}
- if (unindexed > 0) {
- // Not a failure: the pool composers read the index as it stands, and the
- // editor downloads between index builds. Said so a published number that
- // lags the disk has its reason in the log.
+ // Neither is a failure of this build, and both are said, so a published
+ // number that lags the disk has its reason in the log. Only the first kind
+ // resolves itself.
+ if (notIndexedYet > 0) {
+ log(
+ `${notIndexedYet} video(s) were downloaded after the last index build; their transcripts reach the stats on the first run after the next one.`,
+ );
+ }
+ if (notIndexable > 0) {
log(
- `${unindexed} video(s) are not in the index yet; their transcripts reach the stats on the first run after the next index build.`,
+ `${notIndexable} video(s) are not in the index although its last build saw them (no upload_date, or it failed on them: see that build's log). Their stats show no transcript until that is fixed.`,
);
}
const removedKeys: PathKey[] = [];
+ const keptHeld = new Map<string, number>();
for (const { key } of statsByPath.getRange()) {
const k = key as PathKey;
+ if (held.has(k[0])) {
+ keptHeld.set(k[0], (keptHeld.get(k[0]) ?? 0) + 1);
+ continue;
+ }
if (!liveIds.has(pathKeyId(k))) removedKeys.push(k);
}
+ for (const [slug, why] of held) {
+ log(
+ `Channel ${slug}: media not reachable (${why}); its ${keptHeld.get(slug) ?? 0} cached stat(s) are kept as they are, not rescanned.`,
+ );
+ }
const removed = removedKeys.length;
const BATCH = 200;
@@ -383,7 +500,7 @@ export async function buildStats({
signal?.throwIfAborted();
const slice = toProcess.slice(i, i + BATCH);
await Promise.all(
- slice.map(async ({ e, idxMs }) => {
+ slice.map(async ({ e, idx, indexKey: recordedKey }) => {
try {
const metaRaw = await readFile(e.metaPath, "utf8");
const parsedMeta = JSON.parse(metaRaw) as RawMetadata;
@@ -394,7 +511,12 @@ export async function buildStats({
e.configName,
);
if (!base.uploadDate) return;
- const indexKey: IndexKey = [base.uploadDate, e.channelSlug, base.id];
+ // The key buildIndex stored this video's cues under, when it has a
+ // record: its uploadDate comes from transcript.cues.json when that
+ // is fresh, which the metadata's can differ from. The computed key
+ // is only a fallback for a video the index does not have.
+ const indexKey: IndexKey =
+ recordedKey ?? [base.uploadDate, e.channelSlug, base.id];
const cueList = cues.get(indexKey);
const cueCount = cueList ? cueList.length : null;
const coverage = transcriptCoverage(cueList, base.duration).coverage;
@@ -429,7 +551,7 @@ export async function buildStats({
);
await statsByPath.put([e.channelSlug, e.videoDir], {
metaMs: e.metaMs,
- idxMs,
+ idx,
stat,
});
} catch (err) {
@@ -632,7 +754,9 @@ export async function buildStats({
added,
changed,
removed,
- unindexed,
+ notIndexedYet,
+ notIndexable,
+ heldChannels: [...held.keys()],
pagesWritten: aggregatePages,
shortCircuited: needBuild.length === 0,
durationMs: Date.now() - t0,
diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts
@@ -111,6 +111,7 @@ const DECLARED: EnvVarDecl[] = [
{ name: "TRANSCRIPT_PLATFORM_LINKS", audience: "runtime", default: "off", readBy: "common/lib/archive/reader-fs.ts", doc: "`1` cites platform watch pages instead of the archive's own pages." },
{ name: "AUDIO_CHECK_RESUME_DURING_PROBE", audience: "runtime", default: "the channel's `audioCheck.resumeDuringProbe`", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "`1` or `true` resumes yt-dlp during the audio check's probe, anything else holds it, for a one-off comparison run; unset = the channel's setting." },
{ name: "AUDIO_CHECK_BACKOFF_FACTOR", audience: "runtime", default: "the built-in factor", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "The audio check's interval backoff factor, in (0, 1], for a one-off run." },
+ { name: "ARCHILYZER_STATS_ALLOW_DOWNGRADE", audience: "runtime", default: "off", readBy: "common/controller/buildStats.ts", doc: "`1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions." },
{ name: "MCP_IO_STATS", audience: "runtime", default: "off", readBy: "common/lib/archive/io-stats.ts", doc: "`1` turns on per-call I/O accounting, for `mcp/bench`." },
{ name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool", doc: "Which editor `pnpm ops` and the MCP's `fetch_clip` talk to." },
{ name: "ARCHILYZER_AGENT", audience: "runtime", default: "`cli`", readBy: "scripts/archilyzer-ops.mjs", doc: "Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`." },