commit 2e6144647711fae5b94c9d2343c795fcc24113f9
parent bc5b8f9713050b4eb9f5974e2852655e31e21d5c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 28 Sep 2026 19:12:36 -0400
Merge fix/stats-cache-key — a video's stats follow its transcript: the stats cache is keyed on the index's record of each video as well as its metadata, a transcript always has a date, and a site's card counts every transcript; a stats build refuses while a channel's media is unreachable and refuses to clear a cache written by a newer build; reviewed SHIP; the recount is the operator's (plans/stats-cache-key.md)
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
16 files changed, 1466 insertions(+), 43 deletions(-)
diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md
@@ -75,6 +75,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not
| `TRANSCRIPT_PLATFORM_LINKS` | off | `1` cites platform watch pages instead of the archive's own pages. | common/lib/archive/reader-fs.ts |
| `AUDIO_CHECK_RESUME_DURING_PROBE` | the channel's `audioCheck.resumeDuringProbe` | `1` or `true` resumes yt-dlp during the audio check's probe, anything else holds it, for a one-off comparison run; unset = the channel's setting. | common/ytdlp/audioCheckedDownload.ts |
| `AUDIO_CHECK_BACKOFF_FACTOR` | the built-in factor | The audio check's interval backoff factor, in (0, 1], for a one-off run. | common/ytdlp/audioCheckedDownload.ts |
+| `ARCHILYZER_STATS_ALLOW_DOWNGRADE` | off | `1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions. | common/controller/buildStats.ts |
| `MCP_IO_STATS` | off | `1` turns on per-call I/O accounting, for `mcp/bench`. | common/lib/archive/io-stats.ts |
| `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's `fetch_clip` talk to. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool |
| `ARCHILYZER_AGENT` | `cli` | Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`. | scripts/archilyzer-ops.mjs |
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -64,6 +64,7 @@ import {
pageFileName,
} from "../lib/manifest";
import { getSettings } from "../lib/settings";
+import { INDEX_SCANNED_AT_KEY } from "../lib/stats";
import {
listSites,
siteDigestsDir,
@@ -556,6 +557,11 @@ export async function buildIndex({
await meta.put("schema", SCHEMA_VERSION);
}
+ // Recorded as INDEX_SCANNED_AT_KEY only when this build completes, so
+ // buildStats can tell apart a video with no `mtimes` record: metadata newer
+ // than this is "not indexed yet"; older, and this build saw it and skipped it
+ // (no upload_date, or processing failed).
+ const scanStartedAt = Date.now();
const { live, channels: channelConfigs } = await scanSource(channelsDir, log);
const livePathIds = new Set<string>();
const liveByPathId = new Map<string, LiveEntry>();
@@ -1997,6 +2003,7 @@ export async function buildIndex({
}
}
for (const k of staleFpKeys) meta.remove(k);
+ await meta.put(INDEX_SCANNED_AT_KEY, scanStartedAt);
await meta.flushed;
await root.close();
diff --git a/common/controller/buildStats.test.ts b/common/controller/buildStats.test.ts
@@ -0,0 +1,605 @@
+// Integration: the stats cache, through the REAL buildIndex and buildStats, over
+// a temp corpus.
+//
+// The stats cache (`statsByPath`) used to be keyed on metadata.info.json's
+// mtime alone, while hasTranscript / cueCount / coverage / transcribedDate come
+// from the index and the transcript files. So a transcript that arrived after a
+// video was first seen never reached its stat, and a caption video (a VTT, no
+// transcript.json, no outcome sidecar) could never be dated at all. Measured on
+// a real corpus: one site served 1,889 videos and the homepage said 0
+// transcripts, 0 channels, 0 hours. These cases pin the fix: the key also holds
+// the index's own record for the video, and a transcript always has a date.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/buildStats.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { spawnSync } from "node:child_process";
+import { createRequire, syncBuiltinESMExports } from "node:module";
+import {
+ mkdirSync,
+ mkdtempSync,
+ readFileSync,
+ renameSync,
+ rmSync,
+ symlinkSync,
+ utimesSync,
+ writeFileSync,
+} from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+
+// EVERY PATH getPaths() CAN RESOLVE TO A PLACE THIS FILE'S CODE MAY WRITE IS
+// PINNED UNDER ROOT, before anything calls it (it is lazy and cached — the
+// maybeMissingBuild.test.ts pattern). From common/lib/paths.ts: TRANSCRIPTS_DIR
+// (the LMDB, channels, jobs), SAVED_VIDEOS_DIR, SITES_DIR, SETTINGS_FILE,
+// EXPORT_PUBLIC_DIR, EXPORT_INDEX_DIR, EXPORT_BUILDS_DIR, EDITOR_CHANGELOG_FILE,
+// EXPORT_CHANGELOG_FILE, CHARTS_CONFIG_FILE, SEARCH_ALIASES_FILE,
+// CURATED_TAGS_FILE, ARCHILYZER_CONFIG_DIR, ARCHILYZER_SOURCE_SCRATCH. The rest
+// of its variables name binaries and a URL, which nothing here runs. The last
+// test proves no write this file caused landed outside ROOT.
+const ROOT = mkdtempSync(path.join(tmpdir(), "build-stats-"));
+const PINNED: Record<string, string> = {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"),
+ EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"),
+ CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"),
+ SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"),
+ CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+ ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"),
+};
+Object.assign(process.env, PINNED);
+delete process.env.ARCHILYZER_STATS_ALLOW_DOWNGRADE;
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("./buildIndex");
+const { buildStats, STATS_DOWNGRADE_ENV } = await import("./buildStats");
+const { normalizeTranscript } = await import("./normalizeTranscript");
+const { readStatsPages } = await import("./poolSummary");
+const { STATS_SCHEMA_VERSION } = await import("../lib/stats");
+const { open } = await import("lmdb");
+
+const paths = getPaths();
+const CHANNEL = "test-channel";
+const DRIVE_CHANNEL = "drive-channel";
+const SITE = "testsite";
+const POOL = path.join(ROOT, "pool-stats");
+const COMMON = fileURLToPath(new URL("..", import.meta.url));
+
+// ── an fs spy over the whole file ───────────────────────────────────────────
+// Wraps the node:fs and node:fs/promises functions on their CJS exports objects
+// and syncs them into the named ESM imports the code under test holds. Records
+// every call's path(s), and whether it writes. Case (e) reads the reads; the
+// last case reads the writes.
+type FsCall = { fn: string; p: string; write: boolean };
+const fsCalls: FsCall[] = [];
+{
+ const req = createRequire(import.meta.url);
+ const fsCjs = req("node:fs") as Record<string, unknown>;
+ const fspCjs = req("node:fs/promises") as Record<string, unknown>;
+ const READS = ["readFile", "stat", "lstat", "readdir", "readlink"];
+ const WRITES = ["writeFile", "appendFile", "rename", "mkdir", "rm", "rmdir", "unlink", "copyFile", "cp", "symlink", "link", "utimes", "truncate", "mkdtemp"];
+ const TWO_PATHS = new Set(["rename", "copyFile", "cp", "symlink", "link"]);
+ const opensForWrite = (flags: unknown) =>
+ (typeof flags === "string" && /[wa+]/.test(flags)) ||
+ (typeof flags === "number" && (flags & 3) !== 0);
+ const asPath = (v: unknown) =>
+ typeof v === "string" ? v : v instanceof URL ? fileURLToPath(v) : Buffer.isBuffer(v) ? v.toString() : null;
+ const wrap = (mod: Record<string, unknown>, name: string, write: boolean | "open") => {
+ const fn = mod[name];
+ if (typeof fn !== "function") return;
+ mod[name] = function (this: unknown, ...args: unknown[]) {
+ const isWrite = write === "open" ? opensForWrite(args[1]) : write;
+ const paths = TWO_PATHS.has(name.replace(/Sync$/, "")) ? [args[0], args[1]] : [args[0]];
+ for (const a of paths) {
+ const p = asPath(a);
+ if (p !== null) fsCalls.push({ fn: name, p: path.resolve(p), write: isWrite });
+ }
+ return (fn as (...a: unknown[]) => unknown).apply(this, args);
+ };
+ };
+ for (const n of READS) wrap(fspCjs, n, false);
+ for (const n of WRITES) {
+ wrap(fspCjs, n, true);
+ wrap(fsCjs, n, true);
+ wrap(fsCjs, `${n}Sync`, true);
+ }
+ wrap(fspCjs, "open", "open");
+ wrap(fsCjs, "open", "open");
+ wrap(fsCjs, "openSync", "open");
+ wrap(fsCjs, "createWriteStream", true);
+ syncBuiltinESMExports();
+}
+
+const at = (iso: string) => new Date(iso);
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const videoDir = (id: string, channel = CHANNEL) =>
+ path.join(paths.channelsDir, channel, "data", id);
+const touch = (file: string, iso: string) => utimesSync(file, at(iso), at(iso));
+
+// A fresh corpus (and a fresh LMDB) per test: every count below is exact.
+function resetCorpus(): void {
+ for (const p of [paths.transcriptsDir, PINNED.EXPORT_INDEX_DIR, POOL, path.join(ROOT, "media")]) {
+ rmSync(p, { recursive: true, force: true });
+ }
+ mkdirSync(paths.transcriptsDir, { recursive: true });
+ writeFileSync(paths.settingsFile, JSON.stringify({}));
+ writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), {
+ handling: "youtube",
+ name: "Test Channel",
+ url: "https://www.youtube.com/@example/videos",
+ });
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [{ slug: CHANNEL, groupId: "default" }],
+ });
+}
+
+// Metadata only — what a download leaves before any transcript exists. With
+// `metaIso` null the file keeps its real mtime (now): "downloaded just now".
+function seedVideo(
+ id: string,
+ metaIso: string | null = "2026-07-11T11:00:00Z",
+ extra: Record<string, unknown> = {},
+ channel = CHANNEL,
+): void {
+ const file = path.join(videoDir(id, channel), "metadata.info.json");
+ writeJson(file, {
+ id,
+ title: `Video ${id}`,
+ channel: "Test Channel",
+ upload_date: "20260601",
+ duration: 120,
+ description: "fixture",
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ ...extra,
+ });
+ if (metaIso) touch(file, metaIso);
+}
+
+// YouTube's own captions, as a youtube-handled download writes them. parseVtt
+// keeps only lines carrying inline timing tags (YouTube's rolling-caption
+// shape), so the fixture has them.
+function addCaptions(id: string, iso: string, channel = CHANNEL): void {
+ const file = path.join(videoDir(id, channel), "transcript.en.vtt");
+ writeFileSync(
+ file,
+ "WEBVTT\nKind: captions\nLanguage: en\n\n" +
+ "00:00:00.000 --> 00:00:05.000 align:start position:0%\n" +
+ "First<00:00:01.000><c> caption</c><00:00:02.000><c> line.</c>\n\n" +
+ "00:01:00.000 --> 00:01:50.000 align:start position:0%\n" +
+ "Second<00:01:10.000><c> caption</c><00:01:20.000><c> line.</c>\n",
+ );
+ touch(file, iso);
+}
+
+// A Whisper-family run: transcript.json, and (unless `outcome` is false) the
+// transcribe-outcome.json sidecar every run writes beside it.
+function addWhisper(id: string, iso: string, outcome = true): void {
+ const file = path.join(videoDir(id), "transcript.json");
+ writeJson(file, {
+ duration_seconds: 120,
+ chunks: 2,
+ text: "one two",
+ chunk_data: [
+ { start_time: 0, end_time: 5, text: "Spoken line one." },
+ { start_time: 60, end_time: 110, text: "Spoken line two." },
+ ],
+ });
+ touch(file, iso);
+ if (outcome) {
+ writeJson(path.join(videoDir(id), "transcribe-outcome.json"), {
+ videoId: id,
+ transcribedAt: iso,
+ });
+ }
+}
+
+type Stat = Awaited<ReturnType<typeof readStatsPages>>[number];
+
+async function runStats(log: string[] = []) {
+ const res = await buildStats({
+ paths,
+ onLog: (s) => log.push(s),
+ wholePoolStatsDir: POOL,
+ });
+ const byId = new Map<string, Stat>(
+ (await readStatsPages(POOL)).map((s) => [s.id, s]),
+ );
+ return { res, byId, log };
+}
+
+const runIndex = () => buildIndex({ paths, onLog: () => {} });
+
+function statOf(byId: Map<string, Stat>, id: string): Stat {
+ const s = byId.get(id);
+ assert.ok(s, `${id} has a stat`);
+ return s;
+}
+
+test("(a) a transcript that arrives after the stat was cached reaches it on the next run", async () => {
+ resetCorpus();
+ seedVideo("late");
+ await runIndex();
+ const first = await runStats();
+ assert.equal(statOf(first.byId, "late").hasTranscript, false);
+
+ // Whisper runs days later. metadata.info.json — the old key — is untouched.
+ addWhisper("late", "2026-07-17T05:11:14Z");
+ await runIndex();
+ const second = await runStats();
+ assert.equal(second.res.changed, 1, "the index's record moved, so the stat is redone");
+ const s = statOf(second.byId, "late");
+ assert.equal(s.hasTranscript, true);
+ assert.equal(s.cueCount, 2);
+ assert.equal(s.transcribedDate, "20260717");
+ // The MCP's "covers only N% — truncated" note reads this field. A stale
+ // record said 0 for a complete transcript.
+ assert.ok(s.coverage != null && s.coverage > 0.9, `coverage ${s.coverage}`);
+});
+
+test("(b) stats built before the index had the video heal after the index build", async () => {
+ resetCorpus();
+ seedVideo("early");
+ addCaptions("early", "2026-07-11T12:00:00Z");
+ // The pool composers run buildStats against the index as it stands; this
+ // video was downloaded after the last index build.
+ const first = await runStats();
+ assert.equal(statOf(first.byId, "early").hasTranscript, false);
+
+ await runIndex();
+ const second = await runStats();
+ assert.equal(second.res.changed, 1, "the key moved off NOT_INDEXED, so the stat is redone");
+ const s = statOf(second.byId, "early");
+ assert.equal(s.hasTranscript, true);
+ assert.equal(s.transcribedDate, "20260711");
+
+ // And the lag is said, not silent: counted in the result and logged. (No
+ // index build has completed yet, so it is "not indexed yet".)
+ assert.equal(first.res.notIndexedYet, 1);
+ assert.equal(first.res.notIndexable, 0);
+ assert.ok(
+ first.log.some((l) => l.startsWith("1 video(s) were downloaded after the last index build")),
+ first.log.join("\n"),
+ );
+ assert.equal(second.res.notIndexedYet, 0);
+});
+
+test("(c) a caption-only video is dated by its captions' arrival, not by a later Normalize", async () => {
+ resetCorpus();
+ // Captions only: no transcript.json, no outcome sidecar, never normalized.
+ seedVideo("vtt-only");
+ addCaptions("vtt-only", "2026-07-11T12:00:00Z");
+ // Captions that arrived the same day, normalized a month later.
+ seedVideo("normalized");
+ addCaptions("normalized", "2026-07-11T12:00:00Z");
+ const norm = await normalizeTranscript({
+ videoDir: videoDir("normalized"),
+ channelSlug: CHANNEL,
+ });
+ assert.equal(norm.status, "wrote");
+ touch(path.join(videoDir("normalized"), "transcript.cues.json"), "2026-08-10T13:44:00Z");
+
+ await runIndex();
+ const { byId } = await runStats();
+ for (const id of ["vtt-only", "normalized"]) {
+ const s = statOf(byId, id);
+ assert.equal(s.hasTranscript, true, id);
+ assert.equal(s.transcribedDate, "20260711", id);
+ }
+});
+
+test("(d) a Whisper video resolves exactly as before: the outcome sidecar, else transcript.json", async () => {
+ resetCorpus();
+ // Transcribed before the stats first saw it, with and without the sidecar.
+ seedVideo("with-outcome");
+ addWhisper("with-outcome", "2026-06-01T10:00:00Z");
+ seedVideo("no-outcome");
+ addWhisper("no-outcome", "2026-06-02T10:00:00Z", false);
+ // The sidecar wins over every mtime, even a caption file's.
+ seedVideo("hybrid");
+ addCaptions("hybrid", "2026-05-01T10:00:00Z");
+ addWhisper("hybrid", "2026-06-04T10:00:00Z");
+ // Transcribed AFTER the stats first saw it.
+ seedVideo("after");
+ await runIndex();
+ await runStats();
+ addWhisper("after", "2026-06-03T10:00:00Z");
+ await runIndex();
+ const { byId } = await runStats();
+
+ const dates = Object.fromEntries(
+ ["with-outcome", "no-outcome", "hybrid", "after"].map((id) => [
+ id,
+ statOf(byId, id).transcribedDate,
+ ]),
+ );
+ assert.deepEqual(dates, {
+ "with-outcome": "20260601",
+ "no-outcome": "20260602",
+ hybrid: "20260604",
+ after: "20260603",
+ });
+});
+
+test("(e) the key does not churn: a heal redoes one stat, then an unchanged run reads nothing per video", async () => {
+ resetCorpus();
+ seedVideo("w");
+ addWhisper("w", "2026-06-01T10:00:00Z");
+ seedVideo("c");
+ addCaptions("c", "2026-06-01T10:00:00Z");
+ seedVideo("m");
+ await runIndex();
+ const first = await runStats();
+ assert.equal(first.res.added, 3);
+
+ addWhisper("m", "2026-06-05T10:00:00Z");
+ await runIndex();
+ const heal = await runStats();
+ assert.equal(heal.res.added, 0);
+ assert.equal(heal.res.changed, 1, "only the video whose transcript arrived");
+
+ // An index build over an unchanged corpus rewrites nothing the key reads.
+ await runIndex();
+ const dataDir = path.join(paths.channelsDir, CHANNEL, "data");
+ const from = fsCalls.length;
+ const steady = await runStats();
+ assert.equal(steady.res.added, 0);
+ assert.equal(steady.res.changed, 0);
+ assert.equal(steady.res.notIndexedYet + steady.res.notIndexable, 0);
+ // The spy sees node:fs and node:fs/promises, async and sync.
+ const perVideo = fsCalls.slice(from).filter((c) => c.p.startsWith(dataDir + path.sep));
+ assert.deepEqual(
+ perVideo.map((c) => `${c.fn} ${path.relative(dataDir, c.p)}`).sort(),
+ [
+ `stat ${path.join("c", "metadata.info.json")}`,
+ `stat ${path.join("m", "metadata.info.json")}`,
+ `stat ${path.join("w", "metadata.info.json")}`,
+ ],
+ "the unchanged path stats each metadata file and touches nothing else in a video dir",
+ );
+});
+
+test("(f) cues are read under the index's own key, even when the metadata's upload date moved", async () => {
+ resetCorpus();
+ seedVideo("moved", "2026-07-11T11:00:00Z");
+ addCaptions("moved", "2026-07-11T12:00:00Z");
+ const norm = await normalizeTranscript({ videoDir: videoDir("moved"), channelSlug: CHANNEL });
+ assert.equal(norm.status, "wrote");
+ touch(path.join(videoDir("moved"), "transcript.cues.json"), "2026-08-10T13:44:00Z");
+ // The metadata is rewritten with another upload date (a stream's date
+ // settled) but stays older than the normalized cues, so buildIndex still
+ // trusts transcript.cues.json — and keys the cues by ITS upload date.
+ seedVideo("moved", "2026-07-12T11:00:00Z", { upload_date: "20260602" });
+ await runIndex();
+ const { byId } = await runStats();
+ const s = statOf(byId, "moved");
+ assert.equal(s.uploadDate, "20260602");
+ assert.equal(s.hasTranscript, true, "the cues are found under the key the index used");
+ assert.equal(s.cueCount, 2);
+});
+
+test("(g) a re-index for another reason that changes the cues redoes the stat", async () => {
+ resetCorpus();
+ seedVideo("drift");
+ addCaptions("drift", "2026-07-11T12:00:00Z");
+ await runIndex();
+ const first = await runStats();
+ assert.equal(statOf(first.byId, "drift").cueCount, 2);
+
+ // A Normalize run writes transcript.cues.json (here with one cue fewer than
+ // the raw parse). buildIndex does not re-index for that alone...
+ await normalizeTranscript({ videoDir: videoDir("drift"), channelSlug: CHANNEL });
+ const cuesPath = path.join(videoDir("drift"), "transcript.cues.json");
+ const doc = JSON.parse(readFileSync(cuesPath, "utf8")) as { cues: unknown[] };
+ doc.cues = doc.cues.slice(0, 1);
+ writeFileSync(cuesPath, JSON.stringify(doc));
+ // ...but an availability recheck makes it re-process the video, and then it
+ // reads the fresher cues.json. The transcript mtime did not move.
+ writeJson(path.join(videoDir("drift"), "availability.json"), {
+ checkedAt: "2026-08-20T00:00:00.000Z",
+ availability: "public",
+ });
+ await runIndex();
+ const second = await runStats();
+ assert.equal(second.res.changed, 1);
+ assert.equal(statOf(second.byId, "drift").cueCount, 1);
+});
+
+test("(h) a video the index skipped is not announced as pending on every run", async () => {
+ resetCorpus();
+ seedVideo("fine");
+ // No upload_date: buildIndex skips it (it needs one for the index key).
+ seedVideo("undated", "2026-07-11T11:00:00Z", { upload_date: undefined });
+ await runIndex();
+ // Downloaded after that index build.
+ seedVideo("fresh", null);
+ for (let run = 0; run < 2; run++) {
+ const { res, log } = await runStats();
+ assert.equal(res.notIndexedYet, 1, `run ${run}`);
+ assert.equal(res.notIndexable, 1, `run ${run}`);
+ assert.ok(log.some((l) => l.startsWith("1 video(s) were downloaded after the last index build")), log.join("\n"));
+ assert.ok(
+ log.some(
+ (l) =>
+ l.startsWith("1 video(s) are not in the index although they are older than its last build") &&
+ l.includes("or its channel's media was unreachable during that build"),
+ ),
+ log.join("\n"),
+ );
+ }
+ await runIndex();
+ const { res } = await runStats();
+ assert.equal(res.notIndexedYet, 0, "the next index build takes the fresh one");
+ assert.equal(res.notIndexable, 1, "the undated one stays, and is said as such");
+});
+
+// A second channel whose data/ is a relocated symlink, the way the editor's
+// Storage panel leaves it: channels/<slug>/data -> <root>/<slug>/data, with
+// config.dataDir recording the target.
+function seedDriveChannel(): { target: string } {
+ const target = path.join(ROOT, "media", DRIVE_CHANNEL, "data");
+ mkdirSync(target, { recursive: true });
+ writeJson(path.join(paths.channelsDir, DRIVE_CHANNEL, "config.json"), {
+ handling: "youtube",
+ name: "Drive Channel",
+ url: "https://www.youtube.com/@drive/videos",
+ dataDir: target,
+ });
+ symlinkSync(target, path.join(paths.channelsDir, DRIVE_CHANNEL, "data"));
+ for (const id of ["d1", "d2"]) {
+ seedVideo(id, "2026-07-11T11:00:00Z", {}, DRIVE_CHANNEL);
+ addCaptions(id, "2026-07-11T12:00:00Z", DRIVE_CHANNEL);
+ }
+ return { target };
+}
+
+test("(i) an unmounted media drive keeps its channel's stats; a cache clear refuses", async () => {
+ resetCorpus();
+ seedVideo("local");
+ seedDriveChannel();
+ await runIndex();
+ const mounted = await runStats();
+ assert.equal(statOf(mounted.byId, "d1").hasTranscript, true);
+
+ // The drive is a storage location, as /storage records it; the refusal names
+ // it by its label, never by a path.
+ const media = path.join(ROOT, "media");
+ writeFileSync(
+ paths.settingsFile,
+ JSON.stringify({ storage: { locations: [{ id: "usb", label: "USB drive", root: media, autoRepoint: false }] } }),
+ );
+ // Unmount: the link now dangles, exactly as an absent USB drive leaves it.
+ renameSync(media, `${media}-away`);
+ const log: string[] = [];
+ const away = await runStats(log);
+ assert.equal(away.res.removed, 0, "not read as a channel with no videos");
+ for (const id of ["d1", "d2", "local"]) assert.ok(away.byId.has(id), `${id} still published`);
+ assert.deepEqual(away.res.heldChannels, [DRIVE_CHANNEL]);
+ assert.equal(statOf(away.byId, "d1").hasTranscript, true);
+ assert.ok(
+ log.some((l) => l.startsWith(`Channel ${DRIVE_CHANNEL}: its media is not reachable`) && l.includes("its 2 cached stat(s) are kept")),
+ log.join("\n"),
+ );
+
+ // A schema change needs the whole cache rebuilt, which cannot include a
+ // channel it cannot read: refuse, and leave the cache as it is.
+ setStoredSchema(STATS_SCHEMA_VERSION - 1);
+ await assert.rejects(runStats(), (err: Error) => {
+ assert.match(
+ err.message,
+ /must be rebuilt .* cannot be read: drive-channel \(its media is not reachable \(drive not mounted\?\), on location "USB drive"\)/,
+ );
+ // The ways out, mounting first, and no path in the message.
+ assert.match(
+ err.message,
+ /For each: mount its media and run this again; or repair or re-point its location on \/storage; or finish or clear its move .*; or, if it is gone for good, delete the channel or set excludeFromBuild/,
+ );
+ assert.ok(!err.message.includes(ROOT), err.message);
+ return true;
+ });
+ assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION - 1);
+ assert.equal(countStats(), 3);
+
+ renameSync(`${media}-away`, media);
+ const back = await runStats();
+ assert.deepEqual(back.res.heldChannels, []);
+ assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION);
+ assert.equal(back.byId.size, 3);
+});
+
+// Direct access to the temp LMDB's stats cache, for the schema cases.
+function withDb<T>(fn: (dbs: { meta: ReturnType<ReturnType<typeof open>["openDB"]>; stats: ReturnType<ReturnType<typeof open>["openDB"]> }) => T): T {
+ const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true });
+ try {
+ return fn({
+ meta: root.openDB({ name: "statsMeta", encoding: "msgpack" }),
+ stats: root.openDB({ name: "statsByPath", encoding: "msgpack" }),
+ });
+ } finally {
+ root.close();
+ }
+}
+const setStoredSchema = (v: number) =>
+ withDb(({ meta }) => meta.putSync("schema", v));
+const readStoredSchema = () => withDb(({ meta }) => meta.get("schema"));
+const countStats = () => withDb(({ stats }) => [...stats.getKeys()].length);
+
+test("(j) the schema guard: an older cache is cleared, a newer one is refused unless overridden", async () => {
+ resetCorpus();
+ seedVideo("v1");
+ addCaptions("v1", "2026-07-11T12:00:00Z");
+ await runIndex();
+ await runStats();
+ assert.equal(countStats(), 1);
+
+ // Older: cleared and rebuilt, as every schema bump has always done.
+ setStoredSchema(STATS_SCHEMA_VERSION - 1);
+ const log: string[] = [];
+ const older = await runStats(log);
+ assert.ok(log.some((l) => l.includes("clearing stats cache")), log.join("\n"));
+ assert.equal(older.res.added, 1);
+ assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION);
+
+ // Newer: refused, naming both versions and the override; nothing touched.
+ setStoredSchema(STATS_SCHEMA_VERSION + 1);
+ await assert.rejects(
+ runStats(),
+ new RegExp(
+ `newer build \\(stats schema ${STATS_SCHEMA_VERSION + 1}; this build's is ${STATS_SCHEMA_VERSION}\\).*${STATS_DOWNGRADE_ENV}=1`,
+ ),
+ );
+ assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION + 1);
+ assert.equal(countStats(), 1);
+
+ // The CLI exits non-zero on it.
+ const cli = spawnSync(
+ path.join(COMMON, "node_modules", ".bin", "tsx"),
+ ["bin/archilyzer.ts", "build", "stats"],
+ { cwd: COMMON, env: { ...process.env }, encoding: "utf8" },
+ );
+ assert.notEqual(cli.status, 0, cli.stdout + cli.stderr);
+ assert.match(cli.stderr, /written by a newer build/);
+ assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION + 1);
+ assert.equal(countStats(), 1);
+
+ // A deliberate rollback, overridden: cleared and rebuilt at this version.
+ process.env[STATS_DOWNGRADE_ENV] = "1";
+ try {
+ const rolled = await runStats();
+ assert.equal(rolled.res.added, 1);
+ assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION);
+ } finally {
+ delete process.env[STATS_DOWNGRADE_ENV];
+ }
+});
+
+test("(z) no write this file caused landed outside its temp root", () => {
+ // LMDB writes natively, past the spy: its file must be under the root too.
+ assert.ok(paths.lmdbPath.startsWith(ROOT + path.sep), paths.lmdbPath);
+ const outside = fsCalls.filter(
+ (c) => c.write && c.p !== ROOT && !c.p.startsWith(ROOT + path.sep),
+ );
+ assert.deepEqual(outside, []);
+ assert.ok(fsCalls.some((c) => c.write), "the spy saw the writes");
+});
diff --git a/common/controller/buildStats.ts b/common/controller/buildStats.ts
@@ -2,13 +2,34 @@
// page-NNNN}.json for the viewer charts feature. Engagement metrics
// (view/like/comment counts, follower count, categories, language) live in each
// video's metadata.info.json but are NOT carried by the search index, so this
-// reads the raw metadata. Incremental: per-video mtime state is kept in a
-// dedicated `statsByPath` LMDB sub-DB so re-runs only re-parse changed videos.
+// reads the raw metadata. Incremental: per-video state is kept in a dedicated
+// `statsByPath` LMDB sub-DB so re-runs only re-parse changed videos.
//
// Transcript presence + cue count are read from the index LMDB `cues` sub-DB,
// and each video's visibility from the `videoState` sub-DB — both populated by
// buildIndex, so this must run after build:index, which the export prebuild
-// guarantees by chaining build:index && build:stats.
+// guarantees by chaining build:index && build:stats. The pool composers
+// (compose-hub, compose-homepage) do NOT chain it: they read the index as it
+// stands, which the cache key below makes safe.
+//
+// THE CACHE KEY IS TWO THINGS: the metadata file's mtime AND the index's own
+// per-video record (buildIndex's `mtimes` entry, as indexSignature, or
+// NOT_INDEXED). Until schema 6 it was the metadata mtime alone, while
+// hasTranscript / cueCount / coverage / transcribedDate come from the index and
+// the transcript files — so a transcript that arrived after a video was first
+// seen (Whisper days later, or a stats run before build:index had the video)
+// never reached its stat, and a whole channel could publish as untranscribed.
+//
+// A CHANNEL WHOSE MEDIA IS NOT REACHABLE (a relocated `data/` on an unmounted
+// drive, or one mid-relocation) is not rescanned: its cached stats are kept as
+// they are, rather than read as a channel with no videos and removed. This is
+// the stats build's own guard — the job registry's `needsMedia` check is per
+// channel, and this build is pool-wide — so the editor job and the CLI share it.
+//
+// ONE STATS BUILD AT A TIME. Nothing here takes a lock: two runs at once are
+// harmless unless one of them clears the cache (a schema change) after the
+// other scanned, when the other can publish truncated pages. The editor's build
+// jobs share the "build" queue by default; the CLI is outside every queue.
import path from "node:path";
import { createHash } from "node:crypto";
@@ -34,12 +55,24 @@ import {
import type { VideoState } from "../lib/availability";
import { loadDownloadOutcome } from "../lib/downloadOutcome-server";
import { loadTranscribeOutcome } from "../lib/transcribeOutcome-server";
+import {
+ CUES_JSON_FILENAME,
+ pickIndexTranscript,
+ readVideoFiles,
+} from "../lib/videoStatus";
import type { VideoStatus } from "../lib/stats";
import type { ChannelConfig } from "../lib/channelConfig";
import { readChannelConfigFile } from "./channels";
+import {
+ inspectChannelMedia,
+ type ChannelMediaStatus,
+} from "../lib/channelMedia";
+import { getSettings } from "../lib/settings";
+import { locationLabelOfDataDir } from "../lib/storageLocations";
import type { Paths } from "../lib/paths";
import { listSites, siteStatsDir } from "../lib/site";
import {
+ INDEX_SCANNED_AT_KEY,
STATS_SCHEMA_VERSION,
STATS_MANIFEST_VERSION,
STATS_MAX_PAGE_BYTES,
@@ -51,7 +84,56 @@ import {
type IndexKey = [string, string, string];
type PathKey = [string, string];
-type StatsRecord = { metaMs: number; stat: VideoStat };
+
+// `idx` is the index's per-video record as this stat saw it (indexSignature).
+// Optional only because a record written before schema 6 has none; the schema
+// bump clears those, and a missing value compares unequal to every real one, so
+// such a record would be recomputed anyway.
+type StatsRecord = { metaMs: number; idx?: string; stat: VideoStat };
+
+// The part of buildIndex's `mtimes` record this cache reads: every input whose
+// change makes buildIndex re-process the video (and so rewrite its cues), and
+// the key it stored the cues under.
+type IndexRecord = {
+ metaMs: number;
+ transcriptMs: number | null;
+ subsMs?: number | null;
+ availabilityMs?: number | null;
+ digestMs?: number | null;
+ indexKey: IndexKey;
+};
+
+// buildIndex has no `mtimes` record for this video. See the two counts below.
+const NOT_INDEXED = "-";
+
+// The whole index record as one comparable string. Keying on all of it (not
+// the transcript mtime alone) means ANY re-index redoes the stat: a re-index
+// for another reason can read a fresher transcript.cues.json and change the cue
+// count, and a moved index key moves the cues. Numbers print as their shortest
+// round-trip form, so an unchanged record gives an identical string.
+function indexSignature(r: IndexRecord | undefined): string {
+ if (!r) return NOT_INDEXED;
+ const n = (v: number | null | undefined) => (v == null ? "" : String(v));
+ return [
+ n(r.metaMs),
+ n(r.transcriptMs),
+ n(r.subsMs),
+ n(r.availabilityMs),
+ n(r.digestMs),
+ ...r.indexKey,
+ ].join("\u0000");
+}
+
+// Set to 1 (or true/yes/on) to let an OLDER build clear a stats cache a newer
+// one wrote — a deliberate rollback. Declared in lib/envVars.ts.
+export const STATS_DOWNGRADE_ENV = "ARCHILYZER_STATS_ALLOW_DOWNGRADE";
+const TRUTHY = new Set(["1", "true", "yes", "on"]);
+function allowsStatsDowngrade(
+ env: Record<string, string | undefined> = process.env,
+): boolean {
+ const raw = env.ARCHILYZER_STATS_ALLOW_DOWNGRADE;
+ return typeof raw === "string" && TRUTHY.has(raw.trim().toLowerCase());
+}
type ScanEntry = {
channelSlug: string;
@@ -68,6 +150,16 @@ export type BuildStatsResult = {
added: number;
changed: number;
removed: number;
+ // Videos on disk with no index record, in two kinds. `notIndexedYet`: their
+ // metadata is newer than the last completed index build's scan — downloaded
+ // since — and the first stats run after the next index build redoes them.
+ // `notIndexable`: older than the last index build, which did not index them
+ // (no upload_date, a failure, or their channel's media unreachable during
+ // that build); a stats run alone will not change that.
+ notIndexedYet: number;
+ notIndexable: number;
+ // Channels whose media was not reachable, so their cached stats were kept.
+ heldChannels: string[];
pagesWritten: number;
shortCircuited: boolean;
durationMs: number;
@@ -116,6 +208,22 @@ async function fileMtimeMs(p: string): Promise<number | null> {
// "content added over time" progress charts. Prefers the explicit outcome
// sidecars (reliable across the shard rsync model, where file mtimes drift);
// falls back to file mtimes for content added before the sidecars existed.
+//
+// A TRANSCRIPT ALWAYS HAS A DATE (schema 6). The fallbacks, in order:
+// 1. transcribe-outcome.json's `transcribedAt` — every Whisper run writes it;
+// 2. the mtime of the transcript the index takes its cues from
+// (pickIndexTranscript: transcript.json, else the caption VTT);
+// 3. the mtime of transcript.cues.json;
+// 4. downloadedDate, which always resolves.
+// A Whisper video resolves exactly as before: 1, else transcript.json's mtime,
+// which is what (2) picks whenever transcript.json exists (the one difference:
+// a sidecar whose `transcribedAt` will not parse used to leave no date, and now
+// falls through to 2). A CAPTION-handled
+// video is dated by when its captions ARRIVED — the VTT's mtime — not by a later
+// Normalize run: (3) used to be the only file that could date one, so a
+// caption video either had no date (never normalized) or took the Normalize
+// run's date (1,683 of one channel's, all on one day). The homepage fold, the
+// charts and the recent rail all need `hasTranscript` ⇒ `transcribedDate`.
async function resolveAcquisitionDates(
videoDir: string,
metaMs: number,
@@ -128,29 +236,48 @@ async function resolveAcquisitionDates(
let transcribedDate: string | null = null;
if (hasTranscript) {
const tr = await loadTranscribeOutcome(videoDir);
- if (tr?.transcribedAt) {
- transcribedDate = ymdFromIso(tr.transcribedAt);
- } else {
+ transcribedDate = tr?.transcribedAt ? ymdFromIso(tr.transcribedAt) : null;
+ if (transcribedDate === null) {
+ const picked = pickIndexTranscript(await readVideoFiles(videoDir));
const mtime =
- (await fileMtimeMs(path.join(videoDir, "transcript.json"))) ??
- (await fileMtimeMs(path.join(videoDir, "transcript.cues.json")));
- transcribedDate = mtime != null ? ymdFromMs(mtime) : null;
+ (picked ? await fileMtimeMs(path.join(videoDir, picked.filename)) : null) ??
+ (await fileMtimeMs(path.join(videoDir, CUES_JSON_FILENAME)));
+ transcribedDate = (mtime != null ? ymdFromMs(mtime) : null) ?? downloadedDate;
}
}
return { downloadedDate, transcribedDate };
}
+// Why a channel is held, without the paths inspectChannelMedia's `detail`
+// carries (/storage shows those).
+const HELD_REASON: Record<ChannelMediaStatus, string> = {
+ unreachable: "its media is not reachable (drive not mounted?)",
+ "in-transition": "a move of its media is in progress or was interrupted",
+ inconsistent: "its data link and its config disagree",
+ ok: "reachable",
+ "in-place": "reachable",
+};
+
+// `held` maps each channel whose media is not reachable to why, in words with
+// no path in them: it is not scanned, and the caller keeps its cached stats
+// (see the file header).
async function scanSource(
channelsDir: string,
log: (msg: string) => void,
-): Promise<{ entries: ScanEntry[]; channels: Map<string, ChannelConfig> }> {
+): Promise<{
+ entries: ScanEntry[];
+ channels: Map<string, ChannelConfig>;
+ held: Map<string, string>;
+}> {
+ const locations = getSettings().storage.locations;
const channels = new Map<string, ChannelConfig>();
const entries: ScanEntry[] = [];
+ const held = new Map<string, string>();
let channelEntries: Dirent[];
try {
channelEntries = await readdir(channelsDir, { withFileTypes: true });
} catch {
- return { entries, channels };
+ return { entries, channels, held };
}
for (const ch of channelEntries) {
if (!ch.isDirectory()) continue;
@@ -162,6 +289,17 @@ async function scanSource(
}
if (cfg.excludeFromBuild) continue;
channels.set(ch.name, cfg);
+ // An unmounted drive is not an empty channel (lib/channelMedia.ts): the
+ // readdir below would fail and every one of its stats would be removed.
+ const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg);
+ if (media.status !== "ok" && media.status !== "in-place") {
+ const label = locationLabelOfDataDir(cfg.dataDir ?? media.target, locations);
+ held.set(
+ ch.name,
+ `${HELD_REASON[media.status]}${label ? `, on location "${label}"` : ""}`,
+ );
+ continue;
+ }
const dataDir = path.join(channelDir, "data");
let videoEntries: Dirent[];
try {
@@ -187,7 +325,7 @@ async function scanSource(
});
}
}
- return { entries, channels };
+ return { entries, channels, held };
}
// Buffered byte-capped page writer. Stats records are small, so a page fits in
@@ -266,10 +404,53 @@ export async function buildStats({
encoding: "msgpack",
});
const meta = root.openDB<unknown, string>({ name: "statsMeta", encoding: "msgpack" });
+ // Read-only views of buildIndex's per-video `mtimes` record (keyed like
+ // statsByPath: the second half of this cache's key, and the key the video's
+ // cues were stored under) and of its `meta` (when its last scan began). One
+ // LMDB get per video, and no file I/O, on the unchanged path.
+ const indexMtimes = root.openDB<IndexRecord, PathKey>({
+ name: "mtimes",
+ encoding: "msgpack",
+ });
+ const indexMeta = root.openDB<unknown, string>({ name: "meta", encoding: "msgpack" });
+ // NEVER CLEAR A CACHE A NEWER BUILD WROTE. An old build running beside a
+ // new one (an editor not yet restarted onto the new code) would otherwise
+ // clear it, refill it the old way, and the next new run would clear it back:
+ // a full pass each time, and old-logic numbers published in between.
const storedSchema = meta.get("schema") as number | undefined;
+ if (
+ typeof storedSchema === "number" &&
+ storedSchema > STATS_SCHEMA_VERSION &&
+ !allowsStatsDowngrade()
+ ) {
+ await root.close();
+ throw new Error(
+ `The stats cache was written by a newer build (stats schema ${storedSchema}; this build's is ${STATS_SCHEMA_VERSION}). ` +
+ `Refusing to clear it: rebuild and restart onto the current code. ` +
+ `For a deliberate rollback, set ${STATS_DOWNGRADE_ENV}=1.`,
+ );
+ }
+
+ const { entries, channels, held } = await scanSource(paths.channelsDir, log);
+ log(`Scanned ${entries.length} videos across ${channels.size} channels.`);
+
const schemaBumped = storedSchema !== STATS_SCHEMA_VERSION;
if (schemaBumped) {
+ // A clear with a channel's media unreachable would drop that channel's
+ // stats for good (it cannot be rescanned), and the pages built from this
+ // run would publish it as empty. Refuse instead.
+ if (held.size > 0) {
+ await root.close();
+ throw new Error(
+ `The stats cache must be rebuilt (stats schema ${storedSchema ?? "<none>"} -> ${STATS_SCHEMA_VERSION}), ` +
+ `but ${held.size} channel(s) cannot be read: ` +
+ [...held].map(([slug, why]) => `${slug} (${why})`).join("; ") +
+ `. For each: mount its media and run this again; or repair or re-point its location on /storage; ` +
+ `or finish or clear its move (the channel's Storage panel); or, if it is gone for good, ` +
+ `delete the channel or set excludeFromBuild in its config.`,
+ );
+ }
log(
`Stats schema change (${storedSchema ?? "<none>"} -> ${STATS_SCHEMA_VERSION}); clearing stats cache.`,
);
@@ -277,30 +458,65 @@ export async function buildStats({
await meta.put("schema", STATS_SCHEMA_VERSION);
}
- const { entries, channels } = await scanSource(paths.channelsDir, log);
- log(`Scanned ${entries.length} videos across ${channels.size} channels.`);
-
const liveIds = new Set(
entries.map((e) => pathKeyId([e.channelSlug, e.videoDir])),
);
- const toProcess: ScanEntry[] = [];
+ const scannedAt = indexMeta.get(INDEX_SCANNED_AT_KEY) as number | undefined;
+ const toProcess: { e: ScanEntry; idx: string; indexKey?: IndexKey }[] = [];
let added = 0;
let changed = 0;
+ let notIndexedYet = 0;
+ let notIndexable = 0;
for (const e of entries) {
- const prev = statsByPath.get([e.channelSlug, e.videoDir]);
+ const pk: PathKey = [e.channelSlug, e.videoDir];
+ const rec = indexMtimes.get(pk);
+ const idx = indexSignature(rec);
+ if (idx === NOT_INDEXED) {
+ if (typeof scannedAt !== "number" || e.metaMs > scannedAt) notIndexedYet++;
+ else notIndexable++;
+ }
+ const prev = statsByPath.get(pk);
if (!prev) {
added++;
- toProcess.push(e);
- } else if (prev.metaMs !== e.metaMs) {
+ toProcess.push({ e, idx, indexKey: rec?.indexKey });
+ } else if (prev.metaMs !== e.metaMs || prev.idx !== idx) {
+ // The second half is the fix for stats frozen at first sight: a
+ // transcript that arrives later makes buildIndex re-process the video
+ // (and a video first seen before the index had it moves off
+ // NOT_INDEXED), while the metadata file — the old key's only input — is
+ // never touched.
changed++;
- toProcess.push(e);
+ toProcess.push({ e, idx, indexKey: rec?.indexKey });
}
}
+ // Neither is a failure of this build, and both are said, so a published
+ // number that lags the disk has its reason in the log. Only the first kind
+ // resolves itself.
+ if (notIndexedYet > 0) {
+ log(
+ `${notIndexedYet} video(s) were downloaded after the last index build; their transcripts reach the stats on the first run after the next one.`,
+ );
+ }
+ if (notIndexable > 0) {
+ log(
+ `${notIndexable} video(s) are not in the index although they are older than its last build: it skipped them (no upload_date, or it failed on them: see that build's log), or its channel's media was unreachable during that build (run an index build with every drive mounted). Their stats show no transcript until then.`,
+ );
+ }
const removedKeys: PathKey[] = [];
+ const keptHeld = new Map<string, number>();
for (const { key } of statsByPath.getRange()) {
const k = key as PathKey;
+ if (held.has(k[0])) {
+ keptHeld.set(k[0], (keptHeld.get(k[0]) ?? 0) + 1);
+ continue;
+ }
if (!liveIds.has(pathKeyId(k))) removedKeys.push(k);
}
+ for (const [slug, why] of held) {
+ log(
+ `Channel ${slug}: ${why}; its ${keptHeld.get(slug) ?? 0} cached stat(s) are kept as they are, not rescanned.`,
+ );
+ }
const removed = removedKeys.length;
const BATCH = 200;
@@ -308,7 +524,7 @@ export async function buildStats({
signal?.throwIfAborted();
const slice = toProcess.slice(i, i + BATCH);
await Promise.all(
- slice.map(async (e) => {
+ slice.map(async ({ e, idx, indexKey: recordedKey }) => {
try {
const metaRaw = await readFile(e.metaPath, "utf8");
const parsedMeta = JSON.parse(metaRaw) as RawMetadata;
@@ -319,16 +535,22 @@ export async function buildStats({
e.configName,
);
if (!base.uploadDate) return;
- const indexKey: IndexKey = [base.uploadDate, e.channelSlug, base.id];
+ // The key buildIndex stored this video's cues under, when it has a
+ // record: its uploadDate comes from transcript.cues.json when that
+ // is fresh, which the metadata's can differ from. The computed key
+ // is only a fallback for a video the index does not have.
+ const indexKey: IndexKey =
+ recordedKey ?? [base.uploadDate, e.channelSlug, base.id];
const cueList = cues.get(indexKey);
const cueCount = cueList ? cueList.length : null;
const coverage = transcriptCoverage(cueList, base.duration).coverage;
// Placeholder. `status` is NOT a cached field any more: it is applied
// from buildIndex's `videoState` sub-DB at collection time below.
- // This record is keyed on metadata mtime alone, so caching a status
- // here meant a pure availability flip only reached the chart on the
- // next metadata touch or schema bump — a video deleted today did not
- // show as deleted today.
+ // This record's key does not see availability (metadata mtime and
+ // the index's transcript mtime only), so caching a status here meant
+ // a pure availability flip only reached the chart on the next
+ // metadata touch or schema bump — a video deleted today did not show
+ // as deleted today.
const status: VideoStatus = "available";
const hasTranscript = cueCount != null && cueCount > 0;
const { downloadedDate, transcribedDate } =
@@ -353,6 +575,7 @@ export async function buildStats({
);
await statsByPath.put([e.channelSlug, e.videoDir], {
metaMs: e.metaMs,
+ idx,
stat,
});
} catch (err) {
@@ -555,6 +778,9 @@ export async function buildStats({
added,
changed,
removed,
+ notIndexedYet,
+ notIndexable,
+ heldChannels: [...held.keys()],
pagesWritten: aggregatePages,
shortCircuited: needBuild.length === 0,
durationMs: Date.now() - t0,
diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts
@@ -111,6 +111,7 @@ const DECLARED: EnvVarDecl[] = [
{ name: "TRANSCRIPT_PLATFORM_LINKS", audience: "runtime", default: "off", readBy: "common/lib/archive/reader-fs.ts", doc: "`1` cites platform watch pages instead of the archive's own pages." },
{ name: "AUDIO_CHECK_RESUME_DURING_PROBE", audience: "runtime", default: "the channel's `audioCheck.resumeDuringProbe`", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "`1` or `true` resumes yt-dlp during the audio check's probe, anything else holds it, for a one-off comparison run; unset = the channel's setting." },
{ name: "AUDIO_CHECK_BACKOFF_FACTOR", audience: "runtime", default: "the built-in factor", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "The audio check's interval backoff factor, in (0, 1], for a one-off run." },
+ { name: "ARCHILYZER_STATS_ALLOW_DOWNGRADE", audience: "runtime", default: "off", readBy: "common/controller/buildStats.ts", doc: "`1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions." },
{ name: "MCP_IO_STATS", audience: "runtime", default: "off", readBy: "common/lib/archive/io-stats.ts", doc: "`1` turns on per-call I/O accounting, for `mcp/bench`." },
{ name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool", doc: "Which editor `pnpm ops` and the MCP's `fetch_clip` talk to." },
{ name: "ARCHILYZER_AGENT", audience: "runtime", default: "`cli`", readBy: "scripts/archilyzer-ops.mjs", doc: "Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`." },
diff --git a/common/lib/homepageSummary.test.ts b/common/lib/homepageSummary.test.ts
@@ -198,6 +198,46 @@ test("a site's accent is published as a hex: an id becomes its on-dark value", (
assert.ok(plain.sites.every((x) => !("accent" in x)));
});
+// Stats schema 6 guarantees a transcript a date, but a stats page written
+// before it could carry a whole channel of transcripts with none — and the fold
+// used to require the date for everything, so a site serving 1,889 videos
+// showed 0 transcripts, 0 channels, 0 hours. The date is only for the time axis.
+test("a transcript with no transcribedDate (a pre-schema-6 page) is counted, just not charted", () => {
+ const sites = [...SITES, site("gamma", ["g1"], "https://gamma.example")];
+ const undated = [
+ stat({ channelSlug: "g1", id: "u1", uploadDate: "20251101", duration: 3600, transcribedDate: null }),
+ stat({ channelSlug: "g1", id: "u2", uploadDate: "20260210", duration: 3600, transcribedDate: null }),
+ ];
+ const base = buildHomepageSummary(STATS, CHANNEL_SITES, sites, NOW);
+ const s = buildHomepageSummary([...STATS, ...undated], { ...CHANNEL_SITES, g1: ["gamma"] }, sites, NOW);
+
+ const gamma = s.sites.find((x) => x.siteId === "gamma");
+ assert.ok(gamma, "a site whose transcripts are all undated is still on the family page");
+ assert.deepEqual(
+ [gamma.channels, gamma.transcribed.total, gamma.hoursArchived, gamma.recordings],
+ [1, 2, 2, 2],
+ );
+ // No month to put them in: not this month, not the sparkline, not a series.
+ assert.equal(gamma.transcribed.thisMonth, 0);
+ assert.deepEqual(gamma.transcribed.last12, new Array(12).fill(0));
+ assert.equal(s.series.transcribed.month.bySite.gamma, undefined);
+ assert.deepEqual(s.series.transcribed.month.total, base.series.transcribed.month.total);
+ assert.equal(s.totals.transcribedThisMonth, base.totals.transcribedThisMonth);
+ assert.ok(!s.recent.some((r) => r.siteId === "gamma"), "the recent rail needs a date");
+ // Counted everywhere a count lives, and placed by UPLOAD month like any other.
+ assert.equal(s.totals.transcripts, base.totals.transcripts + 2);
+ assert.equal(s.totals.channels, base.totals.channels + 1);
+ assert.equal(s.official!.transcripts, base.official!.transcripts + 2);
+ assert.equal(s.official!.channels, base.official!.channels + 1);
+ assert.equal(s.monthly!.find((m) => m.month === "2025-11")!.bySite.gamma, 1);
+ assert.equal(s.monthly!.find((m) => m.month === "2026-02")!.bySite.gamma, 1);
+ const placed = s.monthly!.reduce(
+ (a, m) => a + Object.values(m.bySite).reduce((x, y) => x + y, 0),
+ 0,
+ );
+ assert.equal(placed + s.monthlyUnplaced!, s.official!.transcripts);
+});
+
test("a named accent also travels as its id; a custom hex and no accent carry none", () => {
const sites = [
{ ...site("alpha", ["a1", "a2"], "https://alpha.example"), accent: " Brass " },
diff --git a/common/lib/homepageSummary.ts b/common/lib/homepageSummary.ts
@@ -61,6 +61,9 @@ export type MetricSeries = {
};
export type SiteMetricStat = {
+ // For `transcribed`, also counts transcripts with no transcribedDate (a stats
+ // page from before stats schema 6) — they have no month, so they are in
+ // `total` and in neither `thisMonth` nor `last12`.
total: number;
thisMonth: number;
last12: number[]; // last 12 month buckets (zero-padded left), for the sparkline
@@ -322,6 +325,12 @@ function siteStatFrom(
return { total, thisMonth, last12 };
}
+// A site's transcribed card counts its undated transcripts too. They have no
+// month, so they join `total` only — never `thisMonth` or the sparkline.
+function withUndated(stat: SiteMetricStat, undated: number): SiteMetricStat {
+ return undated > 0 ? { ...stat, total: stat.total + undated } : stat;
+}
+
export function buildHomepageSummary(
stats: readonly VideoStat[],
channelSites: ChannelSitesMap,
@@ -362,11 +371,14 @@ export function buildHomepageSummary(
// seconds, gone. `transcripts` is counted HERE, in the same pass that places
// or leaves unplaced each transcript, so `placed + unplaced =
// official.transcripts` holds even for a `transcribedDate` that bucketize
- // drops (a future or malformed month).
+ // drops (a future or malformed month). `undated` counts the transcripts with
+ // no `transcribedDate` at all (a stats page from before stats schema 6): no
+ // transcribed bucket can hold them, so they join the card's total directly.
type SiteAcc = {
channels: Set<string>;
recordings: number;
transcripts: number;
+ undated: number;
seconds: number;
gone: number;
};
@@ -376,7 +388,7 @@ export function buildHomepageSummary(
if (!a)
siteAcc.set(
id,
- (a = { channels: new Set(), recordings: 0, transcripts: 0, seconds: 0, gone: 0 }),
+ (a = { channels: new Set(), recordings: 0, transcripts: 0, undated: 0, seconds: 0, gone: 0 }),
);
return a;
};
@@ -385,12 +397,20 @@ export function buildHomepageSummary(
let monthlyUnplaced = 0;
for (const s of stats) {
- const hasTx = s.hasTranscript && !!s.transcribedDate;
+ // A transcript COUNTS whether or not it carries a date: transcripts,
+ // channels, hours, upload-month placement. Only the transcribed time series
+ // (and "this month", and the recent rail) need `transcribedDate`. buildStats
+ // guarantees one since stats schema 6, but a page written before that could
+ // hold a whole channel of transcripts with none — and requiring the date
+ // here made a site serving 1,889 videos show 0 transcripts, 0 channels and
+ // 0 hours.
+ const hasTx = s.hasTranscript;
+ const dated = hasTx && !!s.transcribedDate;
if (hasTx) {
transcripts += 1;
channelSet.add(s.channelSlug);
hoursSeconds += s.duration > 0 ? s.duration : 0;
- if (monthOf(s.transcribedDate) === nowMonth) transcribedThisMonth += 1;
+ if (dated && monthOf(s.transcribedDate) === nowMonth) transcribedThisMonth += 1;
}
if (s.downloadedDate) {
downloads += 1;
@@ -410,7 +430,11 @@ export function buildHomepageSummary(
const um = uploadMonthOf(s.uploadDate);
if (um && um < nowMonth) uploadItems.push({ month: um, siteId });
else monthlyUnplaced += 1;
- transcribedItems.push({ date: s.transcribedDate as string, channelSlug: s.channelSlug, siteId });
+ if (dated) {
+ transcribedItems.push({ date: s.transcribedDate as string, channelSlug: s.channelSlug, siteId });
+ } else {
+ acc.undated += 1;
+ }
}
if (s.downloadedDate) {
downloadedItems.push({ date: s.downloadedDate, channelSlug: s.channelSlug, siteId });
@@ -433,7 +457,10 @@ export function buildHomepageSummary(
siteTitle: s.siteTitle,
siteDescription: s.siteDescription,
siteUrl: s.siteUrl,
- transcribed: siteStatFrom(series.transcribed.month, s.siteId, nowMonth),
+ transcribed: withUndated(
+ siteStatFrom(series.transcribed.month, s.siteId, nowMonth),
+ siteAcc.get(s.siteId)?.undated ?? 0,
+ ),
downloaded: siteStatFrom(series.downloaded.month, s.siteId, nowMonth),
channels: siteAcc.get(s.siteId)?.channels.size ?? 0,
recordings: siteAcc.get(s.siteId)?.recordings ?? 0,
diff --git a/common/lib/stats.ts b/common/lib/stats.ts
@@ -3,10 +3,21 @@ import { pageFileName } from "./manifest";
import type { VideoState } from "./availability";
// Bumping this invalidates the LMDB `statsByPath` incremental cache and forces
-// a full re-extraction (e.g. when a new field is added below).
-export const STATS_SCHEMA_VERSION = 5;
+// a full re-extraction (e.g. when a new field is added below). It versions the
+// CACHE, not the published pages: STATS_MANIFEST_VERSION is theirs.
+// 6 — the cache key gained the index's transcript record, and a transcript
+// always has a `transcribedDate` (caption videos: the VTT's arrival).
+// The page shape did not change.
+export const STATS_SCHEMA_VERSION = 6;
export const STATS_MANIFEST_VERSION = 1;
+// A key in buildIndex's `meta` sub-DB (not this cache's): the ms timestamp the
+// last COMPLETED index build's scan began at. buildIndex writes it; buildStats
+// reads it to tell a video downloaded since that scan from one the scan saw and
+// did not index. Here rather than in buildIndex.ts so buildStats need not load
+// the index builder. Additive: the index schema did not move.
+export const INDEX_SCANNED_AT_KEY = "scannedAt";
+
// Visibility of a video on its source platform. One type with the viewer's, so
// the status chart and the search filter can never drift apart; see VideoState
// in lib/availability for the taxonomy (available + the five missing leaves).
@@ -38,6 +49,10 @@ export type VideoStat = {
// "content added over time" progress charts; the time X-axis can bin on any of
// these date fields.
downloadedDate: string | null;
+ // Non-null whenever `hasTranscript` is, since schema 6 (a caption video takes
+ // its captions' arrival). A page written by an older build can still carry a
+ // transcript with a null date: readers count it and only leave it off a time
+ // axis (homepageSummary does exactly that).
transcribedDate: string | null;
timestamp: number | null; // unix seconds
duration: number; // seconds
diff --git a/common/publish/source.test.ts b/common/publish/source.test.ts
@@ -549,9 +549,11 @@ test("a denied literal no rule removes: refused, nothing written, the report nev
// --keep-scratch keeps the clone for a look, never the scrub rules.
const kept = /scratch kept at (\S+) \(replace\.txt, the scrub rules, deleted\)/.exec(report);
assert.ok(kept, report);
- assert.ok(existsSync(path.join(kept[1], "bare")));
- assert.ok(!existsSync(path.join(kept[1], "replace.txt")));
- rmSync(kept[1], { recursive: true, force: true });
+ // The log tildifies with the real home dir, so a TMPDIR under it prints "~/…".
+ const keptDir = kept[1].replace(/^~(?=\/|$)/, os.homedir());
+ assert.ok(existsSync(path.join(keptDir, "bare")));
+ assert.ok(!existsSync(path.join(keptDir, "replace.txt")));
+ rmSync(keptDir, { recursive: true, force: true });
});
test("a refusal naming a tree path masks a literal that spans path components (review R2-L1)", async (t) => {
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,8 @@
# Changelog
## [Unreleased]
+- **Transcripts that arrived after a video was first seen are counted.** The stats behind the homepage, the hub and every site's charts were cached per video and refreshed only when the video's metadata changed, so a transcript that came later — a Whisper run days after the download, or a video downloaded after the last index build — never reached them, and a video with YouTube captions alone had no transcription date. Counts and charts were low; the homepage could show a site with 0 transcripts, 0 channels and 0 hours while it served its videos. A stat is now also redone whenever the index re-reads the video, every transcript has a date, and a captioned video is dated by when its captions arrived rather than by a later Normalize run, so its place on "Transcribed over time" can move. **After updating, rebuild and restart the editor before anything else:** until then, **Build stats dataset** runs the old code and would undo the new stats, while a site, hub or homepage build already runs the new code — and the first stats build of any kind re-reads every video once (about 10–30 minutes on a large archive; it can be stopped and picks up where it stopped). Then build the index, the stats, the homepage, the hub, and the sites.
+- **A stats build keeps the stats of a channel whose drive is not mounted, and will not undo a newer version's stats.** A channel whose media is on a drive that is not mounted (or is being moved) is left as it was instead of being read as a channel with no videos; a stats rebuild that has to start over refuses until the drive is back. A stats build refuses to clear stats written by a newer version of the editor; set `ARCHILYZER_STATS_ALLOW_DOWNGRADE=1` to roll back on purpose. Its log also says apart how many videos were downloaded since the last index build (they catch up after the next one) and how many the index skipped (no upload date, or it failed on them).
- **Building the homepage now publishes the source: a read-only git mirror, its raw tree and a fresh tarball, behind a gate.** `archilyzer build homepage`, the `/sites` Homepage jobs and `pnpm ops build-homepage` run `archilyzer source publish` between compose and `next build`. It makes a fresh clone of the private `main` (the repository itself is never rewritten), rewrites that copy with git-filter-repo using your scrub rules (file contents and commit messages; your home directory becomes `/home/user` without a rule), and publishes it under `homepage/public` for `git clone https://archilyzer.pages.dev/source/archilyzer.git`, beside `/source/tree/` and the Downloads tarball. Before anything is written, every object of the rewritten history and every file about to be published is searched for every string you have denied; **one hit refuses the build**, and its log names the string only by where you wrote it (`denylist line 3 (len 5)`) and each hit by its object, field and byte offset — never a byte of the object. **A refusal withdraws the source**: the last publish is removed from `homepage/public` and the last build's copy from `homepage/out`, and **Deploy homepage refuses** a build whose source was not audited under today's rules and today's `main` ("run `archilyzer build homepage`, then deploy"). The rules live outside the repo, in `~/.config/archilyzer/source-scrub.txt` and `source-denylist.txt` (`ARCHILYZER_CONFIG_DIR`, `SOURCE_SCRUB_FILE`, `SOURCE_DENYLIST_FILE`); **without them the build refuses**, naming the missing file. **Put everything private in the denylist before any deploy, a preview included**: previews are public, and every deployment stays reachable at its own address until you delete it. Install git-filter-repo once (`pipx install git-filter-repo`; the editor's process needs `~/.local/bin` on its `PATH` to find it) — without it the build fetches it through `pipx run`, which needs the network — and gitleaks if you want its secret scan too. An unchanged `main` with unchanged rules is skipped, so a rebuild costs about 20 seconds only when something moved. A checkout with no git repository (the docker image, a tarball install) builds with the /source page's empty state. `archilyzer source publish --check` audits without writing, `archilyzer source audit <clone>/.git` checks any clone, `archilyzer build homepage --no-source` removes the published source instead, and `archilyzer doctor` reports the tools, the two files (rule counts and permissions, never their contents) and the last publish. `create-archives.sh` is gone. See PUBLISH.md, "The source mirror (homepage)".
- **umtool reads the corpus from its checkout (or `TRANSCRIPTS_DIR`), and the song project's data defaults to `~/.local/share/archilyzer/song`.** If yours is elsewhere, link it there before restarting umtool: `mkdir -p ~/.local/share/archilyzer && ln -s <where the data is> ~/.local/share/archilyzer/song` (the data stays where it is). With no `CHANNELS_DIR`, umtool reads the corpus at `$TRANSCRIPTS_DIR/channels`, else the checkout's own `transcripts/channels`; it used to fall back to an absolute path that existed on one machine only. The song project's videos default to `~/reports/quartering-uh-song/videos`; `SONG_DIR` and `VIDEO_ROOT` still win. The song project's tracked manifests record their paths relative to the song folders, and the twenty one-off `umtool/song/*.sh` run logs, which only ever ran on the machine that wrote them, are gone.
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,5 +1,8 @@
# Changelog
+## [Unreleased]
+- **The charts count every transcript, once the site is rebuilt.** A transcript that arrived after its video was first indexed was missing from the charts' transcript and cue counts and from "Transcribed over time", and a video with YouTube captions alone had no transcription date. Both are counted now, and a captioned video is dated by when its captions arrived.
+
## [0.10.0] - 2026-09-28
- **A video whose recheck failed shows as possibly missing rather than available.** When a video drops out of its channel's listing it is marked "Missing?" until a recheck says why. A recheck that could not reach the video — a blocked request or a network error — used to clear the mark as if the video had been found. It now leaves "Missing?" in place until a recheck actually reaches the video. Needs a rebuild and deploy of every export site.
- **The hub's Ask AI says when the hub has no archives.** On a hub whose list of archives is empty or could not be read, with none added in this browser, the question box read "Loading transcripts…" forever. It is now disabled, and one line under it says the hub has no archives yet, with a link to the front page, where one can be added.
diff --git a/homepage/CHANGELOG.md b/homepage/CHANGELOG.md
@@ -2,6 +2,7 @@
## [Unreleased]
+- **A site's card counts every transcript, and never shows 0 channels while it serves recordings.** A transcript that arrived after its video was first indexed, or a video with YouTube captions alone, could be left out of the family's numbers: one site served 1,889 recordings and its card said 0 transcripts, 0 channels and 0 hours. Such transcripts are counted now — in the card, the family totals and the archive-growth chart — and one with no transcription date is left off only what is placed by that date: the charts by transcription date, "this month" and the recent list. The official-instance figures on the hub move with them.
- **The source is on the site, with its history: `/source/`.** A new **Source** page (and nav entry) gives `git clone https://archilyzer.pages.dev/source/archilyzer.git`, a read-only mirror of the main branch regenerated with every deploy, with its head, the private commit it reflects, a link to browse every file raw at `/source/tree/`, and the tarball with its size and sha256. Commit ids differ from the private repository's, because machine paths are scrubbed on the way out, and the page says so. A build without a published source says "No source published in this build." instead of offering a clone. The Downloads tarball is now regenerated by every build (its commit is the mirror's), and the page points at the mirror for history. The docs that said there is no public repository (*Install*, the FAQ, *What is Archilyzer*) now say how to clone. Below `md` the header's nav drops to its own row, as it did below `sm`, because five labels no longer fit beside the wordmark. `_headers` serves the raw tree as plain text.
- **The docs' *Building several sites at once* page says what Build all does.** It called the container pipeline opt-in, turned on in the settings. Build all sites builds every site in parallel in containers whenever a container engine is available, and one after another when none is; there is nothing to switch on.
- **A single-colour social icon shows on every ground.** The footer's social icons are the operator's (`homepage.json`'s, else `settings.socialLinks`), normalized when they are saved (`normalizeSocialSvg`, release 11 slice O1). An icon drawn in one colour now takes the footer's colour throughout; before, a part that carried its own colour kept it, so X's official logo, which is white, was invisible on the Light ground. An icon of two or more colours, such as YouTube's red mark with its white triangle, keeps its colours as pasted. "No fill", gradients, masks, clip paths and animation timing are never changed, and a clip path's own colour does not count, so a one-colour icon exported from Figma follows the footer too. It applies when the settings are next saved, then needs a rebuild and deploy of the homepage.
diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts
@@ -15,6 +15,7 @@ import type {
} from "yt-dlp-transcript-common/lib/posts";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
+import type { VideoStat } from "yt-dlp-transcript-common/lib/stats";
import type { ChannelGroup } from "yt-dlp-transcript-common/lib/channelGroups";
import { newGroup, newLeaf } from "yt-dlp-transcript-common/lib/searchQuery";
import type {
@@ -1296,6 +1297,81 @@ test("server: get_video_metadata reports a missing id as an error", async () =>
await client.close();
});
+// The "## Stats" block is read straight off the archive's stats pages, so its
+// "covers only N% — truncated" warning is exactly as good as the stat. Before
+// stats schema 6 the stat of a video transcribed after it was first indexed
+// stayed at "no transcript, coverage 0" for good, and this warned that a
+// complete transcript was truncated. buildStats.test.ts (a) pins the stat now
+// being recomputed; this pins what the tool then says about it.
+function statFor(
+ record: TranscriptDetail,
+ s: Pick<VideoStat, "hasTranscript" | "cueCount" | "coverage" | "transcribedDate">,
+): VideoStat {
+ return {
+ slug: record.slug,
+ id: record.id,
+ channelSlug: record.channelSlug,
+ channel: "Channel A",
+ title: record.title,
+ platform: "youtube",
+ uploadDate: record.uploadDate,
+ downloadedDate: "20260711",
+ timestamp: null,
+ duration: 3600,
+ viewCount: null,
+ likeCount: null,
+ commentCount: null,
+ channelFollowerCount: null,
+ categories: [],
+ tags: [],
+ language: null,
+ isLivestream: false,
+ mediaType: "video",
+ status: "available",
+ ...s,
+ };
+}
+
+class StatsStubSource extends StubSource {
+ constructor(private stat: VideoStat) {
+ super();
+ }
+ async statsIndex(): Promise<ReadonlyMap<string, VideoStat>> {
+ return new Map([[this.stat.slug, this.stat]]);
+ }
+}
+
+test("server: get_video_metadata on a recomputed stat reports the cues and no truncation", async () => {
+ const a1 = CHAN_A[0];
+ const client = await connectClient(
+ new StatsStubSource(
+ statFor(a1, { hasTranscript: true, cueCount: 7461, coverage: 0.998, transcribedDate: "20260918" }),
+ ),
+ );
+ const out = firstText(
+ await client.callTool({ name: "get_video_metadata", arguments: { video_id: "a1" } }),
+ );
+ assert.match(out, /## Stats/);
+ assert.match(out, /- transcript cues: 7461/);
+ assert.match(out, /- transcribed: /);
+ assert.doesNotMatch(out, /COVERS ONLY/);
+ await client.close();
+});
+
+test("server: get_video_metadata still warns for a transcript that really stops early", async () => {
+ const a1 = CHAN_A[0];
+ const client = await connectClient(
+ new StatsStubSource(
+ statFor(a1, { hasTranscript: true, cueCount: 900, coverage: 0.41, transcribedDate: "20260918" }),
+ ),
+ );
+ const out = firstText(
+ await client.callTool({ name: "get_video_metadata", arguments: { video_id: "a1" } }),
+ );
+ assert.match(out, /TRANSCRIPT COVERS ONLY 41% OF THE RUNTIME/);
+ await client.close();
+});
+
test("server: open_link decodes AND searches in one call, returning the handle", async () => {
const client = await connectRegistry();
const out = firstText(
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -655,7 +655,10 @@ chunk they agree to within ~2%.
#### The chunk census
Computed free from `statsByPath.cueCount` — present on **all 73,367** transcribed
-videos, so it needed no GPU time and no transcript reads. At the shipped config
+videos, so it needed no GPU time and no transcript reads. (Corrected 2026-09-28: "all" was
+all the CACHE knew of. Until stats schema 6 a video transcribed after its stat was first
+cached kept `cueCount: null` — about 2,500 of them on 2026-09-28; see "The stats cache key"
+at the end of this file.) At the shipped config
(`maxCues` 600, overlap 40, step 560):
```
@@ -5702,7 +5705,7 @@ Every `file:line` below was grepped at `7dfd7508`, whose code is byte-identical
- Export pages (buildIndex, buildStats, compose-homepage): compact, no newline.
- Chart, alias and tag stores: indented, no newline.
- Everything else: indented, with a newline.
- The six local `function writeJsonAtomic` left in `buildIndex.ts:404`, `buildStats.ts:238`,
+ The six local `function writeJsonAtomic` left in `buildIndex.ts:404`, `buildStats.ts:376`,
`bin/compose-homepage.ts:40`, `aliasesStore.ts:23`, `chartsStore.ts:50` and
`curatedTagsStore.ts:61` are one-line wrappers that pin those bytes over the shared writer.
They are not copies.
@@ -5807,7 +5810,7 @@ Every `file:line` below was grepped at `7dfd7508`, whose code is byte-identical
- **`readChannelConfigFile(file)`** `:165-170`. It never throws, and answers null for absent,
unreadable, not JSON, or not a channel. `readChannelConfig(paths, slug)` `:172` wraps it,
- and `buildIndex.ts:296` and `buildStats.ts:158` call it directly. The header `:152-157` names
+ and `buildIndex.ts:296` and `buildStats.ts:285` call it directly. The header `:152-157` names
the raw readers that bypass it: `channelMedia.ts:179` (the `dataDir` guard) and the legacy
migrations (`migrateToSites.ts:101` for `group`, `bin/migrate-channel-priority.ts` for
`excludeFromSync`, which also WRITES raw at `:206`).
@@ -5891,7 +5894,7 @@ which is the same race class 4b fixed for `config.json`.
Not in the 14:
-- The export build's streamed page writers `buildIndex.ts:971` and `buildStats.ts:220`. These
+- The export build's streamed page writers `buildIndex.ts:971` and `buildStats.ts:358`. These
are JSON writers still on the per-pid name, not in the list above only because there is one
writer per build. They are owed with the rest (16 + 2).
- `metadataScanStore.ts:188-206` and `autoQueueState.ts:141-156`. They carry a module-level
@@ -6139,7 +6142,7 @@ complete with this release.
- **The two write counters are deleted.** `git grep -n 'writeSeq\|nextWriteSeq' -- common
editor` is empty.
- **What `git grep -n 'tmp-${process.pid}' -- common editor` still finds:**
- - `buildIndex.ts:971` and `buildStats.ts:220`, the export page writers, out of scope;
+ - `buildIndex.ts:971` and `buildStats.ts:358`, the export page writers, out of scope;
- `transcode.ts:31` (ffmpeg's output, renamed at `:55`);
- `transcribeOne.ts:142` (the transcription app's `outputBase`);
- the shared writer's own comment `:9`, code `:136`, and test `:68`.
@@ -7317,3 +7320,114 @@ Slices Q (`4855f70b`) and R (`ffdeb2cd`): [`release-12.md`](release-12.md), the
- **The live :3001 editor runs its BUILT bundle.** Until it is rebuilt on a tree with release 12,
its `/sites` Homepage jobs have no source step, no withdrawal and no deploy check.
+## The stats cache key (verified 2026-09-28, branch `fix/stats-cache-key`)
+
+The record is [`stats-cache-key.md`](stats-cache-key.md). Anchors are at the branch tip. Two notes
+on anchors elsewhere in this file:
+- The branch added lines to `buildIndex.ts`: anchors above this section that point past `:66`
+ moved by +1, and past `:559` by +6. They are not rewritten in place.
+- The three `buildStats.ts` anchors in the slice-W sections were refreshed here.
+
+- **The stats cache is keyed on the metadata AND the index's own record for the video.**
+ - `statsByPath` (`common/controller/buildStats.ts`) holds `{metaMs, idx, stat}` per
+ `[channelSlug, videoDir]` (`:92`). A stat is recomputed when `metaMs` or `idx` moved (`:482`).
+ - `idx` is `indexSignature` (`:114`) of buildIndex's whole `mtimes` record: `metaMs`,
+ `transcriptMs`, `subsMs`, `availabilityMs`, `digestMs` (every input that makes buildIndex
+ re-process the video, `buildIndex.ts:585`) and `indexKey`. It is `NOT_INDEXED` (`"-"`,
+ `:107`) when the index has no record.
+ - The cues are read under the record's own `indexKey` (`:543`). The key buildIndex used when
+ `transcript.cues.json` is fresh comes from that file's `uploadDate`, which a later metadata
+ rewrite can differ from; the computed key is only the fallback for a video the index lacks.
+ - Cost on the unchanged path: one LMDB get per video, and no file I/O.
+ - **Until schema 6 the key was `metaMs` alone.** A transcript that arrived after a video was
+ first seen never reached its stat: Whisper days later, a Normalize run, or a stats run made
+ before `build:index` had the video. The pool composers make exactly that last kind of run:
+ `poolSummary.ts:76` runs `buildStats` with no index build. A later
+ `build:index && build:stats` then reported "added 0, changed 0" over those videos.
+ - **Why the whole record, not just the transcript mtime:** buildIndex does not re-index when only
+ `transcript.cues.json` changes, but a re-index for another reason (subs, availability, digest)
+ reads the fresher cues.json and changes the cue count. Keying on the whole record redoes the
+ stat then. The cost is a recompute on those rarer changes.
+ - **Race, benign:** buildStats reads the record before the cues, and buildIndex writes the cues
+ (`buildIndex.ts:720`) before `mtimes` (`:841`). A concurrent index build can only pair an
+ older record with newer cues, which the next run recomputes.
+- **Two kinds of video have no index record, and they are counted apart.**
+ - buildIndex writes `meta.scannedAt` (`INDEX_SCANNED_AT_KEY`, `common/lib/stats.ts:19`) at
+ `buildIndex.ts:2006`, when a build completes. The value is the time its scan began (`:564`).
+ - **`notIndexedYet`:** metadata newer than `scannedAt`, or no build has completed yet. The video
+ was downloaded since, and it heals on the first stats run after the next index build.
+ - **`notIndexable`:** older than `scannedAt`, and the index has no record. The build did not
+ index it:
+ - no `upload_date` (`buildIndex.ts:701`);
+ - a processing failure;
+ - or the channel's media was unreachable during that build, since buildIndex has no drive
+ guard.
+
+ It stays until fixed, and is logged as such rather than as pending (`:500`).
+- **A transcript always has a date, and a caption video takes its captions' arrival.**
+ `resolveAcquisitionDates` (`:227`) tries, in order:
+ 1. transcribe-outcome's `transcribedAt`;
+ 2. the mtime of the picked index transcript, which is `transcript.json`, else the caption VTT (`:241`);
+ 3. `transcript.cues.json`;
+ 4. `downloadedDate` (`:245`).
+
+ Whisper videos resolve as before. The one difference is an outcome sidecar whose date will not
+ parse: it now falls through instead of giving null.
+- **The downgrade guard.** A build never clears a cache that a NEWER schema wrote (`:424`). It
+ throws, naming both versions and `ARCHILYZER_STATS_ALLOW_DOWNGRADE` (`:129`). That variable is
+ declared in `envVars.ts` for a deliberate rollback.
+ - The guard helps FUTURE bumps only. Schema-5 code has no guard, and would clear a schema-6 cache.
+ - `STATS_SCHEMA_VERSION` is 6 (`stats.ts:11`). It versions the cache. The pages' version,
+ `STATS_MANIFEST_VERSION`, stays 1.
+ - `digestPlan.ts:395,447` and `duplicateShorts.ts:228` only warn on a mismatch, and read
+ `value.stat` alone.
+- **An unmounted media drive is not an empty channel, for stats either.**
+ - `scanSource` asks `inspectChannelMedia` per channel (`:294`). A channel that is not `ok` or
+ `in-place` is "held": not rescanned, its cached stats kept and published (`:509`), and logged.
+ - A schema clear with any channel held REFUSES (`:443`), because the clear would drop that
+ channel's stats for good.
+ - The message names each held channel, with its storage location's label and no path.
+ - It lists the ways out, mounting first: mount its media; repair or re-point its location on
+ /storage; finish or clear its move; or, for a channel gone for good, delete it or set
+ `excludeFromBuild`. An excluded channel is skipped before the check.
+ - This is the build's own guard. The job registry's `needsMedia` check
+ (`streamCommand.ts refuseForUnreachableMedia`) is per channel and needs a `channelSlug`, so it
+ never covered this pool-wide build, from the editor or from the CLI.
+ - **buildIndex has no such guard.** An index build with a drive unmounted drops those channels'
+ index records, and the site pages built from it lose them.
+ - Its `Diff: … -R removed` line (`buildIndex.ts:641`) shows it.
+ - A proper hold there is a follow-up slice (`stats-cache-key.md`, "Left").
+- **One stats build at a time: an operator rule, not a lock.**
+ - Two concurrent runs are harmless unless one clears the cache (a schema change) after the other
+ has scanned. The other then collects a partly refilled `statsByPath` and publishes truncated
+ pages.
+ - There is no cross-process lock primitive in `common/`: `scripts/queue-lock.mjs` is the e2e
+ queue's flock wrapper.
+ - The editor's build jobs share the queue `"build"` by default. The per-button queue fields can
+ split them.
+ - **The CLI is outside every queue.**
+- **The homepage fold counts a transcript that has no date** (`common/lib/homepageSummary.ts:407-408`).
+ It counts toward totals, channels, hours, the card's `transcribed.total` (`withUndated`) and
+ upload-month placement. Only the transcribed series, "this month" and the recent rail need the
+ date.
+- **MCP `get_video_metadata`** prints its "## Stats" block from the stats pages
+ (`mcp/src/server.ts:2273`). **Still owed:** a video with no transcript has coverage 0 and gets the
+ "covers only 0% — truncated" note (`coverageNote`, `:2309`).
+- **A caption test fixture must carry YouTube's inline timing tags.** `parseVtt` (`vtt.ts:30`) keeps
+ only cue lines containing `<hh:mm:ss.mmm>` (`TIMING_TAG_RE`, `:3`). A plain `WEBVTT` cue parses to
+ zero cues. `maybeMissingBuild.test.ts`'s VTT is such a file, which is harmless there.
+ - The same rule makes a video whose English track is a *manual* caption (no inline tags) index
+ as 0 cues. That is rare: 0 in 3,000 sampled of the-quartering, 6 of chibi-reviews.
+- **Measured before the fix,** on the whole-pool stats of 2026-09-28T20:40Z:
+ - 49,798 transcripts shown, of about 77,000 on disk;
+ - 24,710 records with `hasTranscript` but no `transcribedDate`;
+ - 2,484 with a stale `hasTranscript: false`;
+ - Jasolyzer 0 of 1,889.
+- **The schema bump's cost:** one full re-extraction. That is 79,500 videos and 39.3 GB of
+ metadata, measured at 149 MB/s on NVMe. About a quarter of the video dirs are on a USB drive, at
+ 4–5 random reads per recompute. Expect **about 10–30 minutes, longer with a cold cache**.
+ - The pass commits per batch of 200.
+ - It is interruptible, and it resumes: the schema is written at the clear, so the next run
+ finishes the rest.
+ - Run in the editor, it stalls the editor's event loop for the length of the pass. Prefer the CLI
+ with the editor idle.
diff --git a/plans/STATE.md b/plans/STATE.md
@@ -3,6 +3,31 @@
The working memory for the local-AI derived-corpus work. Rewritten at the end of every
session, before context is cleared. See [`README.md`](README.md) for the protocol.
+**Now (2026-09-28, night): the stats cache key fix — built, reviewed (SHIP AFTER FIXES, then SHIP
+on re-review; every touch-up done), not merged.** The branch is `fix/stats-cache-key`, and [`stats-cache-key.md`](stats-cache-key.md)
+holds the record, the review and the rollout. FACTS has "The stats cache key".
+- **What it fixes:** the homepage showed Jasolyzer as 0 transcripts, 0 channels, 0 hours while it
+ served 1,889 videos. Instance-wide it showed 49,798 transcripts of about 77,000.
+- **The cause:** `statsByPath` was keyed on the metadata mtime alone. It is now keyed on the index's
+ own record as well, and a transcript always has a date.
+- **Added by the review:**
+ - a guard against clearing a newer cache (`ARCHILYZER_STATS_ALLOW_DOWNGRADE`);
+ - an unmounted drive's stats are kept, and a cache clear with one refuses;
+ - "not indexed yet" and "not indexable" are counted apart.
+- **Owed after the merge:** the rollout in the record, in its order. First, rebuild and restart
+ :3001 (until then, never press "Build stats dataset"). Then index, then one full stats pass of
+ 10–30 min, then the homepage, the hub and the sites. The homepage deploy waits on release 12's
+ step 0: it runs the source publish.
+- **Merge note:** against `homepage/social-visible` (tip `afc642fd`) the only conflict is
+ `homepage/CHANGELOG.md`'s `[Unreleased]`. Keep both sides. Against `main` there is none.
+- **FOLLOW-UP, its own slice: the index build still treats an unmounted drive as an empty
+ channel.** It drops that channel's index records, and the next site build publishes the channel
+ as gone.
+ - The fix: give `buildIndex` the stats build's hold, or at least a refusal with an override.
+ - Schedule it before routine builds resume after this rollout.
+ - Until then, the rollout's step 3 `Diff:` check is the safeguard: thousands removed means a
+ drive was missing.
+
**Now (2026-09-28, evening): release 12 — the source mirror — is merged to `main` and NOT rolled
out.** [`release-12.md`](release-12.md) holds Q's and R's records, their reviews, "Merged" and
"Rollout". The operator's runbook is `~/reports/release-12/RUNBOOK.html`, with its scripts in
diff --git a/plans/stats-cache-key.md b/plans/stats-cache-key.md
@@ -0,0 +1,278 @@
+# The stats cache key (fix, 2026-09-28)
+
+Branch `fix/stats-cache-key` from `main` `ac438bbc`. Merged: not yet. Rolled out: not yet.
+
+## What was wrong
+
+- **The stats cache was keyed on the wrong file.** The cache (`statsByPath`, in the index LMDB)
+ was keyed on `metadata.info.json`'s mtime alone. But `hasTranscript`, `cueCount`, `coverage` and
+ `transcribedDate` come from the index and the transcript files.
+- **So a late transcript never reached its stat.** That covers a Whisper run days after the
+ download, a Normalize run weeks later, and a stats run made before `build:index` had the video.
+ The hub and homepage composers make that last kind of run.
+- **Caption videos could not be dated at all.** A caption-only video (a YouTube VTT, no
+ `transcript.json`, no outcome sidecar) got no `transcribedDate` even when computed fresh.
+- **The homepage then dropped every undated transcript.** Its fold required `hasTranscript &&
+ transcribedDate`. Jasolyzer served 1,889 videos and showed 0 transcripts, 0 channels, 0 hours.
+ Every site's counts and "Transcribed over time" charts were low.
+
+## Numbers, before and after (ESTIMATES)
+
+**How they were measured.** The whole-pool stats pages of 2026-09-28T20:40Z (`homepage/public/stats`)
+were classified record by record against the files on disk now. This was read-only: listings,
+`stat` and small JSON reads. No LMDB was opened and nothing was rebuilt.
+
+- **Shown** is the published homepage summary.
+- **After** counts the records a rebuild with this fix would count. These are:
+ - `hasTranscript` with a date source on disk;
+ - stale `hasTranscript: false` records whose `transcript.cues.json` has cues;
+ - caption-only records, which the new date fallback reaches.
+
+The real numbers come from the first rebuild.
+
+| Site | Records | Transcripts shown | Transcripts after (est.) |
+| --- | ---: | ---: | ---: |
+| Anilyzer | 29,836 | 8,370 | ~29,665 |
+| Jeralyzer | 32,994 | 29,983 | ~31,906 (+ up to ~290, see below) |
+| Bonnellyzer | 8,085 | 5,540 | ~7,244 |
+| Hasanalyzer | 3,432 | 3,048 | ~3,335 |
+| Rekietalyzer | 2,931 | 2,855 | ~2,923 |
+| Jasolyzer | 1,889 | 0 | ~1,751 (~3,808 h) |
+| Whole pool | 79,385 | 49,798 | ~76,990 |
+
+- **Why so many were missing:**
+ - 24,710 records had `hasTranscript` with no date. Of those, 23,198 have a date source on disk
+ now, and 1,512 are caption-only.
+ - 2,484 carried a stale `hasTranscript: false`.
+- **About 290 more are probably stale but are not counted above.** These are records with large
+ caption files and `cueCount: null` (Jeralyzer's paramount-tactical, and the pool-only
+ candace-owens). The live Jeralyzer shards serve cues for the three that were checked.
+- **The whole-pool total includes pool-only channels,** so it is more than the sum of the sites.
+
+## The fix
+
+| Commit | What |
+| --- | --- |
+| `ad152529` | `common:` the key becomes the metadata mtime AND buildIndex's `mtimes.transcriptMs`. `transcribedDate` falls back through the outcome sidecar, then the transcript the index read (`transcript.json`, else the caption VTT), then `transcript.cues.json`, then `downloadedDate`, so a transcript always has one. `STATS_SCHEMA_VERSION` 5 → 6. New `buildStats.test.ts`. |
+| `b7a733ad` | `common:` test (b) asserts the heal before the new count. |
+| `9a8bded7` | `common:` the homepage fold counts a transcript with no date (totals, channels, hours, the card's total, upload-month placement). Only the transcribed series, "this month" and the recent rail need the date. |
+| `e765b168` | `mcp:` `get_video_metadata`'s stats block on a recomputed stat: the real cue count and no false "truncated". |
+| `6c7ff664` | `plans:` the first record, FACTS, changelogs. |
+| `9bc3c635` | `common:` buildIndex records `meta.scannedAt`, the time its last completed scan began. |
+| `e4c61c77` | `common:` the review's fixes (see "Review" below): the downgrade guard, the whole index record as the key with the cues read under its `indexKey`, "not indexed yet" and "not indexable" counted apart, an unmounted drive's stats kept (and a clear refused), and test hygiene. |
+| `9e61b119` | `common(test):` `source.test.ts` expands the `~` the kept-scratch log prints (release 12's test). |
+| `000273c0` | `common:` test (i) asserts the kept stats before the new result field. |
+| `22ec383e` | `plans:` the record's review, gates and rollout; FACTS; STATE; the three changelogs. |
+| `b30b52c1` | `common:` re-review R3 and R4: the "not indexable" line names an unreachable drive, and the held-channel refusal names every way out, without paths. Test (z) asserts the LMDB file is under the temp root. |
+| this commit | `plans:` re-review R1, R2, R4 (record), R5 and the nits: the rollout's commands as typed, the index's `Diff:` check, the hub's summary check, numbered preconditions, and the index follow-up in the record and STATE. |
+
+- **Caption videos are dated by when their captions arrived** (the VTT's mtime), not by a later
+ Normalize run. Whisper videos resolve as before. Their "Transcribed over time" curves move: on
+ Jasolyzer, 1,683 videos would otherwise all have landed on the Normalize day.
+- **The published stats page format did not change.** `STATS_MANIFEST_VERSION` stays 1, and
+ `HOMEPAGE_SUMMARY_VERSION` stays 5.
+- **The compose paths warn and do not refuse.** `compose-hub` and `compose-homepage` still run
+ `buildStats` against the index as it stands, and they still do not build the index.
+ - A video they meet before the index has it is keyed `NOT_INDEXED`. It is recomputed on the first
+ run after the next index build.
+ - The log counts separately the videos downloaded since the last index build and the ones older
+ than it that it did not index: no `upload_date`, a failure, or their channel's media unreachable
+ during that build.
+ - A refusal would stop these builds whenever the editor had downloaded since the last index build,
+ which is almost always.
+- **Concurrent stats builds: an operator rule, not a lock** (review L3, brief item 5).
+ - There is no cross-process lock primitive in `common/`. `scripts/queue-lock.mjs` is the e2e
+ queue's flock wrapper, run as a separate holder process. Building one here would be a new
+ lock, with its own stale-lock story.
+ - The rule is: **one stats build at a time.** The editor's build jobs share the queue `"build"`
+ by default. **The CLI is outside every queue.**
+ - Two concurrent runs are harmless unless one clears the cache, which only a schema change does.
+
+## Review
+
+**Verdict: SHIP AFTER FIXES** (the review is `j-review.md` in the job's scratch). The reviewer found
+no code defect; every fix was made on this branch.
+
+| Finding | Where |
+| --- | --- |
+| L1: the record overstated which paths run old code | `22ec383e`: the rollout says only the in-process **Build stats dataset** runs old code, and step 1 is "rebuild the editor bundle, then restart"; the editor changelog says the same. |
+| L2: wrong rollout commands | `22ec383e`: every command checked against `pnpm archilyzer --help` and `pnpm ops --help`. The homepage and the hub are each built and deployed ONE way, the hub before the sites, and `build-deploy` takes `{"all":true}`. |
+| L3: concurrent runs; the removable drive | `e4c61c77`: an unmounted drive's channel is held and its stats kept, and a cache clear with one refuses. Rollout step 2 is the precondition. The lock was **left**: see "Concurrent stats builds" above. |
+| L4: test hygiene | `e4c61c77`: every `getPaths()` path is pinned under the temp root, the root is removed in `after`, and case (z) spies on node:fs and node:fs/promises (async and sync) and fails on any write outside it. |
+| L5: the "not in the index yet" line was wrong for skipped videos | `9bc3c635` + `e4c61c77`: `notIndexedYet` and `notIndexable`, logged apart, with case (h). |
+| L6: the time estimate left out the USB drive | `22ec383e`: 10–30 minutes, with the reason; interruptible and resumes. |
+| L7: no homepage changelog bullet | `22ec383e`: `homepage/CHANGELOG.md`, and the export bullet says "once the site is rebuilt". |
+| The guard (ruled: add it) | `e4c61c77`: a build refuses to clear a cache a newer schema wrote, and names both versions and `ARCHILYZER_STATS_ALLOW_DOWNGRADE`. The variable is declared in `envVars.ts`; `ENVIRONMENT.md` is regenerated and `--check` is clean. Case (j) covers older → cleared, newer → refused (also the CLI, exit non-zero, cache untouched), and the override → cleared. |
+| O1 (ruled: do it) | `e4c61c77`: the cues are read under the `mtimes` record's `indexKey` (case (f)), and the key is the whole record (case (g)). About 40 lines with comments, and no new I/O on the unchanged path: the same one LMDB get per video. |
+| O2: date from the index's mtime | **Left.** The fresh readdir happens only on the recompute path, and it keeps the Whisper rule byte-identical to before. |
+| O3: the spy saw only async fs | `e4c61c77`: the spy now covers node:fs sync and callback APIs too. |
+| O4: the reviewer's extra cases | **Partly.** Case (e) now includes an index rebuild with no churn. The removed-transcript and two-channel cases stay in the reviewer's scratch, where they pass. |
+| O5: manual captions parse to 0 cues | **Left**, noted in FACTS. |
+| `source.test.ts` under a home `TMPDIR` | `9e61b119`: 13 of 13 with `TMPDIR` unset and with it under `~`. |
+
+**The new cases fail on the pre-review code** (`6c7ff664`'s `buildStats.ts`, `buildIndex.ts` and
+`stats.ts` swapped in once):
+
+| Case | Result |
+| --- | --- |
+| (b) | `undefined` for `notIndexedYet` (the field did not exist) |
+| (e) | `NaN` counts, for the same reason |
+| (f) | `hasTranscript` false: the cues missed under the metadata's upload date |
+| (g) | changed 0, not 1: the cue-count drift |
+| (h) | the fields did not exist |
+| (i) | removed 2, not 0: the drive's stats were dropped |
+| (j) | "Missing expected rejection": the newer cache was cleared |
+
+(a), (c), (d) and (z) pass there, as they should: (a) to (d) were fixed before the review.
+
+### Re-review
+
+**Verdict: SHIP.** No code defect; five Low touch-ups and two nits, all made:
+
+| Finding | Where |
+| --- | --- |
+| R1: `archilyzer` is not on PATH | this commit: every rollout command is written as typed from the primary checkout's root, `pnpm archilyzer …`, as release 12's records spell it. Nothing else on the branch spells a bare command: the code's messages name none, and FACTS and the changelogs name scripts or pages. |
+| R2: step 3 had no post-check | this commit: the index step now reads its `Diff:` line. Thousands removed means a drive was missing; mount it and re-run the index before step 4. |
+| R3: "not indexable" blamed the video for a missing drive | `b30b52c1`: the line adds "or its channel's media was unreachable during that build (run an index build with every drive mounted)", and case (h) expects it. |
+| R4: the refusal gave one way out | `b30b52c1` (message) and this commit (record): mount its media first; else repair or re-point its location on /storage, finish or clear its move, or delete the channel or set `excludeFromBuild`. The message carries no path: a held channel is named with its location's label. Case (i) asserts both. |
+| R5: a hub build with no figures still deploys | this commit: step 6 builds the hub, checks the log for `hub-summary.json covers N official instance(s)`, and deploys only then. `hub-summary.json skipped: …` means stop. |
+| Nits | this commit numbers the preconditions; `b30b52c1` asserts `paths.lmdbPath` is under the temp root in case (z). |
+| Follow-up (recommended) | recorded under "Left" and in STATE: **the index build still treats an unmounted drive as an empty channel.** It needs its own slice before routine builds resume. Until then, R2's check is the safeguard. |
+
+## Gates (worktree, 2026-09-28, at the review fixes)
+
+- **tsc:** clean (43 s).
+- **Common tests, with `TMPDIR` unset:** **2,161, all pass**. That is 2,149 at the branch point,
+ plus 11 buildStats cases and 1 homepage fold case.
+- **Other unit suites:**
+ - editor unit: 85 of 85;
+ - `test:scripts`: 185 pass, 1 skipped;
+ - mcp: 271 of 271;
+ - homepage unit: 7 of 7.
+- **Docs:** `pnpm archilyzer docs env --check` is clean.
+- **Builds:** `next build` succeeded for export (36 s), editor (55 s) and homepage (23 s).
+- **e2e:**
+ - Before the review (at `6c7ff664`):
+ - editor `duplicate-shorts`, `build`, `site-scope`, `sites-homepage` and `deploy-page`: 21
+ passed, 0 failed, 2.3 min;
+ - export `charts.spec.ts`: 8 passed, 28 s;
+ - homepage, full suite: 36 passed, 1.0 min.
+ - After the review fixes: the same editor list again, because `duplicate-shorts` drives Build
+ index and Build stats dataset: 21 passed, 0 failed, 1.2 min.
+ - Export and homepage were not rerun. The review fixes change no export or homepage code; the
+ homepage fold is unchanged since `9a8bded7`.
+
+- **At the re-review touch-ups (`b30b52c1`):**
+ - tsc is clean;
+ - `buildStats.test.ts` and `envVars.test.ts` pass 20 of 20;
+ - `pnpm archilyzer docs env --check` is clean;
+ - no e2e was run, since only wording and one assertion changed.
+
+## Rollout (operator) — follow it literally, in this order
+
+Every command below is typed **from the primary checkout's root**. There is no `archilyzer` on
+PATH, so it is `pnpm archilyzer …`.
+
+**What runs which code.**
+- **The live :3001 editor runs its BUILT bundle** until it is rebuilt and restarted. The only
+ stats path that runs inside that bundle is the **Build stats dataset** button (`buildStatsAction`,
+ in-process). On the old code it has no guard: against the new cache it would clear it and refill
+ it the old way.
+- **Everything else spawns the checkout's code from disk**, so it runs the new code the moment
+ `main` has this merge:
+ - a site build's data phase (`pnpm run build:data`);
+ - the hub (`compose:hub`) and the homepage (`compose`);
+ - every CLI command.
+- So **the first of those after the merge is the first schema-6 stats run.** It clears the cache
+ and does the whole pass inside that job.
+
+**Preconditions for steps 3 and 4.**
+1. **The removable media drive is mounted.** `/storage` shows every location **Available**.
+ - The stats build now refuses a cache clear while any channel's media is unreachable.
+ - **The index build has no such guard.** Run with the drive absent, it drops those channels
+ from the index, and the next site build publishes them as gone. Step 3's `Diff:` check is what
+ catches it.
+2. **No other index, stats or site build is running.**
+ - `/jobs` shows no `build-index`, `build-stats`, `build-site`, `build-deploy`, `build-hub` or
+ `build-homepage` job running or queued, on any queue.
+ - No CLI or spawned build is running:
+ `pgrep -af 'archilyzer\.ts (index|build|compose)'` prints nothing. Every CLI build and every
+ spawned data phase or compose goes through `archilyzer.ts`; the in-process editor jobs do not
+ show here, and `/jobs` covers them.
+
+**The steps.**
+
+1. **Rebuild the editor bundle, then restart :3001 onto it:** `pnpm --filter editor build` in the
+ primary checkout, then restart the editor the way it is normally run. Between the merge and this
+ restart:
+ - **never press Build stats dataset**;
+ - **start no site, hub or homepage build**. It would do step 4's full pass itself, inside that
+ job, unannounced.
+2. **Check preconditions 1 and 2 above.**
+3. **Index.** Use `pnpm archilyzer index`, or `/sites` → **Build index**
+ (`pnpm ops build-index --wait`). It writes the same LMDB as the editor, so run it only with no
+ build job running (precondition 2).
+
+ **Then check its `Diff:` line**, `Diff: +A added, ~C changed, -R removed, N total.`:
+ - with `pnpm archilyzer index` or `pnpm ops build-index --wait`, it is in the terminal output;
+ - from the button, it is in the `build-index` job's log on `/jobs`.
+
+ **R should be 0, or a handful.** Thousands removed means a drive was missing during the build,
+ and those channels just left the index. Stop, mount the drive (precondition 1), and run step 3
+ again before step 4.
+4. **Stats.** Use `pnpm archilyzer build stats`: the CLI, with the editor idle. The in-process
+ button stalls the editor for the length of the pass.
+ - The first run logs `Stats schema change (5 -> 6); clearing stats cache.` and re-extracts every
+ video: **about 10–30 minutes, longer with a cold cache.** About a quarter of the video dirs
+ are on the USB drive, at 4–5 random reads each.
+ - It can be interrupted (Ctrl-C, or cancelling the job) and **resumes**: the schema is written at
+ the clear, so the next run only finishes the rest.
+ - **If it refuses because a channel cannot be read,** the message names the channel and its
+ location. The ways out, in order:
+ 1. mount its media and run step 4 again;
+ 2. repair or re-point its location on `/storage`;
+ 3. finish or clear its move, from the channel's Storage panel;
+ 4. if the channel is gone for good, delete it or set `excludeFromBuild` in its config.
+ - **Let it finish before step 5.**
+5. **Homepage.** Use exactly ONE of:
+ - `pnpm archilyzer build homepage && pnpm archilyzer deploy homepage`;
+ - `pnpm ops build-homepage --json '{"deploy":true}' --wait`.
+
+ Building the homepage **also runs release 12's source publish**. So the homepage waits on
+ release 12's rollout step 0: the denylist is complete and `source publish --check` is clean.
+ The homepage and source in `main` at that moment must be the ones the operator has judged.
+6. **Hub, before the sites** (in basic mode the hub and the sites share `export/out`). Build it,
+ check its log, and only then deploy it. Use exactly ONE pair:
+ - `pnpm archilyzer build hub`, then `pnpm archilyzer deploy hub`;
+ - `pnpm ops build-hub --wait`, then `pnpm ops deploy-hub --wait`.
+
+ **Between the two,** the build's output (the compose line, `compose-hub: …`) must end with
+ `hub-summary.json covers N official instance(s)`, where N is the number of public sites (6 today).
+ **`hub-summary.json skipped: …` means the hub would deploy with no figures on its cards.** Stop
+ and fix the cause it names before deploying.
+7. **The six sites:** `pnpm ops build-deploy --json '{"all":true}' --wait`.
+
+**Live check.**
+- The homepage's Jasolyzer card shows about 1,751 transcripts, 1 channel and about 3,808 hours.
+- `https://jasolyzer.pages.dev/stats/page-0000.json` has no record with `hasTranscript: true`
+ and `transcribedDate: null`.
+- Step 4's log has no `Channel …: … cached stat(s) are kept` line, which would mean a held
+ channel.
+
+## Left
+
+- **MCP "truncated" for no transcript at all.** A video with no transcript has coverage 0
+ (`transcriptCoverage(undefined, d > 0)`), so MCP `get_video_metadata` tells it "covers only
+ 0% — truncated". This is older than the cache bug and out of scope. The coverage should be null
+ when there are no cues.
+- **FOLLOW-UP, its own slice: the index build still treats an unmounted drive as an empty
+ channel.**
+ - What it does now: it drops that channel's index records, and the next site build publishes the
+ channel as gone.
+ - What it needs: `buildIndex` gets the same hold the stats build now has (keep the channel's
+ `mtimes`, cues and pages), or at least a refusal with an override.
+ - When: schedule it before routine builds resume after this rollout.
+ - Until then, the rollout's step 3 `Diff:` check is the safeguard.
+- **A cross-process lock for builds** (see "Concurrent stats builds" above). The rule stands in
+ for it.
+- **O5:** a manual English caption (no inline timing tags) indexes as 0 cues (FACTS).