Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 58d877e65bb26e6325232da5a04e920da35cfcc1
parent cc70080585a9fe8c0810d1eb85052edbc57a3cd9
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 28 Sep 2026 18:42:32 -0400

common: buildStats refuses a newer cache, keys on the whole index record, keeps an unmounted drive's stats

Review of the stats cache key fix:

- A build never clears a stats cache a NEWER schema wrote: it throws, naming
  both versions and ARCHILYZER_STATS_ALLOW_DOWNGRADE (declared in envVars.ts,
  ENVIRONMENT.md regenerated) for a deliberate rollback.
- The key is the whole `mtimes` record (every input that makes buildIndex
  re-process the video, and its indexKey), and the cues are read under that
  indexKey: closes the cueCount drift and the uploadDate key split.
- "Not indexed yet" (metadata newer than the last index scan) and "not
  indexable" (the scan saw it and skipped it) are counted and logged apart, so
  a video with no upload_date is not announced as pending on every run.
- A channel whose media is not reachable (inspectChannelMedia) is not
  rescanned and its cached stats are kept; a cache clear with one refuses.
  The job registry's needsMedia guard is per channel and never covered this
  pool-wide build, from the editor or the CLI.
- buildStats.test.ts: cases (f)-(j) for the above, every getPaths() path
  pinned under the temp root, the root removed after, and an fs spy (node:fs
  and node:fs/promises, async and sync) asserting no write lands outside it.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
MENVIRONMENT.md | 1+
Mcommon/controller/buildStats.test.ts | 375++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-------------
Mcommon/controller/buildStats.ts | 224+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------------------
Mcommon/lib/envVars.ts | 1+
4 files changed, 490 insertions(+), 111 deletions(-)

diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md @@ -75,6 +75,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not | `TRANSCRIPT_PLATFORM_LINKS` | off | `1` cites platform watch pages instead of the archive's own pages. | common/lib/archive/reader-fs.ts | | `AUDIO_CHECK_RESUME_DURING_PROBE` | the channel's `audioCheck.resumeDuringProbe` | `1` or `true` resumes yt-dlp during the audio check's probe, anything else holds it, for a one-off comparison run; unset = the channel's setting. | common/ytdlp/audioCheckedDownload.ts | | `AUDIO_CHECK_BACKOFF_FACTOR` | the built-in factor | The audio check's interval backoff factor, in (0, 1], for a one-off run. | common/ytdlp/audioCheckedDownload.ts | +| `ARCHILYZER_STATS_ALLOW_DOWNGRADE` | off | `1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions. | common/controller/buildStats.ts | | `MCP_IO_STATS` | off | `1` turns on per-call I/O accounting, for `mcp/bench`. | common/lib/archive/io-stats.ts | | `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's `fetch_clip` talk to. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool | | `ARCHILYZER_AGENT` | `cli` | Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`. | scripts/archilyzer-ops.mjs | diff --git a/common/controller/buildStats.test.ts b/common/controller/buildStats.test.ts @@ -8,56 +8,132 @@ // transcript.json, no outcome sidecar) could never be dated at all. Measured on // a real corpus: one site served 1,889 videos and the homepage said 0 // transcripts, 0 channels, 0 hours. These cases pin the fix: the key also holds -// the index's transcript record, and a transcript always has a date. +// the index's own record for the video, and a transcript always has a date. // // Run with: node_modules/.bin/tsx --test common/controller/buildStats.test.ts -import { test } from "node:test"; +import { after, test } from "node:test"; import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; import { createRequire, syncBuiltinESMExports } from "node:module"; import { mkdirSync, mkdtempSync, + readFileSync, + renameSync, rmSync, + symlinkSync, utimesSync, writeFileSync, } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; +import { fileURLToPath } from "node:url"; -// getPaths() is lazy and cached, and nothing above calls it at import time, so -// pointing the whole path graph at a temp root here isolates this file's -// process from the real corpus (the maybeMissingBuild.test.ts pattern). +// EVERY PATH getPaths() CAN RESOLVE TO A PLACE THIS FILE'S CODE MAY WRITE IS +// PINNED UNDER ROOT, before anything calls it (it is lazy and cached — the +// maybeMissingBuild.test.ts pattern). From common/lib/paths.ts: TRANSCRIPTS_DIR +// (the LMDB, channels, jobs), SAVED_VIDEOS_DIR, SITES_DIR, SETTINGS_FILE, +// EXPORT_PUBLIC_DIR, EXPORT_INDEX_DIR, EXPORT_BUILDS_DIR, EDITOR_CHANGELOG_FILE, +// EXPORT_CHANGELOG_FILE, CHARTS_CONFIG_FILE, SEARCH_ALIASES_FILE, +// CURATED_TAGS_FILE, ARCHILYZER_CONFIG_DIR, ARCHILYZER_SOURCE_SCRATCH. The rest +// of its variables name binaries and a URL, which nothing here runs. The last +// test proves no write this file caused landed outside ROOT. const ROOT = mkdtempSync(path.join(tmpdir(), "build-stats-")); -process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts"); -process.env.EXPORT_PUBLIC_DIR = path.join(ROOT, "public"); -process.env.EXPORT_INDEX_DIR = path.join(ROOT, ".export-index"); -process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); +const PINNED: Record<string, string> = { + TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), + SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), + SITES_DIR: path.join(ROOT, "transcripts", "sites"), + SETTINGS_FILE: path.join(ROOT, "settings.json"), + EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), + EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), + EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), + EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"), + EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"), + CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"), + SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"), + CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"), + ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), + ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"), +}; +Object.assign(process.env, PINNED); +delete process.env.ARCHILYZER_STATS_ALLOW_DOWNGRADE; +after(() => rmSync(ROOT, { recursive: true, force: true })); const { getPaths } = await import("../lib/paths"); const { buildIndex } = await import("./buildIndex"); -const { buildStats } = await import("./buildStats"); +const { buildStats, STATS_DOWNGRADE_ENV } = await import("./buildStats"); const { normalizeTranscript } = await import("./normalizeTranscript"); const { readStatsPages } = await import("./poolSummary"); +const { STATS_SCHEMA_VERSION } = await import("../lib/stats"); +const { open } = await import("lmdb"); const paths = getPaths(); const CHANNEL = "test-channel"; +const DRIVE_CHANNEL = "drive-channel"; const SITE = "testsite"; const POOL = path.join(ROOT, "pool-stats"); +const COMMON = fileURLToPath(new URL("..", import.meta.url)); + +// ── an fs spy over the whole file ─────────────────────────────────────────── +// Wraps the node:fs and node:fs/promises functions on their CJS exports objects +// and syncs them into the named ESM imports the code under test holds. Records +// every call's path(s), and whether it writes. Case (e) reads the reads; the +// last case reads the writes. +type FsCall = { fn: string; p: string; write: boolean }; +const fsCalls: FsCall[] = []; +{ + const req = createRequire(import.meta.url); + const fsCjs = req("node:fs") as Record<string, unknown>; + const fspCjs = req("node:fs/promises") as Record<string, unknown>; + const READS = ["readFile", "stat", "lstat", "readdir", "readlink"]; + const WRITES = ["writeFile", "appendFile", "rename", "mkdir", "rm", "rmdir", "unlink", "copyFile", "cp", "symlink", "link", "utimes", "truncate", "mkdtemp"]; + const TWO_PATHS = new Set(["rename", "copyFile", "cp", "symlink", "link"]); + const opensForWrite = (flags: unknown) => + (typeof flags === "string" && /[wa+]/.test(flags)) || + (typeof flags === "number" && (flags & 3) !== 0); + const asPath = (v: unknown) => + typeof v === "string" ? v : v instanceof URL ? fileURLToPath(v) : Buffer.isBuffer(v) ? v.toString() : null; + const wrap = (mod: Record<string, unknown>, name: string, write: boolean | "open") => { + const fn = mod[name]; + if (typeof fn !== "function") return; + mod[name] = function (this: unknown, ...args: unknown[]) { + const isWrite = write === "open" ? opensForWrite(args[1]) : write; + const paths = TWO_PATHS.has(name.replace(/Sync$/, "")) ? [args[0], args[1]] : [args[0]]; + for (const a of paths) { + const p = asPath(a); + if (p !== null) fsCalls.push({ fn: name, p: path.resolve(p), write: isWrite }); + } + return (fn as (...a: unknown[]) => unknown).apply(this, args); + }; + }; + for (const n of READS) wrap(fspCjs, n, false); + for (const n of WRITES) { + wrap(fspCjs, n, true); + wrap(fsCjs, n, true); + wrap(fsCjs, `${n}Sync`, true); + } + wrap(fspCjs, "open", "open"); + wrap(fsCjs, "open", "open"); + wrap(fsCjs, "openSync", "open"); + wrap(fsCjs, "createWriteStream", true); + syncBuiltinESMExports(); +} const at = (iso: string) => new Date(iso); const writeJson = (file: string, value: unknown) => { mkdirSync(path.dirname(file), { recursive: true }); writeFileSync(file, JSON.stringify(value, null, 2)); }; -const videoDir = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id); +const videoDir = (id: string, channel = CHANNEL) => + path.join(paths.channelsDir, channel, "data", id); const touch = (file: string, iso: string) => utimesSync(file, at(iso), at(iso)); // A fresh corpus (and a fresh LMDB) per test: every count below is exact. function resetCorpus(): void { - rmSync(paths.transcriptsDir, { recursive: true, force: true }); - rmSync(path.join(ROOT, ".export-index"), { recursive: true, force: true }); - rmSync(POOL, { recursive: true, force: true }); + for (const p of [paths.transcriptsDir, PINNED.EXPORT_INDEX_DIR, POOL, path.join(ROOT, "media")]) { + rmSync(p, { recursive: true, force: true }); + } mkdirSync(paths.transcriptsDir, { recursive: true }); writeFileSync(paths.settingsFile, JSON.stringify({})); writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), { @@ -78,9 +154,15 @@ function resetCorpus(): void { }); } -// Metadata only — what a download leaves before any transcript exists. -function seedVideo(id: string, metaIso = "2026-07-11T11:00:00Z"): void { - const file = path.join(videoDir(id), "metadata.info.json"); +// Metadata only — what a download leaves before any transcript exists. With +// `metaIso` null the file keeps its real mtime (now): "downloaded just now". +function seedVideo( + id: string, + metaIso: string | null = "2026-07-11T11:00:00Z", + extra: Record<string, unknown> = {}, + channel = CHANNEL, +): void { + const file = path.join(videoDir(id, channel), "metadata.info.json"); writeJson(file, { id, title: `Video ${id}`, @@ -90,15 +172,16 @@ function seedVideo(id: string, metaIso = "2026-07-11T11:00:00Z"): void { description: "fixture", webpage_url: `https://www.youtube.com/watch?v=${id}`, extractor_key: "Youtube", + ...extra, }); - touch(file, metaIso); + if (metaIso) touch(file, metaIso); } // YouTube's own captions, as a youtube-handled download writes them. parseVtt // keeps only lines carrying inline timing tags (YouTube's rolling-caption // shape), so the fixture has them. -function addCaptions(id: string, iso: string): void { - const file = path.join(videoDir(id), "transcript.en.vtt"); +function addCaptions(id: string, iso: string, channel = CHANNEL): void { + const file = path.join(videoDir(id, channel), "transcript.en.vtt"); writeFileSync( file, "WEBVTT\nKind: captions\nLanguage: en\n\n" + @@ -165,7 +248,7 @@ test("(a) a transcript that arrives after the stat was cached reaches it on the addWhisper("late", "2026-07-17T05:11:14Z"); await runIndex(); const second = await runStats(); - assert.equal(second.res.changed, 1, "the index's transcript record moved, so the stat is redone"); + assert.equal(second.res.changed, 1, "the index's record moved, so the stat is redone"); const s = statOf(second.byId, "late"); assert.equal(s.hasTranscript, true); assert.equal(s.cueCount, 2); @@ -191,13 +274,15 @@ test("(b) stats built before the index had the video heal after the index build" assert.equal(s.hasTranscript, true); assert.equal(s.transcribedDate, "20260711"); - // And the lag is said, not silent: counted in the result and logged. - assert.equal(first.res.unindexed, 1); + // And the lag is said, not silent: counted in the result and logged. (No + // index build has completed yet, so it is "not indexed yet".) + assert.equal(first.res.notIndexedYet, 1); + assert.equal(first.res.notIndexable, 0); assert.ok( - first.log.some((l) => l.startsWith("1 video(s) are not in the index yet")), + first.log.some((l) => l.startsWith("1 video(s) were downloaded after the last index build")), first.log.join("\n"), ); - assert.equal(second.res.unindexed, 0); + assert.equal(second.res.notIndexedYet, 0); }); test("(c) a caption-only video is dated by its captions' arrival, not by a later Normalize", async () => { @@ -257,33 +342,6 @@ test("(d) a Whisper video resolves exactly as before: the outcome sidecar, else }); }); -// Every fs/promises call inside a video dir, by function name. -function spyFs(names: ("readFile" | "stat" | "readdir" | "open")[]) { - const calls: { fn: string; p: string }[] = []; - // The CJS exports object: patching it and syncing is what reaches the named - // ESM imports buildStats and its helpers hold. - const mod = createRequire(import.meta.url)("node:fs/promises") as Record< - string, - (...a: unknown[]) => unknown - >; - const orig = new Map(names.map((n) => [n, mod[n]])); - for (const n of names) { - const fn = orig.get(n)!; - mod[n] = (p: unknown, ...rest: unknown[]) => { - calls.push({ fn: n, p: String(p) }); - return fn(p, ...rest); - }; - } - syncBuiltinESMExports(); - return { - calls, - restore() { - for (const [n, fn] of orig) mod[n] = fn; - syncBuiltinESMExports(); - }, - }; -} - test("(e) the key does not churn: a heal redoes one stat, then an unchanged run reads nothing per video", async () => { resetCorpus(); seedVideo("w"); @@ -301,18 +359,16 @@ test("(e) the key does not churn: a heal redoes one stat, then an unchanged run assert.equal(heal.res.added, 0); assert.equal(heal.res.changed, 1, "only the video whose transcript arrived"); + // An index build over an unchanged corpus rewrites nothing the key reads. + await runIndex(); const dataDir = path.join(paths.channelsDir, CHANNEL, "data"); - const spy = spyFs(["readFile", "stat", "readdir", "open"]); - let steady; - try { - steady = await runStats(); - } finally { - spy.restore(); - } + const from = fsCalls.length; + const steady = await runStats(); assert.equal(steady.res.added, 0); assert.equal(steady.res.changed, 0); - assert.equal(steady.res.unindexed, 0); - const perVideo = spy.calls.filter((c) => c.p.startsWith(dataDir + path.sep)); + assert.equal(steady.res.notIndexedYet + steady.res.notIndexable, 0); + // The spy sees node:fs and node:fs/promises, async and sync. + const perVideo = fsCalls.slice(from).filter((c) => c.p.startsWith(dataDir + path.sep)); assert.deepEqual( perVideo.map((c) => `${c.fn} ${path.relative(dataDir, c.p)}`).sort(), [ @@ -323,3 +379,200 @@ test("(e) the key does not churn: a heal redoes one stat, then an unchanged run "the unchanged path stats each metadata file and touches nothing else in a video dir", ); }); + +test("(f) cues are read under the index's own key, even when the metadata's upload date moved", async () => { + resetCorpus(); + seedVideo("moved", "2026-07-11T11:00:00Z"); + addCaptions("moved", "2026-07-11T12:00:00Z"); + const norm = await normalizeTranscript({ videoDir: videoDir("moved"), channelSlug: CHANNEL }); + assert.equal(norm.status, "wrote"); + touch(path.join(videoDir("moved"), "transcript.cues.json"), "2026-08-10T13:44:00Z"); + // The metadata is rewritten with another upload date (a stream's date + // settled) but stays older than the normalized cues, so buildIndex still + // trusts transcript.cues.json — and keys the cues by ITS upload date. + seedVideo("moved", "2026-07-12T11:00:00Z", { upload_date: "20260602" }); + await runIndex(); + const { byId } = await runStats(); + const s = statOf(byId, "moved"); + assert.equal(s.uploadDate, "20260602"); + assert.equal(s.hasTranscript, true, "the cues are found under the key the index used"); + assert.equal(s.cueCount, 2); +}); + +test("(g) a re-index for another reason that changes the cues redoes the stat", async () => { + resetCorpus(); + seedVideo("drift"); + addCaptions("drift", "2026-07-11T12:00:00Z"); + await runIndex(); + const first = await runStats(); + assert.equal(statOf(first.byId, "drift").cueCount, 2); + + // A Normalize run writes transcript.cues.json (here with one cue fewer than + // the raw parse). buildIndex does not re-index for that alone... + await normalizeTranscript({ videoDir: videoDir("drift"), channelSlug: CHANNEL }); + const cuesPath = path.join(videoDir("drift"), "transcript.cues.json"); + const doc = JSON.parse(readFileSync(cuesPath, "utf8")) as { cues: unknown[] }; + doc.cues = doc.cues.slice(0, 1); + writeFileSync(cuesPath, JSON.stringify(doc)); + // ...but an availability recheck makes it re-process the video, and then it + // reads the fresher cues.json. The transcript mtime did not move. + writeJson(path.join(videoDir("drift"), "availability.json"), { + checkedAt: "2026-08-20T00:00:00.000Z", + availability: "public", + }); + await runIndex(); + const second = await runStats(); + assert.equal(second.res.changed, 1); + assert.equal(statOf(second.byId, "drift").cueCount, 1); +}); + +test("(h) a video the index skipped is not announced as pending on every run", async () => { + resetCorpus(); + seedVideo("fine"); + // No upload_date: buildIndex skips it (it needs one for the index key). + seedVideo("undated", "2026-07-11T11:00:00Z", { upload_date: undefined }); + await runIndex(); + // Downloaded after that index build. + seedVideo("fresh", null); + for (let run = 0; run < 2; run++) { + const { res, log } = await runStats(); + assert.equal(res.notIndexedYet, 1, `run ${run}`); + assert.equal(res.notIndexable, 1, `run ${run}`); + assert.ok(log.some((l) => l.startsWith("1 video(s) were downloaded after the last index build")), log.join("\n")); + assert.ok(log.some((l) => l.startsWith("1 video(s) are not in the index although its last build saw them")), log.join("\n")); + } + await runIndex(); + const { res } = await runStats(); + assert.equal(res.notIndexedYet, 0, "the next index build takes the fresh one"); + assert.equal(res.notIndexable, 1, "the undated one stays, and is said as such"); +}); + +// A second channel whose data/ is a relocated symlink, the way the editor's +// Storage panel leaves it: channels/<slug>/data -> <root>/<slug>/data, with +// config.dataDir recording the target. +function seedDriveChannel(): { target: string } { + const target = path.join(ROOT, "media", DRIVE_CHANNEL, "data"); + mkdirSync(target, { recursive: true }); + writeJson(path.join(paths.channelsDir, DRIVE_CHANNEL, "config.json"), { + handling: "youtube", + name: "Drive Channel", + url: "https://www.youtube.com/@drive/videos", + dataDir: target, + }); + symlinkSync(target, path.join(paths.channelsDir, DRIVE_CHANNEL, "data")); + for (const id of ["d1", "d2"]) { + seedVideo(id, "2026-07-11T11:00:00Z", {}, DRIVE_CHANNEL); + addCaptions(id, "2026-07-11T12:00:00Z", DRIVE_CHANNEL); + } + return { target }; +} + +test("(i) an unmounted media drive keeps its channel's stats; a cache clear refuses", async () => { + resetCorpus(); + seedVideo("local"); + seedDriveChannel(); + await runIndex(); + const mounted = await runStats(); + assert.equal(statOf(mounted.byId, "d1").hasTranscript, true); + + // Unmount: the link now dangles, exactly as an absent USB drive leaves it. + const media = path.join(ROOT, "media"); + renameSync(media, `${media}-away`); + const log: string[] = []; + const away = await runStats(log); + assert.deepEqual(away.res.heldChannels, [DRIVE_CHANNEL]); + assert.equal(away.res.removed, 0, "not read as a channel with no videos"); + for (const id of ["d1", "d2", "local"]) assert.ok(away.byId.has(id), `${id} still published`); + assert.equal(statOf(away.byId, "d1").hasTranscript, true); + assert.ok( + log.some((l) => l.startsWith(`Channel ${DRIVE_CHANNEL}: media not reachable`) && l.includes("its 2 cached stat(s) are kept")), + log.join("\n"), + ); + + // A schema change needs the whole cache rebuilt, which cannot include a + // channel it cannot read: refuse, and leave the cache as it is. + setStoredSchema(STATS_SCHEMA_VERSION - 1); + await assert.rejects(runStats(), /must be rebuilt .* cannot be read: drive-channel/); + assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION - 1); + assert.equal(countStats(), 3); + + renameSync(`${media}-away`, media); + const back = await runStats(); + assert.deepEqual(back.res.heldChannels, []); + assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION); + assert.equal(back.byId.size, 3); +}); + +// Direct access to the temp LMDB's stats cache, for the schema cases. +function withDb<T>(fn: (dbs: { meta: ReturnType<ReturnType<typeof open>["openDB"]>; stats: ReturnType<ReturnType<typeof open>["openDB"]> }) => T): T { + const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true }); + try { + return fn({ + meta: root.openDB({ name: "statsMeta", encoding: "msgpack" }), + stats: root.openDB({ name: "statsByPath", encoding: "msgpack" }), + }); + } finally { + root.close(); + } +} +const setStoredSchema = (v: number) => + withDb(({ meta }) => meta.putSync("schema", v)); +const readStoredSchema = () => withDb(({ meta }) => meta.get("schema")); +const countStats = () => withDb(({ stats }) => [...stats.getKeys()].length); + +test("(j) the schema guard: an older cache is cleared, a newer one is refused unless overridden", async () => { + resetCorpus(); + seedVideo("v1"); + addCaptions("v1", "2026-07-11T12:00:00Z"); + await runIndex(); + await runStats(); + assert.equal(countStats(), 1); + + // Older: cleared and rebuilt, as every schema bump has always done. + setStoredSchema(STATS_SCHEMA_VERSION - 1); + const log: string[] = []; + const older = await runStats(log); + assert.ok(log.some((l) => l.includes("clearing stats cache")), log.join("\n")); + assert.equal(older.res.added, 1); + assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION); + + // Newer: refused, naming both versions and the override; nothing touched. + setStoredSchema(STATS_SCHEMA_VERSION + 1); + await assert.rejects( + runStats(), + new RegExp( + `newer build \\(stats schema ${STATS_SCHEMA_VERSION + 1}; this build's is ${STATS_SCHEMA_VERSION}\\).*${STATS_DOWNGRADE_ENV}=1`, + ), + ); + assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION + 1); + assert.equal(countStats(), 1); + + // The CLI exits non-zero on it. + const cli = spawnSync( + path.join(COMMON, "node_modules", ".bin", "tsx"), + ["bin/archilyzer.ts", "build", "stats"], + { cwd: COMMON, env: { ...process.env }, encoding: "utf8" }, + ); + assert.notEqual(cli.status, 0, cli.stdout + cli.stderr); + assert.match(cli.stderr, /written by a newer build/); + assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION + 1); + assert.equal(countStats(), 1); + + // A deliberate rollback, overridden: cleared and rebuilt at this version. + process.env[STATS_DOWNGRADE_ENV] = "1"; + try { + const rolled = await runStats(); + assert.equal(rolled.res.added, 1); + assert.equal(readStoredSchema(), STATS_SCHEMA_VERSION); + } finally { + delete process.env[STATS_DOWNGRADE_ENV]; + } +}); + +test("(z) no write this file caused landed outside its temp root", () => { + const outside = fsCalls.filter( + (c) => c.write && c.p !== ROOT && !c.p.startsWith(ROOT + path.sep), + ); + assert.deepEqual(outside, []); + assert.ok(fsCalls.some((c) => c.write), "the spy saw the writes"); +}); diff --git a/common/controller/buildStats.ts b/common/controller/buildStats.ts @@ -13,12 +13,23 @@ // stands, which the cache key below makes safe. // // THE CACHE KEY IS TWO THINGS: the metadata file's mtime AND the index's own -// per-video transcript record (buildIndex's `mtimes.transcriptMs`, or +// per-video record (buildIndex's `mtimes` entry, as indexSignature, or // NOT_INDEXED). Until schema 6 it was the metadata mtime alone, while // hasTranscript / cueCount / coverage / transcribedDate come from the index and // the transcript files — so a transcript that arrived after a video was first // seen (Whisper days later, or a stats run before build:index had the video) // never reached its stat, and a whole channel could publish as untranscribed. +// +// A CHANNEL WHOSE MEDIA IS NOT REACHABLE (a relocated `data/` on an unmounted +// drive, or one mid-relocation) is not rescanned: its cached stats are kept as +// they are, rather than read as a channel with no videos and removed. This is +// the stats build's own guard — the job registry's `needsMedia` check is per +// channel, and this build is pool-wide — so the editor job and the CLI share it. +// +// ONE STATS BUILD AT A TIME. Nothing here takes a lock: two runs at once are +// harmless unless one of them clears the cache (a schema change) after the +// other scanned, when the other can publish truncated pages. The editor's build +// jobs share the "build" queue by default; the CLI is outside every queue. import path from "node:path"; import { createHash } from "node:crypto"; @@ -52,9 +63,11 @@ import { import type { VideoStatus } from "../lib/stats"; import type { ChannelConfig } from "../lib/channelConfig"; import { readChannelConfigFile } from "./channels"; +import { inspectChannelMedia } from "../lib/channelMedia"; import type { Paths } from "../lib/paths"; import { listSites, siteStatsDir } from "../lib/site"; import { + INDEX_SCANNED_AT_KEY, STATS_SCHEMA_VERSION, STATS_MANIFEST_VERSION, STATS_MAX_PAGE_BYTES, @@ -67,15 +80,55 @@ import { type IndexKey = [string, string, string]; type PathKey = [string, string]; -// `idxMs` is the index's per-video transcript record as this stat saw it — see -// indexTranscriptMs. Optional only because a record written before schema 6 -// has none; the schema bump clears those, and a missing value compares -// unequal to every real one, so such a record would be recomputed anyway. -type StatsRecord = { metaMs: number; idxMs?: number | null; stat: VideoStat }; +// `idx` is the index's per-video record as this stat saw it (indexSignature). +// Optional only because a record written before schema 6 has none; the schema +// bump clears those, and a missing value compares unequal to every real one, so +// such a record would be recomputed anyway. +type StatsRecord = { metaMs: number; idx?: string; stat: VideoStat }; + +// The part of buildIndex's `mtimes` record this cache reads: every input whose +// change makes buildIndex re-process the video (and so rewrite its cues), and +// the key it stored the cues under. +type IndexRecord = { + metaMs: number; + transcriptMs: number | null; + subsMs?: number | null; + availabilityMs?: number | null; + digestMs?: number | null; + indexKey: IndexKey; +}; + +// buildIndex has no `mtimes` record for this video. See the two counts below. +const NOT_INDEXED = "-"; + +// The whole index record as one comparable string. Keying on all of it (not +// the transcript mtime alone) means ANY re-index redoes the stat: a re-index +// for another reason can read a fresher transcript.cues.json and change the cue +// count, and a moved index key moves the cues. Numbers print as their shortest +// round-trip form, so an unchanged record gives an identical string. +function indexSignature(r: IndexRecord | undefined): string { + if (!r) return NOT_INDEXED; + const n = (v: number | null | undefined) => (v == null ? "" : String(v)); + return [ + n(r.metaMs), + n(r.transcriptMs), + n(r.subsMs), + n(r.availabilityMs), + n(r.digestMs), + ...r.indexKey, + ].join("\u0000"); +} -// buildIndex has no `mtimes` record for this video yet: it was downloaded after -// the last index build. Distinct from `null` (indexed, no transcript file). -const NOT_INDEXED = -1; +// Set to 1 (or true/yes/on) to let an OLDER build clear a stats cache a newer +// one wrote — a deliberate rollback. Declared in lib/envVars.ts. +export const STATS_DOWNGRADE_ENV = "ARCHILYZER_STATS_ALLOW_DOWNGRADE"; +const TRUTHY = new Set(["1", "true", "yes", "on"]); +function allowsStatsDowngrade( + env: Record<string, string | undefined> = process.env, +): boolean { + const raw = env.ARCHILYZER_STATS_ALLOW_DOWNGRADE; + return typeof raw === "string" && TRUTHY.has(raw.trim().toLowerCase()); +} type ScanEntry = { channelSlug: string; @@ -92,10 +145,15 @@ export type BuildStatsResult = { added: number; changed: number; removed: number; - // Videos on disk that the index does not have yet. Their stats say "no - // transcript" until the first run after the next index build, which - // recomputes them (the key moves from NOT_INDEXED). - unindexed: number; + // Videos on disk with no index record, in two kinds. `notIndexedYet`: their + // metadata is newer than the last completed index build's scan — downloaded + // since — and the first stats run after the next index build redoes them. + // `notIndexable`: the last index build saw them and did not index them (no + // upload_date, or it failed on them); nothing here will change that. + notIndexedYet: number; + notIndexable: number; + // Channels whose media was not reachable, so their cached stats were kept. + heldChannels: string[]; pagesWritten: number; shortCircuited: boolean; durationMs: number; @@ -184,17 +242,24 @@ async function resolveAcquisitionDates( return { downloadedDate, transcribedDate }; } +// `held` maps each channel whose media is not reachable to the reason: it is +// not scanned, and the caller keeps its cached stats (see the file header). async function scanSource( channelsDir: string, log: (msg: string) => void, -): Promise<{ entries: ScanEntry[]; channels: Map<string, ChannelConfig> }> { +): Promise<{ + entries: ScanEntry[]; + channels: Map<string, ChannelConfig>; + held: Map<string, string>; +}> { const channels = new Map<string, ChannelConfig>(); const entries: ScanEntry[] = []; + const held = new Map<string, string>(); let channelEntries: Dirent[]; try { channelEntries = await readdir(channelsDir, { withFileTypes: true }); } catch { - return { entries, channels }; + return { entries, channels, held }; } for (const ch of channelEntries) { if (!ch.isDirectory()) continue; @@ -206,6 +271,13 @@ async function scanSource( } if (cfg.excludeFromBuild) continue; channels.set(ch.name, cfg); + // An unmounted drive is not an empty channel (lib/channelMedia.ts): the + // readdir below would fail and every one of its stats would be removed. + const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg); + if (media.status !== "ok" && media.status !== "in-place") { + held.set(ch.name, media.detail ?? media.status); + continue; + } const dataDir = path.join(channelDir, "data"); let videoEntries: Dirent[]; try { @@ -231,7 +303,7 @@ async function scanSource( }); } } - return { entries, channels }; + return { entries, channels, held }; } // Buffered byte-capped page writer. Stats records are small, so a page fits in @@ -310,25 +382,51 @@ export async function buildStats({ encoding: "msgpack", }); const meta = root.openDB<unknown, string>({ name: "statsMeta", encoding: "msgpack" }); - // Read-only view of buildIndex's per-video mtime record, keyed like - // statsByPath. Only `transcriptMs` is read: the mtime of the transcript file - // the index took this video's cues from (null when it had none). buildIndex - // rewrites a video's `cues` when that number moves (buildIndex.ts, the - // added/changed diff), so it is exactly the "has the transcript this stat was - // computed from changed" signal — at the cost of one LMDB get per video, and - // no file I/O, on the unchanged path. - const indexMtimes = root.openDB<{ transcriptMs: number | null }, PathKey>({ + // Read-only views of buildIndex's per-video `mtimes` record (keyed like + // statsByPath: the second half of this cache's key, and the key the video's + // cues were stored under) and of its `meta` (when its last scan began). One + // LMDB get per video, and no file I/O, on the unchanged path. + const indexMtimes = root.openDB<IndexRecord, PathKey>({ name: "mtimes", encoding: "msgpack", }); - const indexTranscriptMs = (k: PathKey): number | null => { - const rec = indexMtimes.get(k); - return rec ? (rec.transcriptMs ?? null) : NOT_INDEXED; - }; + const indexMeta = root.openDB<unknown, string>({ name: "meta", encoding: "msgpack" }); + // NEVER CLEAR A CACHE A NEWER BUILD WROTE. An old build running beside a + // new one (an editor not yet restarted onto the new code) would otherwise + // clear it, refill it the old way, and the next new run would clear it back: + // a full pass each time, and old-logic numbers published in between. const storedSchema = meta.get("schema") as number | undefined; + if ( + typeof storedSchema === "number" && + storedSchema > STATS_SCHEMA_VERSION && + !allowsStatsDowngrade() + ) { + await root.close(); + throw new Error( + `The stats cache was written by a newer build (stats schema ${storedSchema}; this build's is ${STATS_SCHEMA_VERSION}). ` + + `Refusing to clear it: rebuild and restart onto the current code. ` + + `For a deliberate rollback, set ${STATS_DOWNGRADE_ENV}=1.`, + ); + } + + const { entries, channels, held } = await scanSource(paths.channelsDir, log); + log(`Scanned ${entries.length} videos across ${channels.size} channels.`); + const schemaBumped = storedSchema !== STATS_SCHEMA_VERSION; if (schemaBumped) { + // A clear with a channel's media unreachable would drop that channel's + // stats for good (it cannot be rescanned), and the pages built from this + // run would publish it as empty. Refuse instead. + if (held.size > 0) { + await root.close(); + throw new Error( + `The stats cache must be rebuilt (stats schema ${storedSchema ?? "<none>"} -> ${STATS_SCHEMA_VERSION}), ` + + `but ${held.size} channel(s) cannot be read: ` + + [...held].map(([slug, why]) => `${slug} (${why})`).join("; ") + + `. Mount their media (see /storage) and run it again.`, + ); + } log( `Stats schema change (${storedSchema ?? "<none>"} -> ${STATS_SCHEMA_VERSION}); clearing stats cache.`, ); @@ -336,46 +434,65 @@ export async function buildStats({ await meta.put("schema", STATS_SCHEMA_VERSION); } - const { entries, channels } = await scanSource(paths.channelsDir, log); - log(`Scanned ${entries.length} videos across ${channels.size} channels.`); - const liveIds = new Set( entries.map((e) => pathKeyId([e.channelSlug, e.videoDir])), ); - const toProcess: { e: ScanEntry; idxMs: number | null }[] = []; + const scannedAt = indexMeta.get(INDEX_SCANNED_AT_KEY) as number | undefined; + const toProcess: { e: ScanEntry; idx: string; indexKey?: IndexKey }[] = []; let added = 0; let changed = 0; - let unindexed = 0; + let notIndexedYet = 0; + let notIndexable = 0; for (const e of entries) { const pk: PathKey = [e.channelSlug, e.videoDir]; - const idxMs = indexTranscriptMs(pk); - if (idxMs === NOT_INDEXED) unindexed++; + const rec = indexMtimes.get(pk); + const idx = indexSignature(rec); + if (idx === NOT_INDEXED) { + if (typeof scannedAt !== "number" || e.metaMs > scannedAt) notIndexedYet++; + else notIndexable++; + } const prev = statsByPath.get(pk); if (!prev) { added++; - toProcess.push({ e, idxMs }); - } else if (prev.metaMs !== e.metaMs || prev.idxMs !== idxMs) { + toProcess.push({ e, idx, indexKey: rec?.indexKey }); + } else if (prev.metaMs !== e.metaMs || prev.idx !== idx) { // The second half is the fix for stats frozen at first sight: a - // transcript that arrives later moves transcriptMs (and a video first - // seen before the index had it moves off NOT_INDEXED), while the - // metadata file — the old key's only input — is never touched. + // transcript that arrives later makes buildIndex re-process the video + // (and a video first seen before the index had it moves off + // NOT_INDEXED), while the metadata file — the old key's only input — is + // never touched. changed++; - toProcess.push({ e, idxMs }); + toProcess.push({ e, idx, indexKey: rec?.indexKey }); } } - if (unindexed > 0) { - // Not a failure: the pool composers read the index as it stands, and the - // editor downloads between index builds. Said so a published number that - // lags the disk has its reason in the log. + // Neither is a failure of this build, and both are said, so a published + // number that lags the disk has its reason in the log. Only the first kind + // resolves itself. + if (notIndexedYet > 0) { + log( + `${notIndexedYet} video(s) were downloaded after the last index build; their transcripts reach the stats on the first run after the next one.`, + ); + } + if (notIndexable > 0) { log( - `${unindexed} video(s) are not in the index yet; their transcripts reach the stats on the first run after the next index build.`, + `${notIndexable} video(s) are not in the index although its last build saw them (no upload_date, or it failed on them: see that build's log). Their stats show no transcript until that is fixed.`, ); } const removedKeys: PathKey[] = []; + const keptHeld = new Map<string, number>(); for (const { key } of statsByPath.getRange()) { const k = key as PathKey; + if (held.has(k[0])) { + keptHeld.set(k[0], (keptHeld.get(k[0]) ?? 0) + 1); + continue; + } if (!liveIds.has(pathKeyId(k))) removedKeys.push(k); } + for (const [slug, why] of held) { + log( + `Channel ${slug}: media not reachable (${why}); its ${keptHeld.get(slug) ?? 0} cached stat(s) are kept as they are, not rescanned.`, + ); + } const removed = removedKeys.length; const BATCH = 200; @@ -383,7 +500,7 @@ export async function buildStats({ signal?.throwIfAborted(); const slice = toProcess.slice(i, i + BATCH); await Promise.all( - slice.map(async ({ e, idxMs }) => { + slice.map(async ({ e, idx, indexKey: recordedKey }) => { try { const metaRaw = await readFile(e.metaPath, "utf8"); const parsedMeta = JSON.parse(metaRaw) as RawMetadata; @@ -394,7 +511,12 @@ export async function buildStats({ e.configName, ); if (!base.uploadDate) return; - const indexKey: IndexKey = [base.uploadDate, e.channelSlug, base.id]; + // The key buildIndex stored this video's cues under, when it has a + // record: its uploadDate comes from transcript.cues.json when that + // is fresh, which the metadata's can differ from. The computed key + // is only a fallback for a video the index does not have. + const indexKey: IndexKey = + recordedKey ?? [base.uploadDate, e.channelSlug, base.id]; const cueList = cues.get(indexKey); const cueCount = cueList ? cueList.length : null; const coverage = transcriptCoverage(cueList, base.duration).coverage; @@ -429,7 +551,7 @@ export async function buildStats({ ); await statsByPath.put([e.channelSlug, e.videoDir], { metaMs: e.metaMs, - idxMs, + idx, stat, }); } catch (err) { @@ -632,7 +754,9 @@ export async function buildStats({ added, changed, removed, - unindexed, + notIndexedYet, + notIndexable, + heldChannels: [...held.keys()], pagesWritten: aggregatePages, shortCircuited: needBuild.length === 0, durationMs: Date.now() - t0, diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts @@ -111,6 +111,7 @@ const DECLARED: EnvVarDecl[] = [ { name: "TRANSCRIPT_PLATFORM_LINKS", audience: "runtime", default: "off", readBy: "common/lib/archive/reader-fs.ts", doc: "`1` cites platform watch pages instead of the archive's own pages." }, { name: "AUDIO_CHECK_RESUME_DURING_PROBE", audience: "runtime", default: "the channel's `audioCheck.resumeDuringProbe`", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "`1` or `true` resumes yt-dlp during the audio check's probe, anything else holds it, for a one-off comparison run; unset = the channel's setting." }, { name: "AUDIO_CHECK_BACKOFF_FACTOR", audience: "runtime", default: "the built-in factor", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "The audio check's interval backoff factor, in (0, 1], for a one-off run." }, + { name: "ARCHILYZER_STATS_ALLOW_DOWNGRADE", audience: "runtime", default: "off", readBy: "common/controller/buildStats.ts", doc: "`1` lets a stats build clear a stats cache that a NEWER build wrote, for a deliberate rollback. Unset, such a build refuses and names both versions." }, { name: "MCP_IO_STATS", audience: "runtime", default: "off", readBy: "common/lib/archive/io-stats.ts", doc: "`1` turns on per-call I/O accounting, for `mcp/bench`." }, { name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool", doc: "Which editor `pnpm ops` and the MCP's `fetch_clip` talk to." }, { name: "ARCHILYZER_AGENT", audience: "runtime", default: "`cli`", readBy: "scripts/archilyzer-ops.mjs", doc: "Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`." },