Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 1e6bc8c32deb4d28c281aa23856261277ff29938
parent 653bd65287646770448172696fff68aaac176c9c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat,  8 Aug 2026 00:16:20 -0400

Make the corpus census a report, not a permanently-red test

digestPlan.test.ts asserted that this machine's LMDB held exactly 191,116 chunks
across 73,367 videos. It opened the operator's real corpus and skipped when
absent, so it ran on exactly one machine — and on that machine it was RED,
because the number moves whenever a video is downloaded. Measured today it is
194,053 across 74,321. That is the corpus growing, not a regression, and a suite
carrying a permanent red for doing its job correctly teaches people to ignore
reds.

An absolute count of a live corpus is a MEASUREMENT. It now lives in
`bin/digest-plan.ts --census`, which prints the band table by duration, the
totals, the projected sweep days, and the drift against
CORPUS_CHUNKS_PER_AUDIO_HOUR — with a loud warning when the constant has stopped
describing the corpus.

The test's own comment gave it two jobs, and both are kept, hermetically:

  1. "catch the plan and the chunker drifting apart" — a pure property, and it
     never needed real data. buildChunkCensus is now a pure function over an
     iterable of stats, asserted against the real chunker at the chunk-size
     boundaries.
  2. "catch a DEFAULT_DIGEST_* change that alters the shipped chunk size without
     re-pricing the sweep" — this is the one that mattered, and it was pinned on
     the wrong thing. The cost model consumes the DENSITY
     (CORPUS_CHUNKS_PER_AUDIO_HOUR), not a video count, and the old test's
     `maxCuesForContext(8192)` hardcoded the default it was supposed to be
     guarding — so changing DEFAULT_DIGEST_NUM_CTX would have sailed straight
     through it. It is now pinned against the DEFAULTS themselves, with a failure
     message that says to re-run the census and update the constant.

So the tripwire that mattered survives, is stronger than it was, and the common
suite reads none of the operator's data.

Verified: common 563/563 — the suite is fully green for the first time. tsc clean
in common/editor/export. `--census` run against the real corpus reproduces
194,053 / 74,321 with 0 rows missing a cueCount, density 2.460 against the
constant's 2.47 (drift 0.010, under the 0.05 warning threshold) — so the constant
has held to within 0.01 across ~3,000 new videos, which is the property actually
worth having. `--census --json` and the no-corpus path (what CI or a fresh
checkout hits) both exercised.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

Diffstat:
Mcommon/bin/digest-plan.ts | 97+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/digestPlan.test.ts | 187++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------------------
Mcommon/controller/digestPlan.ts | 134+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
3 files changed, 368 insertions(+), 50 deletions(-)

diff --git a/common/bin/digest-plan.ts b/common/bin/digest-plan.ts @@ -24,18 +24,28 @@ // --channels a,b restrict to these channel slugs // --top N how many channels to list (default 15, 0 = all) // --json machine-readable, for diffing two runs +// --census print the CORPUS CHUNK CENSUS by duration band and stop. +// This is the measurement the density constant +// CORPUS_CHUNKS_PER_AUDIO_HOUR is derived from, and the +// table frozen in digestPlan.ts's comment is one run of +// it. Re-run it after changing any DEFAULT_DIGEST_*, and +// after any large ingest, to re-price the sweep. import { getPaths } from "../lib/paths"; import { buildDigestSweepPlan, audioHours, chunksPerAudioHour, + readChunkCensus, sweepDays, sweepDaysFromAudio, + CORPUS_CHUNKS_PER_AUDIO_HOUR, DIGEST_PLAN_ROLES, MEASURED_SECONDS_PER_AUDIO_HOUR, MEASURED_SECONDS_PER_CHUNK, roleMustGenerate, } from "../controller/digestPlan"; +import { maxCuesForContext } from "../lib/digestPrompt"; +import { DEFAULT_DIGEST_NUM_CTX } from "../lib/digest"; import { parseFlags } from "./_parseFlags"; const flags = parseFlags(process.argv.slice(2)); @@ -73,7 +83,94 @@ function pct(part: number, whole: number): string { return whole > 0 ? `${((part / whole) * 100).toFixed(1)}%` : "—"; } +// The census: the corpus totals the cost model is derived from, printed rather +// than asserted. +// +// It used to be a unit test that opened this machine's LMDB and asserted the +// corpus was exactly 191,116 chunks across 73,367 videos. That could only pass +// on one machine, and it went red there the moment a video was downloaded — the +// corpus growing is not a regression. An absolute count of a live corpus is a +// MEASUREMENT, and this is where a measurement goes. The properties that test +// was really guarding (plan-vs-chunker drift, and a DEFAULT_DIGEST_* change +// silently re-pricing the sweep) are pinned hermetically in +// controller/digestPlan.test.ts instead. +async function census(): Promise<void> { + const maxCues = maxCuesForContext(DEFAULT_DIGEST_NUM_CTX); + const result = await readChunkCensus(getPaths(), maxCues); + if (!result) { + console.log(""); + console.log("No stats cache on this machine — run build:stats first."); + console.log(""); + return; + } + if (asJson) { + console.log(JSON.stringify({ ...result, maxCues }, null, 2)); + return; + } + + console.log(""); + console.log( + `Corpus chunk census — ${maxCues} cues/chunk at the shipped defaults.`, + ); + if (result.statsSchemaStale) + console.log("!! stats cache is stale — run build:stats."); + console.log(""); + console.log( + ` ${"band".padEnd(11)} ${"videos".padStart(7)} ${"audio-h".padStart(8)} ` + + `${"% audio".padStart(8)} ${"chunks".padStart(9)} ${"chunks/audio-h".padStart(15)} ${"% chunks".padStart(9)}`, + ); + for (const band of result.bands) { + console.log( + ` ${band.label.padEnd(11)} ${count(band.videos).padStart(7)} ${hours(band.audioSeconds).padStart(8)} ` + + `${pct(band.audioSeconds, result.audioSeconds).padStart(8)} ${count(band.chunks).padStart(9)} ` + + `${chunksPerAudioHour(band.chunks, band.audioSeconds).toFixed(2).padStart(15)} ` + + `${pct(band.chunks, result.chunks).padStart(9)}`, + ); + } + console.log( + ` ${"total".padEnd(11)} ${count(result.videos).padStart(7)} ${hours(result.audioSeconds).padStart(8)} ` + + `${"".padStart(8)} ${count(result.chunks).padStart(9)} ` + + `${result.chunksPerAudioHour.toFixed(2).padStart(15)}`, + ); + console.log(""); + if (result.estimated > 0) { + // The census's claim to costing nothing is that cueCount is already on every + // row. A non-zero here means part of the total above is a guess from + // duration, and it must not pass unremarked. + console.log( + `!! ${count(result.estimated)} row(s) had no cueCount; their chunks were ESTIMATED from duration.`, + ); + } + console.log( + `Projected sweep: ${sweepDays(result.chunks, perChunk).toFixed(1)} days at ${perChunk}s per chunk.`, + ); + + // THE TRIPWIRE, and the reason to run this after touching a DEFAULT_DIGEST_*: + // the density constant is what the cost model consumes, and nothing else + // notices when the shipped chunk size stops matching the corpus it was + // measured on. + const drift = Math.abs(result.chunksPerAudioHour - CORPUS_CHUNKS_PER_AUDIO_HOUR); + console.log( + `Density ${result.chunksPerAudioHour.toFixed(3)} vs CORPUS_CHUNKS_PER_AUDIO_HOUR ` + + `${CORPUS_CHUNKS_PER_AUDIO_HOUR} (drift ${drift.toFixed(3)}).`, + ); + if (drift >= 0.05) { + console.log( + `!! The constant no longer describes this corpus. Update ` + + `CORPUS_CHUNKS_PER_AUDIO_HOUR (and the band table beside it) in ` + + `controller/digestPlan.ts, or the sweep projection is priced on a mix ` + + `that no longer exists.`, + ); + } + console.log(""); +} + async function main(): Promise<void> { + if (flags.census === "true") { + await census(); + return; + } + const plan = await buildDigestSweepPlan({ paths: getPaths(), lane, diff --git a/common/controller/digestPlan.test.ts b/common/controller/digestPlan.test.ts @@ -3,6 +3,7 @@ import assert from "node:assert/strict"; import type { DigestClusterRole } from "./digestSharing"; import { audioHours, + buildChunkCensus, chunksForVideo, chunksPerAudioHour, classifyDigestRole, @@ -16,6 +17,10 @@ import { } from "./digestPlan"; import { chunkCuesForContext, countCueChunks } from "../lib/transcriptWindow"; import { DIGEST_OVERLAP_CUES, maxCuesForContext } from "../lib/digestPrompt"; +import { + DEFAULT_DIGEST_MAX_CUES_PER_CHUNK, + DEFAULT_DIGEST_NUM_CTX, +} from "../lib/digest"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/digestPlan.test.ts @@ -185,62 +190,144 @@ test("audio-hours, not video count, is what a saving is measured in", () => { ); }); -// The census, as a live regression pin against the REAL corpus. +// --------------------------------------------------------------------------- +// The census // -// The pure test above pins the formula; this one pins the number the whole -// re-pricing rests on. It re-derives the corpus chunk total from the same LMDB row -// the plan reads, using the same helper, so it fails if the plan and the chunker -// ever drift — or if a DEFAULT_DIGEST_* changes the shipped chunk size without -// anyone re-pricing the sweep. +// There used to be a test HERE that opened this machine's real LMDB and asserted +// the corpus was exactly 191,116 chunks across 73,367 videos. It was skipped +// everywhere else and permanently RED here, because the number moves whenever a +// video is downloaded — 191,116 had already become 194,053. That is the corpus +// growing, not a regression, and a suite that reports a red for it teaches people +// to ignore reds. // -// Skipped when there is no corpus on this machine (CI, a fresh checkout): a pin on -// data that isn't there would just be a broken test. -test("the corpus chunk census reproduces 191,116 at the shipped config", async (t) => { - const { existsSync } = await import("node:fs"); - const { getPaths } = await import("../lib/paths"); - const paths = getPaths(); - if (!existsSync(paths.lmdbPath)) { - t.skip("no LMDB corpus on this machine"); - return; - } - const { open } = await import("lmdb"); - const { STATS_SCHEMA_VERSION } = await import("../lib/stats"); - const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true }); - const meta = root.openDB<unknown, string>({ - name: "statsMeta", - encoding: "msgpack", - }); - if ((meta.get("schema") as number | undefined) !== STATS_SCHEMA_VERSION) { - t.skip("stats cache is at another schema version — run build:stats"); - return; - } - const statsByPath = root.openDB< - { metaMs: number; stat: import("../lib/stats").VideoStat }, - [string, string] - >({ name: "statsByPath", encoding: "msgpack" }); +// An absolute count of a live corpus is a MEASUREMENT. It now lives in +// `bin/digest-plan.ts --census`, which prints the band table and warns when the +// density constant stops describing the corpus. +// +// Its comment claimed two jobs. Both are kept below, hermetically: +// 1. catch the plan and the chunker drifting apart — a pure property, and one +// the test directly above already covers for countCueChunks; +// 2. catch a DEFAULT_DIGEST_* change that alters the shipped chunk size +// without re-pricing the sweep — pinned on the DENSITY constant, which is +// what the cost model actually consumes, rather than on a video count that +// moves on its own. +// --------------------------------------------------------------------------- + +// One synthetic row per band, with cue counts chosen so the expected chunk count +// is computable by hand. +function stat(duration: number, cueCount: number | null) { + return { duration, cueCount, hasTranscript: true }; +} + +test("the census buckets by duration and totals chunks per band", () => { + const maxCues = maxCuesForContext(8192); + const census = buildChunkCensus( + [ + stat(10 * 60, 600), // < 15 min → 1 chunk + stat(30 * 60, 601), // 15–60 min → 2 + stat(90 * 60, 1_161), // 1–2 h → 3 + stat(3 * 3600, 600), // 2–4 h → 1 + stat(6 * 3600, 600), // 4–8 h → 1 + stat(9 * 3600, 600), // > 8 h → 1 + ], + maxCues, + ); + assert.deepEqual( + census.bands.map((b) => [b.label, b.videos, b.chunks]), + [ + ["< 15 min", 1, 1], + ["15–60 min", 1, 2], + ["1–2 h", 1, 3], + ["2–4 h", 1, 1], + ["4–8 h", 1, 1], + ["> 8 h", 1, 1], + ], + ); + assert.equal(census.videos, 6); + assert.equal(census.chunks, 9); + assert.equal(census.estimated, 0); + // Band EDGES land in the lower band, so a boundary video cannot be counted + // twice or dropped — 15:00 exactly is "15–60 min", not "< 15 min". + const edges = buildChunkCensus( + [stat(15 * 60, 600), stat(8 * 3600, 600)], + maxCues, + ); + assert.equal(edges.bands[0].videos, 0); + assert.equal(edges.bands[1].videos, 1); + assert.equal(edges.bands[4].videos, 0); + assert.equal(edges.bands[5].videos, 1); +}); + +test("the census counts only transcribed videos, and flags estimated rows", () => { + const maxCues = maxCuesForContext(8192); + const census = buildChunkCensus( + [ + stat(3600, 600), + { duration: 3600, cueCount: 600, hasTranscript: false }, + // A zero-duration row cannot be banded and is not work. + stat(0, 600), + // No cueCount: estimated from duration rather than dropped, because + // understating the sweep is the worse failure — and COUNTED, because the + // census's claim to costing nothing is that cueCount is on every row. + stat(4 * 3600, null), + ], + maxCues, + ); + assert.equal(census.videos, 2); + assert.equal(census.estimated, 1); + assert.equal( + census.chunks, + 1 + Math.round(4 * CORPUS_CHUNKS_PER_AUDIO_HOUR), + ); +}); +// The plan and the chunker, one function — asserted through the census this time, +// so the path the report actually takes is the path under test. +test("the census prices a video exactly as the real chunker slices it", () => { const maxCues = maxCuesForContext(8192); - let chunks = 0; - let videos = 0; - let audioSeconds = 0; - let missingCueCount = 0; - for (const { value } of statsByPath.getRange()) { - const stat = value.stat; - if (!stat.hasTranscript || !(stat.duration > 0)) continue; - videos++; - audioSeconds += stat.duration; - if (typeof stat.cueCount !== "number") missingCueCount++; - chunks += chunksForVideo(stat, maxCues).chunks; + const opts = { maxCues, overlapCues: DIGEST_OVERLAP_CUES }; + for (const n of [1, 599, 600, 601, 1_160, 1_161, 5_000]) { + const cues = Array.from({ length: n }, (_, i) => ({ + start: i, + end: i + 1, + text: "x", + })); + assert.equal( + buildChunkCensus([stat(3600, n)], maxCues).chunks, + chunkCuesForContext(cues, opts).length, + `census disagrees with the chunker at ${n} cues`, + ); } +}); - assert.equal(chunks, 191_116, "corpus chunk total moved"); - assert.equal(videos, 73_367, "transcribed video count moved"); - // The census cost nothing precisely because cueCount was already on every row. - assert.equal(missingCueCount, 0, "some rows lost their cueCount"); - // And the derived density constant must still describe this corpus. - const density = chunksPerAudioHour(chunks, audioSeconds); +// THE TRIPWIRE THE LIVE TEST WAS REALLY FOR, and the whole reason it can be +// hermetic: the cost model consumes the DENSITY, not a video count. +// +// CORPUS_CHUNKS_PER_AUDIO_HOUR was measured at the shipped chunk size. Change +// DEFAULT_DIGEST_NUM_CTX or DEFAULT_DIGEST_MAX_CUES_PER_CHUNK and the real +// density moves while the constant does not — so every sweep projection is +// silently priced on a mix that no longer exists. Pinning the shipped size +// against the DEFAULTS (not against a hardcoded 8192, which is what let this +// through before) makes that change fail here and say what to do about it. +test("changing a DEFAULT_DIGEST_* forces the sweep to be re-priced", () => { + assert.equal( + maxCuesForContext(DEFAULT_DIGEST_NUM_CTX), + 600, + "the shipped chunk size changed — re-run `digest-plan --census` and update " + + "CORPUS_CHUNKS_PER_AUDIO_HOUR (and the band table beside it) before " + + "trusting any sweep projection", + ); + assert.equal(DEFAULT_DIGEST_MAX_CUES_PER_CHUNK, 600); + // And the recorded census must still reproduce the constant it was derived + // from. These are the totals from the run frozen in digestPlan.ts's comment — + // constants here, not a live read, so this asserts the ARITHMETIC rather than + // the operator's current corpus. + const RECORDED = { chunks: 191_116, audioHours: 77_298 }; assert.ok( - Math.abs(density - CORPUS_CHUNKS_PER_AUDIO_HOUR) < 0.01, - `census density ${density.toFixed(3)} != CORPUS_CHUNKS_PER_AUDIO_HOUR`, + Math.abs( + chunksPerAudioHour(RECORDED.chunks, RECORDED.audioHours * 3600) - + CORPUS_CHUNKS_PER_AUDIO_HOUR, + ) < 0.01, + "CORPUS_CHUNKS_PER_AUDIO_HOUR no longer matches the census it came from", ); }); diff --git a/common/controller/digestPlan.ts b/common/controller/digestPlan.ts @@ -90,6 +90,15 @@ export const MEASURED_SECONDS_PER_CHUNK = 24.7; // AUDIO but only 37% of the WORK, while sub-hour videos are 16% of audio and 34% // of chunks. Any reasoning that prices the sweep in audio-hours gets this // backwards. +// +// THE TABLE ABOVE IS ONE RUN, not a live number, and the corpus grows underneath +// it: re-measured 2026-08-08 it reads 74,321 videos / 194,053 chunks / density +// 2.46. That drift is why the census is `bin/digest-plan.ts --census` and not a +// unit test — an absolute count of a live corpus can only ever go red. Re-run it +// after a large ingest or ANY change to a DEFAULT_DIGEST_*; it prints the drift +// against this constant and says when the constant needs updating. The density +// is what the cost model consumes, and it has held to within 0.01 across ~3,000 +// new videos, which is the property worth having. export const CORPUS_CHUNKS_PER_AUDIO_HOUR = 2.47; // DERIVED, not measured — kept only so `bin/digest-plan.ts --rate` and anything @@ -246,6 +255,131 @@ export function chunksForVideo( }; } +// --------------------------------------------------------------------------- +// The chunk census +// --------------------------------------------------------------------------- + +// The duration bands the census is reported in. They exist because the headline +// result is a SHAPE, not a total: chunk density falls monotonically from 6.8 to +// 1.7 chunks per audio-hour as videos get longer, so 4 h+ videos are 52% of the +// audio but only 37% of the work. A single total hides exactly the thing that +// makes seconds-per-audio-hour an invalid unit. +export const CENSUS_BANDS: readonly { label: string; maxSeconds: number }[] = [ + { label: "< 15 min", maxSeconds: 15 * 60 }, + { label: "15–60 min", maxSeconds: 3600 }, + { label: "1–2 h", maxSeconds: 2 * 3600 }, + { label: "2–4 h", maxSeconds: 4 * 3600 }, + { label: "4–8 h", maxSeconds: 8 * 3600 }, + { label: "> 8 h", maxSeconds: Infinity }, +]; + +export type CensusBand = { + label: string; + videos: number; + audioSeconds: number; + chunks: number; +}; + +export type ChunkCensus = { + bands: CensusBand[]; + videos: number; + audioSeconds: number; + chunks: number; + // Rows with no recorded cueCount, whose chunks were ESTIMATED from duration. + // Reported rather than silently absorbed: the census's whole claim to being + // free is that cueCount was already on every row, and a non-zero here means + // part of the headline is a guess. + estimated: number; + chunksPerAudioHour: number; +}; + +// Bucket transcribed videos into the bands above and total their chunks. +// +// PURE, over an iterable of stats, and that is the point. This used to be +// inlined in a TEST that opened the operator's real LMDB and asserted the corpus +// totalled exactly 191,116 chunks across 73,367 videos. That test could only run +// on one machine, and it failed there the moment a video was downloaded — which +// is not a regression, it is the corpus growing. An absolute count of a live, +// changing corpus is a MEASUREMENT, not an invariant, and pinning it as a test +// meant the suite reported a permanent red for doing its job correctly. +// +// The two things that test was actually for are both kept, split apart: +// - "the plan and the chunker have drifted" is a pure property, asserted +// hermetically against this function and against countCueChunks. +// - "a DEFAULT_DIGEST_* changed the shipped chunk size and nobody re-priced +// the sweep" is pinned on the DENSITY constant, which is what the cost model +// consumes, rather than on a video count that moves on its own. +// The live numbers are printed by `bin/digest-plan.ts --census`, where a +// measurement belongs. +export function buildChunkCensus( + rows: Iterable<Pick<VideoStat, "cueCount" | "duration" | "hasTranscript">>, + maxCues: number, +): ChunkCensus { + const bands: CensusBand[] = CENSUS_BANDS.map((b) => ({ + label: b.label, + videos: 0, + audioSeconds: 0, + chunks: 0, + })); + let videos = 0; + let audioSeconds = 0; + let chunks = 0; + let estimated = 0; + + for (const stat of rows) { + // The same eligibility the plan uses: a video with no transcript is not work + // and a zero duration is a row we cannot band. + if (!stat.hasTranscript || !(stat.duration > 0)) continue; + const index = CENSUS_BANDS.findIndex((b) => stat.duration < b.maxSeconds); + const band = bands[index === -1 ? bands.length - 1 : index]; + const forVideo = chunksForVideo(stat, maxCues); + videos++; + audioSeconds += stat.duration; + chunks += forVideo.chunks; + if (forVideo.estimated) estimated++; + band.videos++; + band.audioSeconds += stat.duration; + band.chunks += forVideo.chunks; + } + + return { + bands, + videos, + audioSeconds, + chunks, + estimated, + chunksPerAudioHour: chunksPerAudioHour(chunks, audioSeconds), + }; +} + +// Run the census over the real corpus. The only impure part, and it does nothing +// but feed rows to the function above. +// +// Returns null when there is no stats cache on this machine, so the caller can +// say so rather than reporting a corpus of zero. +export async function readChunkCensus( + paths: Paths, + maxCues: number, +): Promise<(ChunkCensus & { statsSchemaStale: boolean }) | null> { + const { existsSync } = await import("node:fs"); + if (!existsSync(paths.lmdbPath)) return null; + const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true }); + const statsByPath = root.openDB< + { metaMs: number; stat: VideoStat }, + [string, string] + >({ name: "statsByPath", encoding: "msgpack" }); + const meta = root.openDB<unknown, string>({ + name: "statsMeta", + encoding: "msgpack", + }); + const statsSchemaStale = + (meta.get("schema") as number | undefined) !== STATS_SCHEMA_VERSION; + function* rows(): Generator<VideoStat> { + for (const { value } of statsByPath.getRange()) yield value.stat; + } + return { ...buildChunkCensus(rows(), maxCues), statsSchemaStale }; +} + // The whole role decision, as one pure function so it can be asserted on // without an LMDB corpus behind it. //