Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit b9a871a8249bd7ad773ff57691211adcb0357fe3
parent 653bd65287646770448172696fff68aaac176c9c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat,  8 Aug 2026 00:16:25 -0400

Merge fix/census-as-report: make the corpus census a report, not a permanently-red test

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

Diffstat:
Mcommon/bin/digest-plan.ts | 97+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/digestPlan.test.ts | 187++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------------------
Mcommon/controller/digestPlan.ts | 134+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
3 files changed, 368 insertions(+), 50 deletions(-)

diff --git a/common/bin/digest-plan.ts b/common/bin/digest-plan.ts @@ -24,18 +24,28 @@ // --channels a,b restrict to these channel slugs // --top N how many channels to list (default 15, 0 = all) // --json machine-readable, for diffing two runs +// --census print the CORPUS CHUNK CENSUS by duration band and stop. +// This is the measurement the density constant +// CORPUS_CHUNKS_PER_AUDIO_HOUR is derived from, and the +// table frozen in digestPlan.ts's comment is one run of +// it. Re-run it after changing any DEFAULT_DIGEST_*, and +// after any large ingest, to re-price the sweep. import { getPaths } from "../lib/paths"; import { buildDigestSweepPlan, audioHours, chunksPerAudioHour, + readChunkCensus, sweepDays, sweepDaysFromAudio, + CORPUS_CHUNKS_PER_AUDIO_HOUR, DIGEST_PLAN_ROLES, MEASURED_SECONDS_PER_AUDIO_HOUR, MEASURED_SECONDS_PER_CHUNK, roleMustGenerate, } from "../controller/digestPlan"; +import { maxCuesForContext } from "../lib/digestPrompt"; +import { DEFAULT_DIGEST_NUM_CTX } from "../lib/digest"; import { parseFlags } from "./_parseFlags"; const flags = parseFlags(process.argv.slice(2)); @@ -73,7 +83,94 @@ function pct(part: number, whole: number): string { return whole > 0 ? `${((part / whole) * 100).toFixed(1)}%` : "—"; } +// The census: the corpus totals the cost model is derived from, printed rather +// than asserted. +// +// It used to be a unit test that opened this machine's LMDB and asserted the +// corpus was exactly 191,116 chunks across 73,367 videos. That could only pass +// on one machine, and it went red there the moment a video was downloaded — the +// corpus growing is not a regression. An absolute count of a live corpus is a +// MEASUREMENT, and this is where a measurement goes. The properties that test +// was really guarding (plan-vs-chunker drift, and a DEFAULT_DIGEST_* change +// silently re-pricing the sweep) are pinned hermetically in +// controller/digestPlan.test.ts instead. +async function census(): Promise<void> { + const maxCues = maxCuesForContext(DEFAULT_DIGEST_NUM_CTX); + const result = await readChunkCensus(getPaths(), maxCues); + if (!result) { + console.log(""); + console.log("No stats cache on this machine — run build:stats first."); + console.log(""); + return; + } + if (asJson) { + console.log(JSON.stringify({ ...result, maxCues }, null, 2)); + return; + } + + console.log(""); + console.log( + `Corpus chunk census — ${maxCues} cues/chunk at the shipped defaults.`, + ); + if (result.statsSchemaStale) + console.log("!! stats cache is stale — run build:stats."); + console.log(""); + console.log( + ` ${"band".padEnd(11)} ${"videos".padStart(7)} ${"audio-h".padStart(8)} ` + + `${"% audio".padStart(8)} ${"chunks".padStart(9)} ${"chunks/audio-h".padStart(15)} ${"% chunks".padStart(9)}`, + ); + for (const band of result.bands) { + console.log( + ` ${band.label.padEnd(11)} ${count(band.videos).padStart(7)} ${hours(band.audioSeconds).padStart(8)} ` + + `${pct(band.audioSeconds, result.audioSeconds).padStart(8)} ${count(band.chunks).padStart(9)} ` + + `${chunksPerAudioHour(band.chunks, band.audioSeconds).toFixed(2).padStart(15)} ` + + `${pct(band.chunks, result.chunks).padStart(9)}`, + ); + } + console.log( + ` ${"total".padEnd(11)} ${count(result.videos).padStart(7)} ${hours(result.audioSeconds).padStart(8)} ` + + `${"".padStart(8)} ${count(result.chunks).padStart(9)} ` + + `${result.chunksPerAudioHour.toFixed(2).padStart(15)}`, + ); + console.log(""); + if (result.estimated > 0) { + // The census's claim to costing nothing is that cueCount is already on every + // row. A non-zero here means part of the total above is a guess from + // duration, and it must not pass unremarked. + console.log( + `!! ${count(result.estimated)} row(s) had no cueCount; their chunks were ESTIMATED from duration.`, + ); + } + console.log( + `Projected sweep: ${sweepDays(result.chunks, perChunk).toFixed(1)} days at ${perChunk}s per chunk.`, + ); + + // THE TRIPWIRE, and the reason to run this after touching a DEFAULT_DIGEST_*: + // the density constant is what the cost model consumes, and nothing else + // notices when the shipped chunk size stops matching the corpus it was + // measured on. + const drift = Math.abs(result.chunksPerAudioHour - CORPUS_CHUNKS_PER_AUDIO_HOUR); + console.log( + `Density ${result.chunksPerAudioHour.toFixed(3)} vs CORPUS_CHUNKS_PER_AUDIO_HOUR ` + + `${CORPUS_CHUNKS_PER_AUDIO_HOUR} (drift ${drift.toFixed(3)}).`, + ); + if (drift >= 0.05) { + console.log( + `!! The constant no longer describes this corpus. Update ` + + `CORPUS_CHUNKS_PER_AUDIO_HOUR (and the band table beside it) in ` + + `controller/digestPlan.ts, or the sweep projection is priced on a mix ` + + `that no longer exists.`, + ); + } + console.log(""); +} + async function main(): Promise<void> { + if (flags.census === "true") { + await census(); + return; + } + const plan = await buildDigestSweepPlan({ paths: getPaths(), lane, diff --git a/common/controller/digestPlan.test.ts b/common/controller/digestPlan.test.ts @@ -3,6 +3,7 @@ import assert from "node:assert/strict"; import type { DigestClusterRole } from "./digestSharing"; import { audioHours, + buildChunkCensus, chunksForVideo, chunksPerAudioHour, classifyDigestRole, @@ -16,6 +17,10 @@ import { } from "./digestPlan"; import { chunkCuesForContext, countCueChunks } from "../lib/transcriptWindow"; import { DIGEST_OVERLAP_CUES, maxCuesForContext } from "../lib/digestPrompt"; +import { + DEFAULT_DIGEST_MAX_CUES_PER_CHUNK, + DEFAULT_DIGEST_NUM_CTX, +} from "../lib/digest"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/digestPlan.test.ts @@ -185,62 +190,144 @@ test("audio-hours, not video count, is what a saving is measured in", () => { ); }); -// The census, as a live regression pin against the REAL corpus. +// --------------------------------------------------------------------------- +// The census // -// The pure test above pins the formula; this one pins the number the whole -// re-pricing rests on. It re-derives the corpus chunk total from the same LMDB row -// the plan reads, using the same helper, so it fails if the plan and the chunker -// ever drift — or if a DEFAULT_DIGEST_* changes the shipped chunk size without -// anyone re-pricing the sweep. +// There used to be a test HERE that opened this machine's real LMDB and asserted +// the corpus was exactly 191,116 chunks across 73,367 videos. It was skipped +// everywhere else and permanently RED here, because the number moves whenever a +// video is downloaded — 191,116 had already become 194,053. That is the corpus +// growing, not a regression, and a suite that reports a red for it teaches people +// to ignore reds. // -// Skipped when there is no corpus on this machine (CI, a fresh checkout): a pin on -// data that isn't there would just be a broken test. -test("the corpus chunk census reproduces 191,116 at the shipped config", async (t) => { - const { existsSync } = await import("node:fs"); - const { getPaths } = await import("../lib/paths"); - const paths = getPaths(); - if (!existsSync(paths.lmdbPath)) { - t.skip("no LMDB corpus on this machine"); - return; - } - const { open } = await import("lmdb"); - const { STATS_SCHEMA_VERSION } = await import("../lib/stats"); - const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true }); - const meta = root.openDB<unknown, string>({ - name: "statsMeta", - encoding: "msgpack", - }); - if ((meta.get("schema") as number | undefined) !== STATS_SCHEMA_VERSION) { - t.skip("stats cache is at another schema version — run build:stats"); - return; - } - const statsByPath = root.openDB< - { metaMs: number; stat: import("../lib/stats").VideoStat }, - [string, string] - >({ name: "statsByPath", encoding: "msgpack" }); +// An absolute count of a live corpus is a MEASUREMENT. It now lives in +// `bin/digest-plan.ts --census`, which prints the band table and warns when the +// density constant stops describing the corpus. +// +// Its comment claimed two jobs. Both are kept below, hermetically: +// 1. catch the plan and the chunker drifting apart — a pure property, and one +// the test directly above already covers for countCueChunks; +// 2. catch a DEFAULT_DIGEST_* change that alters the shipped chunk size +// without re-pricing the sweep — pinned on the DENSITY constant, which is +// what the cost model actually consumes, rather than on a video count that +// moves on its own. +// --------------------------------------------------------------------------- + +// One synthetic row per band, with cue counts chosen so the expected chunk count +// is computable by hand. +function stat(duration: number, cueCount: number | null) { + return { duration, cueCount, hasTranscript: true }; +} + +test("the census buckets by duration and totals chunks per band", () => { + const maxCues = maxCuesForContext(8192); + const census = buildChunkCensus( + [ + stat(10 * 60, 600), // < 15 min → 1 chunk + stat(30 * 60, 601), // 15–60 min → 2 + stat(90 * 60, 1_161), // 1–2 h → 3 + stat(3 * 3600, 600), // 2–4 h → 1 + stat(6 * 3600, 600), // 4–8 h → 1 + stat(9 * 3600, 600), // > 8 h → 1 + ], + maxCues, + ); + assert.deepEqual( + census.bands.map((b) => [b.label, b.videos, b.chunks]), + [ + ["< 15 min", 1, 1], + ["15–60 min", 1, 2], + ["1–2 h", 1, 3], + ["2–4 h", 1, 1], + ["4–8 h", 1, 1], + ["> 8 h", 1, 1], + ], + ); + assert.equal(census.videos, 6); + assert.equal(census.chunks, 9); + assert.equal(census.estimated, 0); + // Band EDGES land in the lower band, so a boundary video cannot be counted + // twice or dropped — 15:00 exactly is "15–60 min", not "< 15 min". + const edges = buildChunkCensus( + [stat(15 * 60, 600), stat(8 * 3600, 600)], + maxCues, + ); + assert.equal(edges.bands[0].videos, 0); + assert.equal(edges.bands[1].videos, 1); + assert.equal(edges.bands[4].videos, 0); + assert.equal(edges.bands[5].videos, 1); +}); + +test("the census counts only transcribed videos, and flags estimated rows", () => { + const maxCues = maxCuesForContext(8192); + const census = buildChunkCensus( + [ + stat(3600, 600), + { duration: 3600, cueCount: 600, hasTranscript: false }, + // A zero-duration row cannot be banded and is not work. + stat(0, 600), + // No cueCount: estimated from duration rather than dropped, because + // understating the sweep is the worse failure — and COUNTED, because the + // census's claim to costing nothing is that cueCount is on every row. + stat(4 * 3600, null), + ], + maxCues, + ); + assert.equal(census.videos, 2); + assert.equal(census.estimated, 1); + assert.equal( + census.chunks, + 1 + Math.round(4 * CORPUS_CHUNKS_PER_AUDIO_HOUR), + ); +}); +// The plan and the chunker, one function — asserted through the census this time, +// so the path the report actually takes is the path under test. +test("the census prices a video exactly as the real chunker slices it", () => { const maxCues = maxCuesForContext(8192); - let chunks = 0; - let videos = 0; - let audioSeconds = 0; - let missingCueCount = 0; - for (const { value } of statsByPath.getRange()) { - const stat = value.stat; - if (!stat.hasTranscript || !(stat.duration > 0)) continue; - videos++; - audioSeconds += stat.duration; - if (typeof stat.cueCount !== "number") missingCueCount++; - chunks += chunksForVideo(stat, maxCues).chunks; + const opts = { maxCues, overlapCues: DIGEST_OVERLAP_CUES }; + for (const n of [1, 599, 600, 601, 1_160, 1_161, 5_000]) { + const cues = Array.from({ length: n }, (_, i) => ({ + start: i, + end: i + 1, + text: "x", + })); + assert.equal( + buildChunkCensus([stat(3600, n)], maxCues).chunks, + chunkCuesForContext(cues, opts).length, + `census disagrees with the chunker at ${n} cues`, + ); } +}); - assert.equal(chunks, 191_116, "corpus chunk total moved"); - assert.equal(videos, 73_367, "transcribed video count moved"); - // The census cost nothing precisely because cueCount was already on every row. - assert.equal(missingCueCount, 0, "some rows lost their cueCount"); - // And the derived density constant must still describe this corpus. - const density = chunksPerAudioHour(chunks, audioSeconds); +// THE TRIPWIRE THE LIVE TEST WAS REALLY FOR, and the whole reason it can be +// hermetic: the cost model consumes the DENSITY, not a video count. +// +// CORPUS_CHUNKS_PER_AUDIO_HOUR was measured at the shipped chunk size. Change +// DEFAULT_DIGEST_NUM_CTX or DEFAULT_DIGEST_MAX_CUES_PER_CHUNK and the real +// density moves while the constant does not — so every sweep projection is +// silently priced on a mix that no longer exists. Pinning the shipped size +// against the DEFAULTS (not against a hardcoded 8192, which is what let this +// through before) makes that change fail here and say what to do about it. +test("changing a DEFAULT_DIGEST_* forces the sweep to be re-priced", () => { + assert.equal( + maxCuesForContext(DEFAULT_DIGEST_NUM_CTX), + 600, + "the shipped chunk size changed — re-run `digest-plan --census` and update " + + "CORPUS_CHUNKS_PER_AUDIO_HOUR (and the band table beside it) before " + + "trusting any sweep projection", + ); + assert.equal(DEFAULT_DIGEST_MAX_CUES_PER_CHUNK, 600); + // And the recorded census must still reproduce the constant it was derived + // from. These are the totals from the run frozen in digestPlan.ts's comment — + // constants here, not a live read, so this asserts the ARITHMETIC rather than + // the operator's current corpus. + const RECORDED = { chunks: 191_116, audioHours: 77_298 }; assert.ok( - Math.abs(density - CORPUS_CHUNKS_PER_AUDIO_HOUR) < 0.01, - `census density ${density.toFixed(3)} != CORPUS_CHUNKS_PER_AUDIO_HOUR`, + Math.abs( + chunksPerAudioHour(RECORDED.chunks, RECORDED.audioHours * 3600) - + CORPUS_CHUNKS_PER_AUDIO_HOUR, + ) < 0.01, + "CORPUS_CHUNKS_PER_AUDIO_HOUR no longer matches the census it came from", ); }); diff --git a/common/controller/digestPlan.ts b/common/controller/digestPlan.ts @@ -90,6 +90,15 @@ export const MEASURED_SECONDS_PER_CHUNK = 24.7; // AUDIO but only 37% of the WORK, while sub-hour videos are 16% of audio and 34% // of chunks. Any reasoning that prices the sweep in audio-hours gets this // backwards. +// +// THE TABLE ABOVE IS ONE RUN, not a live number, and the corpus grows underneath +// it: re-measured 2026-08-08 it reads 74,321 videos / 194,053 chunks / density +// 2.46. That drift is why the census is `bin/digest-plan.ts --census` and not a +// unit test — an absolute count of a live corpus can only ever go red. Re-run it +// after a large ingest or ANY change to a DEFAULT_DIGEST_*; it prints the drift +// against this constant and says when the constant needs updating. The density +// is what the cost model consumes, and it has held to within 0.01 across ~3,000 +// new videos, which is the property worth having. export const CORPUS_CHUNKS_PER_AUDIO_HOUR = 2.47; // DERIVED, not measured — kept only so `bin/digest-plan.ts --rate` and anything @@ -246,6 +255,131 @@ export function chunksForVideo( }; } +// --------------------------------------------------------------------------- +// The chunk census +// --------------------------------------------------------------------------- + +// The duration bands the census is reported in. They exist because the headline +// result is a SHAPE, not a total: chunk density falls monotonically from 6.8 to +// 1.7 chunks per audio-hour as videos get longer, so 4 h+ videos are 52% of the +// audio but only 37% of the work. A single total hides exactly the thing that +// makes seconds-per-audio-hour an invalid unit. +export const CENSUS_BANDS: readonly { label: string; maxSeconds: number }[] = [ + { label: "< 15 min", maxSeconds: 15 * 60 }, + { label: "15–60 min", maxSeconds: 3600 }, + { label: "1–2 h", maxSeconds: 2 * 3600 }, + { label: "2–4 h", maxSeconds: 4 * 3600 }, + { label: "4–8 h", maxSeconds: 8 * 3600 }, + { label: "> 8 h", maxSeconds: Infinity }, +]; + +export type CensusBand = { + label: string; + videos: number; + audioSeconds: number; + chunks: number; +}; + +export type ChunkCensus = { + bands: CensusBand[]; + videos: number; + audioSeconds: number; + chunks: number; + // Rows with no recorded cueCount, whose chunks were ESTIMATED from duration. + // Reported rather than silently absorbed: the census's whole claim to being + // free is that cueCount was already on every row, and a non-zero here means + // part of the headline is a guess. + estimated: number; + chunksPerAudioHour: number; +}; + +// Bucket transcribed videos into the bands above and total their chunks. +// +// PURE, over an iterable of stats, and that is the point. This used to be +// inlined in a TEST that opened the operator's real LMDB and asserted the corpus +// totalled exactly 191,116 chunks across 73,367 videos. That test could only run +// on one machine, and it failed there the moment a video was downloaded — which +// is not a regression, it is the corpus growing. An absolute count of a live, +// changing corpus is a MEASUREMENT, not an invariant, and pinning it as a test +// meant the suite reported a permanent red for doing its job correctly. +// +// The two things that test was actually for are both kept, split apart: +// - "the plan and the chunker have drifted" is a pure property, asserted +// hermetically against this function and against countCueChunks. +// - "a DEFAULT_DIGEST_* changed the shipped chunk size and nobody re-priced +// the sweep" is pinned on the DENSITY constant, which is what the cost model +// consumes, rather than on a video count that moves on its own. +// The live numbers are printed by `bin/digest-plan.ts --census`, where a +// measurement belongs. +export function buildChunkCensus( + rows: Iterable<Pick<VideoStat, "cueCount" | "duration" | "hasTranscript">>, + maxCues: number, +): ChunkCensus { + const bands: CensusBand[] = CENSUS_BANDS.map((b) => ({ + label: b.label, + videos: 0, + audioSeconds: 0, + chunks: 0, + })); + let videos = 0; + let audioSeconds = 0; + let chunks = 0; + let estimated = 0; + + for (const stat of rows) { + // The same eligibility the plan uses: a video with no transcript is not work + // and a zero duration is a row we cannot band. + if (!stat.hasTranscript || !(stat.duration > 0)) continue; + const index = CENSUS_BANDS.findIndex((b) => stat.duration < b.maxSeconds); + const band = bands[index === -1 ? bands.length - 1 : index]; + const forVideo = chunksForVideo(stat, maxCues); + videos++; + audioSeconds += stat.duration; + chunks += forVideo.chunks; + if (forVideo.estimated) estimated++; + band.videos++; + band.audioSeconds += stat.duration; + band.chunks += forVideo.chunks; + } + + return { + bands, + videos, + audioSeconds, + chunks, + estimated, + chunksPerAudioHour: chunksPerAudioHour(chunks, audioSeconds), + }; +} + +// Run the census over the real corpus. The only impure part, and it does nothing +// but feed rows to the function above. +// +// Returns null when there is no stats cache on this machine, so the caller can +// say so rather than reporting a corpus of zero. +export async function readChunkCensus( + paths: Paths, + maxCues: number, +): Promise<(ChunkCensus & { statsSchemaStale: boolean }) | null> { + const { existsSync } = await import("node:fs"); + if (!existsSync(paths.lmdbPath)) return null; + const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true }); + const statsByPath = root.openDB< + { metaMs: number; stat: VideoStat }, + [string, string] + >({ name: "statsByPath", encoding: "msgpack" }); + const meta = root.openDB<unknown, string>({ + name: "statsMeta", + encoding: "msgpack", + }); + const statsSchemaStale = + (meta.get("schema") as number | undefined) !== STATS_SCHEMA_VERSION; + function* rows(): Generator<VideoStat> { + for (const { value } of statsByPath.getRange()) yield value.stat; + } + return { ...buildChunkCensus(rows(), maxCues), statsSchemaStale }; +} + // The whole role decision, as one pure function so it can be asserted on // without an LMDB corpus behind it. //