commit b9a871a8249bd7ad773ff57691211adcb0357fe3
parent 653bd65287646770448172696fff68aaac176c9c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 8 Aug 2026 00:16:25 -0400
Merge fix/census-as-report: make the corpus census a report, not a permanently-red test
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Diffstat:
3 files changed, 368 insertions(+), 50 deletions(-)
diff --git a/common/bin/digest-plan.ts b/common/bin/digest-plan.ts
@@ -24,18 +24,28 @@
// --channels a,b restrict to these channel slugs
// --top N how many channels to list (default 15, 0 = all)
// --json machine-readable, for diffing two runs
+// --census print the CORPUS CHUNK CENSUS by duration band and stop.
+// This is the measurement the density constant
+// CORPUS_CHUNKS_PER_AUDIO_HOUR is derived from, and the
+// table frozen in digestPlan.ts's comment is one run of
+// it. Re-run it after changing any DEFAULT_DIGEST_*, and
+// after any large ingest, to re-price the sweep.
import { getPaths } from "../lib/paths";
import {
buildDigestSweepPlan,
audioHours,
chunksPerAudioHour,
+ readChunkCensus,
sweepDays,
sweepDaysFromAudio,
+ CORPUS_CHUNKS_PER_AUDIO_HOUR,
DIGEST_PLAN_ROLES,
MEASURED_SECONDS_PER_AUDIO_HOUR,
MEASURED_SECONDS_PER_CHUNK,
roleMustGenerate,
} from "../controller/digestPlan";
+import { maxCuesForContext } from "../lib/digestPrompt";
+import { DEFAULT_DIGEST_NUM_CTX } from "../lib/digest";
import { parseFlags } from "./_parseFlags";
const flags = parseFlags(process.argv.slice(2));
@@ -73,7 +83,94 @@ function pct(part: number, whole: number): string {
return whole > 0 ? `${((part / whole) * 100).toFixed(1)}%` : "—";
}
+// The census: the corpus totals the cost model is derived from, printed rather
+// than asserted.
+//
+// It used to be a unit test that opened this machine's LMDB and asserted the
+// corpus was exactly 191,116 chunks across 73,367 videos. That could only pass
+// on one machine, and it went red there the moment a video was downloaded — the
+// corpus growing is not a regression. An absolute count of a live corpus is a
+// MEASUREMENT, and this is where a measurement goes. The properties that test
+// was really guarding (plan-vs-chunker drift, and a DEFAULT_DIGEST_* change
+// silently re-pricing the sweep) are pinned hermetically in
+// controller/digestPlan.test.ts instead.
+async function census(): Promise<void> {
+ const maxCues = maxCuesForContext(DEFAULT_DIGEST_NUM_CTX);
+ const result = await readChunkCensus(getPaths(), maxCues);
+ if (!result) {
+ console.log("");
+ console.log("No stats cache on this machine — run build:stats first.");
+ console.log("");
+ return;
+ }
+ if (asJson) {
+ console.log(JSON.stringify({ ...result, maxCues }, null, 2));
+ return;
+ }
+
+ console.log("");
+ console.log(
+ `Corpus chunk census — ${maxCues} cues/chunk at the shipped defaults.`,
+ );
+ if (result.statsSchemaStale)
+ console.log("!! stats cache is stale — run build:stats.");
+ console.log("");
+ console.log(
+ ` ${"band".padEnd(11)} ${"videos".padStart(7)} ${"audio-h".padStart(8)} ` +
+ `${"% audio".padStart(8)} ${"chunks".padStart(9)} ${"chunks/audio-h".padStart(15)} ${"% chunks".padStart(9)}`,
+ );
+ for (const band of result.bands) {
+ console.log(
+ ` ${band.label.padEnd(11)} ${count(band.videos).padStart(7)} ${hours(band.audioSeconds).padStart(8)} ` +
+ `${pct(band.audioSeconds, result.audioSeconds).padStart(8)} ${count(band.chunks).padStart(9)} ` +
+ `${chunksPerAudioHour(band.chunks, band.audioSeconds).toFixed(2).padStart(15)} ` +
+ `${pct(band.chunks, result.chunks).padStart(9)}`,
+ );
+ }
+ console.log(
+ ` ${"total".padEnd(11)} ${count(result.videos).padStart(7)} ${hours(result.audioSeconds).padStart(8)} ` +
+ `${"".padStart(8)} ${count(result.chunks).padStart(9)} ` +
+ `${result.chunksPerAudioHour.toFixed(2).padStart(15)}`,
+ );
+ console.log("");
+ if (result.estimated > 0) {
+ // The census's claim to costing nothing is that cueCount is already on every
+ // row. A non-zero here means part of the total above is a guess from
+ // duration, and it must not pass unremarked.
+ console.log(
+ `!! ${count(result.estimated)} row(s) had no cueCount; their chunks were ESTIMATED from duration.`,
+ );
+ }
+ console.log(
+ `Projected sweep: ${sweepDays(result.chunks, perChunk).toFixed(1)} days at ${perChunk}s per chunk.`,
+ );
+
+ // THE TRIPWIRE, and the reason to run this after touching a DEFAULT_DIGEST_*:
+ // the density constant is what the cost model consumes, and nothing else
+ // notices when the shipped chunk size stops matching the corpus it was
+ // measured on.
+ const drift = Math.abs(result.chunksPerAudioHour - CORPUS_CHUNKS_PER_AUDIO_HOUR);
+ console.log(
+ `Density ${result.chunksPerAudioHour.toFixed(3)} vs CORPUS_CHUNKS_PER_AUDIO_HOUR ` +
+ `${CORPUS_CHUNKS_PER_AUDIO_HOUR} (drift ${drift.toFixed(3)}).`,
+ );
+ if (drift >= 0.05) {
+ console.log(
+ `!! The constant no longer describes this corpus. Update ` +
+ `CORPUS_CHUNKS_PER_AUDIO_HOUR (and the band table beside it) in ` +
+ `controller/digestPlan.ts, or the sweep projection is priced on a mix ` +
+ `that no longer exists.`,
+ );
+ }
+ console.log("");
+}
+
async function main(): Promise<void> {
+ if (flags.census === "true") {
+ await census();
+ return;
+ }
+
const plan = await buildDigestSweepPlan({
paths: getPaths(),
lane,
diff --git a/common/controller/digestPlan.test.ts b/common/controller/digestPlan.test.ts
@@ -3,6 +3,7 @@ import assert from "node:assert/strict";
import type { DigestClusterRole } from "./digestSharing";
import {
audioHours,
+ buildChunkCensus,
chunksForVideo,
chunksPerAudioHour,
classifyDigestRole,
@@ -16,6 +17,10 @@ import {
} from "./digestPlan";
import { chunkCuesForContext, countCueChunks } from "../lib/transcriptWindow";
import { DIGEST_OVERLAP_CUES, maxCuesForContext } from "../lib/digestPrompt";
+import {
+ DEFAULT_DIGEST_MAX_CUES_PER_CHUNK,
+ DEFAULT_DIGEST_NUM_CTX,
+} from "../lib/digest";
// Run with:
// pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/digestPlan.test.ts
@@ -185,62 +190,144 @@ test("audio-hours, not video count, is what a saving is measured in", () => {
);
});
-// The census, as a live regression pin against the REAL corpus.
+// ---------------------------------------------------------------------------
+// The census
//
-// The pure test above pins the formula; this one pins the number the whole
-// re-pricing rests on. It re-derives the corpus chunk total from the same LMDB row
-// the plan reads, using the same helper, so it fails if the plan and the chunker
-// ever drift — or if a DEFAULT_DIGEST_* changes the shipped chunk size without
-// anyone re-pricing the sweep.
+// There used to be a test HERE that opened this machine's real LMDB and asserted
+// the corpus was exactly 191,116 chunks across 73,367 videos. It was skipped
+// everywhere else and permanently RED here, because the number moves whenever a
+// video is downloaded — 191,116 had already become 194,053. That is the corpus
+// growing, not a regression, and a suite that reports a red for it teaches people
+// to ignore reds.
//
-// Skipped when there is no corpus on this machine (CI, a fresh checkout): a pin on
-// data that isn't there would just be a broken test.
-test("the corpus chunk census reproduces 191,116 at the shipped config", async (t) => {
- const { existsSync } = await import("node:fs");
- const { getPaths } = await import("../lib/paths");
- const paths = getPaths();
- if (!existsSync(paths.lmdbPath)) {
- t.skip("no LMDB corpus on this machine");
- return;
- }
- const { open } = await import("lmdb");
- const { STATS_SCHEMA_VERSION } = await import("../lib/stats");
- const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true });
- const meta = root.openDB<unknown, string>({
- name: "statsMeta",
- encoding: "msgpack",
- });
- if ((meta.get("schema") as number | undefined) !== STATS_SCHEMA_VERSION) {
- t.skip("stats cache is at another schema version — run build:stats");
- return;
- }
- const statsByPath = root.openDB<
- { metaMs: number; stat: import("../lib/stats").VideoStat },
- [string, string]
- >({ name: "statsByPath", encoding: "msgpack" });
+// An absolute count of a live corpus is a MEASUREMENT. It now lives in
+// `bin/digest-plan.ts --census`, which prints the band table and warns when the
+// density constant stops describing the corpus.
+//
+// Its comment claimed two jobs. Both are kept below, hermetically:
+// 1. catch the plan and the chunker drifting apart — a pure property, and one
+// the test directly above already covers for countCueChunks;
+// 2. catch a DEFAULT_DIGEST_* change that alters the shipped chunk size
+// without re-pricing the sweep — pinned on the DENSITY constant, which is
+// what the cost model actually consumes, rather than on a video count that
+// moves on its own.
+// ---------------------------------------------------------------------------
+
+// One synthetic row per band, with cue counts chosen so the expected chunk count
+// is computable by hand.
+function stat(duration: number, cueCount: number | null) {
+ return { duration, cueCount, hasTranscript: true };
+}
+
+test("the census buckets by duration and totals chunks per band", () => {
+ const maxCues = maxCuesForContext(8192);
+ const census = buildChunkCensus(
+ [
+ stat(10 * 60, 600), // < 15 min → 1 chunk
+ stat(30 * 60, 601), // 15–60 min → 2
+ stat(90 * 60, 1_161), // 1–2 h → 3
+ stat(3 * 3600, 600), // 2–4 h → 1
+ stat(6 * 3600, 600), // 4–8 h → 1
+ stat(9 * 3600, 600), // > 8 h → 1
+ ],
+ maxCues,
+ );
+ assert.deepEqual(
+ census.bands.map((b) => [b.label, b.videos, b.chunks]),
+ [
+ ["< 15 min", 1, 1],
+ ["15–60 min", 1, 2],
+ ["1–2 h", 1, 3],
+ ["2–4 h", 1, 1],
+ ["4–8 h", 1, 1],
+ ["> 8 h", 1, 1],
+ ],
+ );
+ assert.equal(census.videos, 6);
+ assert.equal(census.chunks, 9);
+ assert.equal(census.estimated, 0);
+ // Band EDGES land in the lower band, so a boundary video cannot be counted
+ // twice or dropped — 15:00 exactly is "15–60 min", not "< 15 min".
+ const edges = buildChunkCensus(
+ [stat(15 * 60, 600), stat(8 * 3600, 600)],
+ maxCues,
+ );
+ assert.equal(edges.bands[0].videos, 0);
+ assert.equal(edges.bands[1].videos, 1);
+ assert.equal(edges.bands[4].videos, 0);
+ assert.equal(edges.bands[5].videos, 1);
+});
+
+test("the census counts only transcribed videos, and flags estimated rows", () => {
+ const maxCues = maxCuesForContext(8192);
+ const census = buildChunkCensus(
+ [
+ stat(3600, 600),
+ { duration: 3600, cueCount: 600, hasTranscript: false },
+ // A zero-duration row cannot be banded and is not work.
+ stat(0, 600),
+ // No cueCount: estimated from duration rather than dropped, because
+ // understating the sweep is the worse failure — and COUNTED, because the
+ // census's claim to costing nothing is that cueCount is on every row.
+ stat(4 * 3600, null),
+ ],
+ maxCues,
+ );
+ assert.equal(census.videos, 2);
+ assert.equal(census.estimated, 1);
+ assert.equal(
+ census.chunks,
+ 1 + Math.round(4 * CORPUS_CHUNKS_PER_AUDIO_HOUR),
+ );
+});
+// The plan and the chunker, one function — asserted through the census this time,
+// so the path the report actually takes is the path under test.
+test("the census prices a video exactly as the real chunker slices it", () => {
const maxCues = maxCuesForContext(8192);
- let chunks = 0;
- let videos = 0;
- let audioSeconds = 0;
- let missingCueCount = 0;
- for (const { value } of statsByPath.getRange()) {
- const stat = value.stat;
- if (!stat.hasTranscript || !(stat.duration > 0)) continue;
- videos++;
- audioSeconds += stat.duration;
- if (typeof stat.cueCount !== "number") missingCueCount++;
- chunks += chunksForVideo(stat, maxCues).chunks;
+ const opts = { maxCues, overlapCues: DIGEST_OVERLAP_CUES };
+ for (const n of [1, 599, 600, 601, 1_160, 1_161, 5_000]) {
+ const cues = Array.from({ length: n }, (_, i) => ({
+ start: i,
+ end: i + 1,
+ text: "x",
+ }));
+ assert.equal(
+ buildChunkCensus([stat(3600, n)], maxCues).chunks,
+ chunkCuesForContext(cues, opts).length,
+ `census disagrees with the chunker at ${n} cues`,
+ );
}
+});
- assert.equal(chunks, 191_116, "corpus chunk total moved");
- assert.equal(videos, 73_367, "transcribed video count moved");
- // The census cost nothing precisely because cueCount was already on every row.
- assert.equal(missingCueCount, 0, "some rows lost their cueCount");
- // And the derived density constant must still describe this corpus.
- const density = chunksPerAudioHour(chunks, audioSeconds);
+// THE TRIPWIRE THE LIVE TEST WAS REALLY FOR, and the whole reason it can be
+// hermetic: the cost model consumes the DENSITY, not a video count.
+//
+// CORPUS_CHUNKS_PER_AUDIO_HOUR was measured at the shipped chunk size. Change
+// DEFAULT_DIGEST_NUM_CTX or DEFAULT_DIGEST_MAX_CUES_PER_CHUNK and the real
+// density moves while the constant does not — so every sweep projection is
+// silently priced on a mix that no longer exists. Pinning the shipped size
+// against the DEFAULTS (not against a hardcoded 8192, which is what let this
+// through before) makes that change fail here and say what to do about it.
+test("changing a DEFAULT_DIGEST_* forces the sweep to be re-priced", () => {
+ assert.equal(
+ maxCuesForContext(DEFAULT_DIGEST_NUM_CTX),
+ 600,
+ "the shipped chunk size changed — re-run `digest-plan --census` and update " +
+ "CORPUS_CHUNKS_PER_AUDIO_HOUR (and the band table beside it) before " +
+ "trusting any sweep projection",
+ );
+ assert.equal(DEFAULT_DIGEST_MAX_CUES_PER_CHUNK, 600);
+ // And the recorded census must still reproduce the constant it was derived
+ // from. These are the totals from the run frozen in digestPlan.ts's comment —
+ // constants here, not a live read, so this asserts the ARITHMETIC rather than
+ // the operator's current corpus.
+ const RECORDED = { chunks: 191_116, audioHours: 77_298 };
assert.ok(
- Math.abs(density - CORPUS_CHUNKS_PER_AUDIO_HOUR) < 0.01,
- `census density ${density.toFixed(3)} != CORPUS_CHUNKS_PER_AUDIO_HOUR`,
+ Math.abs(
+ chunksPerAudioHour(RECORDED.chunks, RECORDED.audioHours * 3600) -
+ CORPUS_CHUNKS_PER_AUDIO_HOUR,
+ ) < 0.01,
+ "CORPUS_CHUNKS_PER_AUDIO_HOUR no longer matches the census it came from",
);
});
diff --git a/common/controller/digestPlan.ts b/common/controller/digestPlan.ts
@@ -90,6 +90,15 @@ export const MEASURED_SECONDS_PER_CHUNK = 24.7;
// AUDIO but only 37% of the WORK, while sub-hour videos are 16% of audio and 34%
// of chunks. Any reasoning that prices the sweep in audio-hours gets this
// backwards.
+//
+// THE TABLE ABOVE IS ONE RUN, not a live number, and the corpus grows underneath
+// it: re-measured 2026-08-08 it reads 74,321 videos / 194,053 chunks / density
+// 2.46. That drift is why the census is `bin/digest-plan.ts --census` and not a
+// unit test — an absolute count of a live corpus can only ever go red. Re-run it
+// after a large ingest or ANY change to a DEFAULT_DIGEST_*; it prints the drift
+// against this constant and says when the constant needs updating. The density
+// is what the cost model consumes, and it has held to within 0.01 across ~3,000
+// new videos, which is the property worth having.
export const CORPUS_CHUNKS_PER_AUDIO_HOUR = 2.47;
// DERIVED, not measured — kept only so `bin/digest-plan.ts --rate` and anything
@@ -246,6 +255,131 @@ export function chunksForVideo(
};
}
+// ---------------------------------------------------------------------------
+// The chunk census
+// ---------------------------------------------------------------------------
+
+// The duration bands the census is reported in. They exist because the headline
+// result is a SHAPE, not a total: chunk density falls monotonically from 6.8 to
+// 1.7 chunks per audio-hour as videos get longer, so 4 h+ videos are 52% of the
+// audio but only 37% of the work. A single total hides exactly the thing that
+// makes seconds-per-audio-hour an invalid unit.
+export const CENSUS_BANDS: readonly { label: string; maxSeconds: number }[] = [
+ { label: "< 15 min", maxSeconds: 15 * 60 },
+ { label: "15–60 min", maxSeconds: 3600 },
+ { label: "1–2 h", maxSeconds: 2 * 3600 },
+ { label: "2–4 h", maxSeconds: 4 * 3600 },
+ { label: "4–8 h", maxSeconds: 8 * 3600 },
+ { label: "> 8 h", maxSeconds: Infinity },
+];
+
+export type CensusBand = {
+ label: string;
+ videos: number;
+ audioSeconds: number;
+ chunks: number;
+};
+
+export type ChunkCensus = {
+ bands: CensusBand[];
+ videos: number;
+ audioSeconds: number;
+ chunks: number;
+ // Rows with no recorded cueCount, whose chunks were ESTIMATED from duration.
+ // Reported rather than silently absorbed: the census's whole claim to being
+ // free is that cueCount was already on every row, and a non-zero here means
+ // part of the headline is a guess.
+ estimated: number;
+ chunksPerAudioHour: number;
+};
+
+// Bucket transcribed videos into the bands above and total their chunks.
+//
+// PURE, over an iterable of stats, and that is the point. This used to be
+// inlined in a TEST that opened the operator's real LMDB and asserted the corpus
+// totalled exactly 191,116 chunks across 73,367 videos. That test could only run
+// on one machine, and it failed there the moment a video was downloaded — which
+// is not a regression, it is the corpus growing. An absolute count of a live,
+// changing corpus is a MEASUREMENT, not an invariant, and pinning it as a test
+// meant the suite reported a permanent red for doing its job correctly.
+//
+// The two things that test was actually for are both kept, split apart:
+// - "the plan and the chunker have drifted" is a pure property, asserted
+// hermetically against this function and against countCueChunks.
+// - "a DEFAULT_DIGEST_* changed the shipped chunk size and nobody re-priced
+// the sweep" is pinned on the DENSITY constant, which is what the cost model
+// consumes, rather than on a video count that moves on its own.
+// The live numbers are printed by `bin/digest-plan.ts --census`, where a
+// measurement belongs.
+export function buildChunkCensus(
+ rows: Iterable<Pick<VideoStat, "cueCount" | "duration" | "hasTranscript">>,
+ maxCues: number,
+): ChunkCensus {
+ const bands: CensusBand[] = CENSUS_BANDS.map((b) => ({
+ label: b.label,
+ videos: 0,
+ audioSeconds: 0,
+ chunks: 0,
+ }));
+ let videos = 0;
+ let audioSeconds = 0;
+ let chunks = 0;
+ let estimated = 0;
+
+ for (const stat of rows) {
+ // The same eligibility the plan uses: a video with no transcript is not work
+ // and a zero duration is a row we cannot band.
+ if (!stat.hasTranscript || !(stat.duration > 0)) continue;
+ const index = CENSUS_BANDS.findIndex((b) => stat.duration < b.maxSeconds);
+ const band = bands[index === -1 ? bands.length - 1 : index];
+ const forVideo = chunksForVideo(stat, maxCues);
+ videos++;
+ audioSeconds += stat.duration;
+ chunks += forVideo.chunks;
+ if (forVideo.estimated) estimated++;
+ band.videos++;
+ band.audioSeconds += stat.duration;
+ band.chunks += forVideo.chunks;
+ }
+
+ return {
+ bands,
+ videos,
+ audioSeconds,
+ chunks,
+ estimated,
+ chunksPerAudioHour: chunksPerAudioHour(chunks, audioSeconds),
+ };
+}
+
+// Run the census over the real corpus. The only impure part, and it does nothing
+// but feed rows to the function above.
+//
+// Returns null when there is no stats cache on this machine, so the caller can
+// say so rather than reporting a corpus of zero.
+export async function readChunkCensus(
+ paths: Paths,
+ maxCues: number,
+): Promise<(ChunkCensus & { statsSchemaStale: boolean }) | null> {
+ const { existsSync } = await import("node:fs");
+ if (!existsSync(paths.lmdbPath)) return null;
+ const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true });
+ const statsByPath = root.openDB<
+ { metaMs: number; stat: VideoStat },
+ [string, string]
+ >({ name: "statsByPath", encoding: "msgpack" });
+ const meta = root.openDB<unknown, string>({
+ name: "statsMeta",
+ encoding: "msgpack",
+ });
+ const statsSchemaStale =
+ (meta.get("schema") as number | undefined) !== STATS_SCHEMA_VERSION;
+ function* rows(): Generator<VideoStat> {
+ for (const { value } of statsByPath.getRange()) yield value.stat;
+ }
+ return { ...buildChunkCensus(rows(), maxCues), statsSchemaStale };
+}
+
// The whole role decision, as one pure function so it can be asserted on
// without an LMDB corpus behind it.
//