import { test } from "node:test"; import assert from "node:assert/strict"; import type { DigestClusterRole } from "./digestSharing"; import { audioHours, buildChunkCensus, chunksForVideo, chunksPerAudioHour, classifyDigestRole, CORPUS_CHUNKS_PER_AUDIO_HOUR, DIGEST_PLAN_ROLES, MEASURED_SECONDS_PER_AUDIO_HOUR, MEASURED_SECONDS_PER_CHUNK, roleMustGenerate, sweepDays, sweepDaysFromAudio, } from "./digestPlan"; import { chunkCuesForContext, countCueChunks } from "../lib/transcriptWindow"; import { DIGEST_OVERLAP_CUES, maxCuesForContext } from "../lib/digestPrompt"; import { DEFAULT_DIGEST_MAX_CUES_PER_CHUNK, DEFAULT_DIGEST_NUM_CTX, } from "../lib/digest"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/digestPlan.test.ts test("a video in no cluster is unclustered work", () => { assert.equal(classifyDigestRole(undefined, undefined), "unclustered"); // Alignment is meaningless without a cluster and must not change the answer. assert.equal(classifyDigestRole(undefined, true), "unclustered"); }); test("a canonical member is work regardless of alignment", () => { const role: DigestClusterRole = { kind: "canonical", clusterId: "c1", mirrors: ["a"], }; assert.equal(classifyDigestRole(role, true), "canonical"); assert.equal(classifyDigestRole(role, false), "canonical"); }); // The distinction the whole plan turns on: only an ALIGNED mirror is free. The // sharing pass refuses to place a digest without measured alignment, so counting // every mirror as free overstates the saving by however many it will refuse — // nearly half, on the current corpus. test("only an explicitly-aligned mirror is free", () => { const role: DigestClusterRole = { kind: "mirror", clusterId: "c1", canonicalSlug: "x", }; assert.equal(classifyDigestRole(role, true), "mirror-aligned"); assert.equal(classifyDigestRole(role, false), "mirror-unaligned"); }); // Absent means "never measured", which sharing treats as NOT aligned. The plan // must under-promise here, never over-promise. test("an unmeasured mirror counts as work, not as a saving", () => { const role: DigestClusterRole = { kind: "mirror", clusterId: "c1", canonicalSlug: "x", }; assert.equal(classifyDigestRole(role, undefined), "mirror-unaligned"); assert.equal(roleMustGenerate(classifyDigestRole(role, undefined)), true); }); test("exactly one role is free; every other role costs GPU time", () => { const free = DIGEST_PLAN_ROLES.filter((r) => !roleMustGenerate(r)); assert.deepEqual(free, ["mirror-aligned"]); }); test("audio-hours and the legacy audio-hour projection still agree", () => { assert.equal(audioHours(3600), 1); // One audio-hour at 90 s/audio-hour is 90 seconds of wall clock. assert.equal(sweepDaysFromAudio(3600, 90), 90 / 86400); }); // The unit fix, pinned. A chunk is one model call, and the projection is chunks × // seconds-per-chunk — NOT audio-hours × seconds-per-audio-hour, which is what made // 27, 90 and 151 s/audio-hour look like three irreconcilable measurements of the // same box. test("the projection is priced per chunk", () => { assert.equal(sweepDays(86_400, 1), 1); // The corpus headline, as a regression pin: 191,116 chunks at the measured // production rate is ~55 days, and the 81 days the audio-hour model claimed was // wrong on the validation run's OWN data. const days = sweepDays(191_116, MEASURED_SECONDS_PER_CHUNK); assert.ok(days > 54 && days < 56, `expected ~55 sweep days, got ${days}`); // Idle-engine cost from bake-off round 2, re-priced: ~25 days, reproducing the // 24.2 it originally claimed. Same corpus, same code, different contention. const idle = sweepDays(191_116, 11.2); assert.ok(idle > 24 && idle < 26, `expected ~25 idle days, got ${idle}`); // gemma2's 5.4× — a cost decision, not a judgement call. assert.ok(sweepDays(191_116, 60.6) / idle > 5); }); // s/audio-hour is kept only as a derived convenience, and it must stay derived: a // hand-edited second number here is exactly how the unit error happened. test("the audio-hour rate is derived from the per-chunk cost", () => { assert.equal( MEASURED_SECONDS_PER_AUDIO_HOUR, MEASURED_SECONDS_PER_CHUNK * CORPUS_CHUNKS_PER_AUDIO_HOUR, ); // And the two bases must agree on the corpus, since that is the mix the // conversion factor was measured on. const viaChunks = sweepDays(191_116, MEASURED_SECONDS_PER_CHUNK); const viaAudio = sweepDaysFromAudio( 77_298 * 3600, MEASURED_SECONDS_PER_AUDIO_HOUR, ); assert.ok( Math.abs(viaChunks - viaAudio) / viaChunks < 0.02, `bases disagree: ${viaChunks} vs ${viaAudio}`, ); }); // The census band table, as arithmetic rather than prose: chunk density spans 4×, // which is the entire reason a single s/audio-hour figure cannot be trusted. test("chunk density varies fourfold across the corpus", () => { const bands = [ { label: "<15m", chunks: 41_960, audioSeconds: 6_203 * 3600 }, { label: ">8h", chunks: 24_888, audioSeconds: 14_850 * 3600 }, ]; const short = chunksPerAudioHour(bands[0].chunks, bands[0].audioSeconds); const long = chunksPerAudioHour(bands[1].chunks, bands[1].audioSeconds); assert.ok(short > 6.5 && short < 7.1, `short band density ${short}`); assert.ok(long > 1.5 && long < 1.9, `long band density ${long}`); assert.ok(short / long > 3.5, `expected ~4x spread, got ${short / long}`); // The result that inverts the intuition: the long band is 2.4× the audio of the // short band but LESS work. assert.ok(bands[1].audioSeconds > bands[0].audioSeconds * 2); assert.ok(bands[1].chunks < bands[0].chunks); }); // The plan's chunk count and the chunker's actual slicing must be the same // function. Pinned against the REAL chunker, because a drift here silently // mis-prices GPU-weeks. test("the chunk-count helper matches the real chunker", () => { const maxCues = maxCuesForContext(8192); assert.equal(maxCues, 600); const opts = { maxCues, overlapCues: DIGEST_OVERLAP_CUES }; // The plan's stated formula: step is 560 at the shipped config. assert.equal(countCueChunks(0, opts), 0); assert.equal(countCueChunks(600, opts), 1); assert.equal(countCueChunks(601, opts), 2); assert.equal(countCueChunks(1160, opts), 2); // 600 + 560 exactly assert.equal(countCueChunks(1161, opts), 3); // …and it agrees with slicing real cues, at the boundaries and either side. for (const n of [0, 1, 599, 600, 601, 1159, 1160, 1161, 5000, 12_345]) { const cues = Array.from({ length: n }, (_, i) => ({ start: i, end: i + 1, text: "x", })); assert.equal( countCueChunks(n, opts), chunkCuesForContext(cues, opts).length, `chunk count disagrees with the chunker at ${n} cues`, ); } }); test("a video with no recorded cueCount is estimated, and says so", () => { const real = chunksForVideo({ cueCount: 601, duration: 3600 }, 600); assert.deepEqual(real, { chunks: 2, estimated: false }); // Falls back to the census density rather than dropping the video from the // bill — understating the sweep is the worse failure. const guessed = chunksForVideo({ cueCount: null, duration: 4 * 3600 }, 600); assert.equal(guessed.estimated, true); assert.equal(guessed.chunks, Math.round(4 * CORPUS_CHUNKS_PER_AUDIO_HOUR)); // Never zero: a transcribed video always costs at least one call. assert.equal(chunksForVideo({ cueCount: null, duration: 30 }, 600).chunks, 1); }); // A saving counted in videos is the mistake this module exists to prevent, so // pin the arithmetic that makes it visible: many short mirrors are worth far // less than one long VOD. test("audio-hours, not video count, is what a saving is measured in", () => { const fourThousandShorts = 4_000 * 120; // 4,000 two-minute mirrors const oneVodChannel = 8_331 * 3600; // HasanAbiVODs3 assert.ok( audioHours(oneVodChannel) > audioHours(fourThousandShorts) * 60, "one VOD channel outweighs thousands of short mirrors", ); }); // --------------------------------------------------------------------------- // The census // // There used to be a test HERE that opened this machine's real LMDB and asserted // the corpus was exactly 191,116 chunks across 73,367 videos. It was skipped // everywhere else and permanently RED here, because the number moves whenever a // video is downloaded — 191,116 had already become 194,053. That is the corpus // growing, not a regression, and a suite that reports a red for it teaches people // to ignore reds. // // An absolute count of a live corpus is a MEASUREMENT. It now lives in // `bin/digest-plan.ts --census`, which prints the band table and warns when the // density constant stops describing the corpus. // // Its comment claimed two jobs. Both are kept below, hermetically: // 1. catch the plan and the chunker drifting apart — a pure property, and one // the test directly above already covers for countCueChunks; // 2. catch a DEFAULT_DIGEST_* change that alters the shipped chunk size // without re-pricing the sweep — pinned on the DENSITY constant, which is // what the cost model actually consumes, rather than on a video count that // moves on its own. // --------------------------------------------------------------------------- // One synthetic row per band, with cue counts chosen so the expected chunk count // is computable by hand. function stat(duration: number, cueCount: number | null) { return { duration, cueCount, hasTranscript: true }; } test("the census buckets by duration and totals chunks per band", () => { const maxCues = maxCuesForContext(8192); const census = buildChunkCensus( [ stat(10 * 60, 600), // < 15 min → 1 chunk stat(30 * 60, 601), // 15–60 min → 2 stat(90 * 60, 1_161), // 1–2 h → 3 stat(3 * 3600, 600), // 2–4 h → 1 stat(6 * 3600, 600), // 4–8 h → 1 stat(9 * 3600, 600), // > 8 h → 1 ], maxCues, ); assert.deepEqual( census.bands.map((b) => [b.label, b.videos, b.chunks]), [ ["< 15 min", 1, 1], ["15–60 min", 1, 2], ["1–2 h", 1, 3], ["2–4 h", 1, 1], ["4–8 h", 1, 1], ["> 8 h", 1, 1], ], ); assert.equal(census.videos, 6); assert.equal(census.chunks, 9); assert.equal(census.estimated, 0); // Band EDGES land in the lower band, so a boundary video cannot be counted // twice or dropped — 15:00 exactly is "15–60 min", not "< 15 min". const edges = buildChunkCensus( [stat(15 * 60, 600), stat(8 * 3600, 600)], maxCues, ); assert.equal(edges.bands[0].videos, 0); assert.equal(edges.bands[1].videos, 1); assert.equal(edges.bands[4].videos, 0); assert.equal(edges.bands[5].videos, 1); }); test("the census counts only transcribed videos, and flags estimated rows", () => { const maxCues = maxCuesForContext(8192); const census = buildChunkCensus( [ stat(3600, 600), { duration: 3600, cueCount: 600, hasTranscript: false }, // A zero-duration row cannot be banded and is not work. stat(0, 600), // No cueCount: estimated from duration rather than dropped, because // understating the sweep is the worse failure — and COUNTED, because the // census's claim to costing nothing is that cueCount is on every row. stat(4 * 3600, null), ], maxCues, ); assert.equal(census.videos, 2); assert.equal(census.estimated, 1); assert.equal( census.chunks, 1 + Math.round(4 * CORPUS_CHUNKS_PER_AUDIO_HOUR), ); }); // The plan and the chunker, one function — asserted through the census this time, // so the path the report actually takes is the path under test. test("the census prices a video exactly as the real chunker slices it", () => { const maxCues = maxCuesForContext(8192); const opts = { maxCues, overlapCues: DIGEST_OVERLAP_CUES }; for (const n of [1, 599, 600, 601, 1_160, 1_161, 5_000]) { const cues = Array.from({ length: n }, (_, i) => ({ start: i, end: i + 1, text: "x", })); assert.equal( buildChunkCensus([stat(3600, n)], maxCues).chunks, chunkCuesForContext(cues, opts).length, `census disagrees with the chunker at ${n} cues`, ); } }); // THE TRIPWIRE THE LIVE TEST WAS REALLY FOR, and the whole reason it can be // hermetic: the cost model consumes the DENSITY, not a video count. // // CORPUS_CHUNKS_PER_AUDIO_HOUR was measured at the shipped chunk size. Change // DEFAULT_DIGEST_NUM_CTX or DEFAULT_DIGEST_MAX_CUES_PER_CHUNK and the real // density moves while the constant does not — so every sweep projection is // silently priced on a mix that no longer exists. Pinning the shipped size // against the DEFAULTS (not against a hardcoded 8192, which is what let this // through before) makes that change fail here and say what to do about it. test("changing a DEFAULT_DIGEST_* forces the sweep to be re-priced", () => { assert.equal( maxCuesForContext(DEFAULT_DIGEST_NUM_CTX), 600, "the shipped chunk size changed — re-run `digest-plan --census` and update " + "CORPUS_CHUNKS_PER_AUDIO_HOUR (and the band table beside it) before " + "trusting any sweep projection", ); assert.equal(DEFAULT_DIGEST_MAX_CUES_PER_CHUNK, 600); // And the recorded census must still reproduce the constant it was derived // from. These are the totals from the run frozen in digestPlan.ts's comment — // constants here, not a live read, so this asserts the ARITHMETIC rather than // the operator's current corpus. const RECORDED = { chunks: 191_116, audioHours: 77_298 }; assert.ok( Math.abs( chunksPerAudioHour(RECORDED.chunks, RECORDED.audioHours * 3600) - CORPUS_CHUNKS_PER_AUDIO_HOUR, ) < 0.01, "CORPUS_CHUNKS_PER_AUDIO_HOUR no longer matches the census it came from", ); });