#!/usr/bin/env tsx // Price the digest backfill, in CHUNKS, before spending GPU-weeks on it. // // Reads the last duplicate-detection run and build:stats' cache; writes nothing. // Run it before and after a cheapness lever (a --near threshold change, a batch // of confirmed clusters) and diff `generate`: that number, not a video count, is // what the sweep's wall clock is made of. // // A chunk is one model call and IS the unit of cost. Audio-hours are reported // alongside because they are what a human thinks in, but chunk density varies 4× // across the corpus, so a projection built on s/audio-hour is only valid for a // slice with the corpus-average mix. Quoting one is what made three separate // measurements look contradictory when per chunk they agreed to within 2%. // // Flags: // --lane local|remote which engine identity to test freshness against // --per-chunk N seconds per model call (default: the measured // production rate, MEASURED_SECONDS_PER_CHUNK) // --rate N LEGACY: seconds per audio-hour. Prices off audio-hours // instead of chunks, so it can reproduce an old estimate; // it is corpus-average only and wrong per channel. // --no-freshness skip the per-video sidecar read; reports the // from-scratch cost instead of the remaining one // --channels a,b restrict to these channel slugs // --top N how many channels to list (default 15, 0 = all) // --json machine-readable, for diffing two runs // --census print the CORPUS CHUNK CENSUS by duration band and stop. // This is the measurement the density constant // CORPUS_CHUNKS_PER_AUDIO_HOUR is derived from, and the // table frozen in digestPlan.ts's comment is one run of // it. Re-run it after changing any DEFAULT_DIGEST_*, and // after any large ingest, to re-price the sweep. import { getPaths } from "../lib/paths"; import { buildDigestSweepPlan, audioHours, chunksPerAudioHour, readChunkCensus, sweepDays, sweepDaysFromAudio, CORPUS_CHUNKS_PER_AUDIO_HOUR, DIGEST_PLAN_ROLES, MEASURED_SECONDS_PER_AUDIO_HOUR, MEASURED_SECONDS_PER_CHUNK, roleMustGenerate, } from "../controller/digestPlan"; import { maxCuesForContext } from "../lib/digestPrompt"; import { DEFAULT_DIGEST_NUM_CTX } from "../lib/digest"; import { parseFlags } from "./_parseFlags"; const flags = parseFlags(process.argv.slice(2)); const lane = flags.lane === "remote" ? "remote" : "local"; // --rate is honoured for reproducing an old audio-hour estimate; --per-chunk (or // nothing) uses the real unit. const legacyRate = flags.rate !== undefined ? Number(flags.rate) : null; const perChunk = flags["per-chunk"] !== undefined ? Number(flags["per-chunk"]) : MEASURED_SECONDS_PER_CHUNK; const top = flags.top !== undefined ? Number(flags.top) : 15; const asJson = flags.json === "true"; // One projection function for every line below, so the header's stated basis and // every number under it cannot disagree. function days(chunks: number, audioSeconds: number): number { return legacyRate !== null ? sweepDaysFromAudio(audioSeconds, legacyRate) : sweepDays(chunks, perChunk); } function hours(seconds: number): string { return audioHours(seconds).toLocaleString("en-US", { maximumFractionDigits: 0, }); } function count(n: number): string { return n.toLocaleString("en-US", { maximumFractionDigits: 0 }); } function pct(part: number, whole: number): string { return whole > 0 ? `${((part / whole) * 100).toFixed(1)}%` : "—"; } // The census: the corpus totals the cost model is derived from, printed rather // than asserted. // // It used to be a unit test that opened this machine's LMDB and asserted the // corpus was exactly 191,116 chunks across 73,367 videos. That could only pass // on one machine, and it went red there the moment a video was downloaded — the // corpus growing is not a regression. An absolute count of a live corpus is a // MEASUREMENT, and this is where a measurement goes. The properties that test // was really guarding (plan-vs-chunker drift, and a DEFAULT_DIGEST_* change // silently re-pricing the sweep) are pinned hermetically in // controller/digestPlan.test.ts instead. async function census(): Promise { const maxCues = maxCuesForContext(DEFAULT_DIGEST_NUM_CTX); const result = await readChunkCensus(getPaths(), maxCues); if (!result) { console.log(""); console.log("No stats cache on this machine — run build:stats first."); console.log(""); return; } if (asJson) { console.log(JSON.stringify({ ...result, maxCues }, null, 2)); return; } console.log(""); console.log( `Corpus chunk census — ${maxCues} cues/chunk at the shipped defaults.`, ); if (result.statsSchemaStale) console.log("!! stats cache is stale — run build:stats."); console.log(""); console.log( ` ${"band".padEnd(11)} ${"videos".padStart(7)} ${"audio-h".padStart(8)} ` + `${"% audio".padStart(8)} ${"chunks".padStart(9)} ${"chunks/audio-h".padStart(15)} ${"% chunks".padStart(9)}`, ); for (const band of result.bands) { console.log( ` ${band.label.padEnd(11)} ${count(band.videos).padStart(7)} ${hours(band.audioSeconds).padStart(8)} ` + `${pct(band.audioSeconds, result.audioSeconds).padStart(8)} ${count(band.chunks).padStart(9)} ` + `${chunksPerAudioHour(band.chunks, band.audioSeconds).toFixed(2).padStart(15)} ` + `${pct(band.chunks, result.chunks).padStart(9)}`, ); } console.log( ` ${"total".padEnd(11)} ${count(result.videos).padStart(7)} ${hours(result.audioSeconds).padStart(8)} ` + `${"".padStart(8)} ${count(result.chunks).padStart(9)} ` + `${result.chunksPerAudioHour.toFixed(2).padStart(15)}`, ); console.log(""); if (result.estimated > 0) { // The census's claim to costing nothing is that cueCount is already on every // row. A non-zero here means part of the total above is a guess from // duration, and it must not pass unremarked. console.log( `!! ${count(result.estimated)} row(s) had no cueCount; their chunks were ESTIMATED from duration.`, ); } console.log( `Projected sweep: ${sweepDays(result.chunks, perChunk).toFixed(1)} days at ${perChunk}s per chunk.`, ); // THE TRIPWIRE, and the reason to run this after touching a DEFAULT_DIGEST_*: // the density constant is what the cost model consumes, and nothing else // notices when the shipped chunk size stops matching the corpus it was // measured on. const drift = Math.abs(result.chunksPerAudioHour - CORPUS_CHUNKS_PER_AUDIO_HOUR); console.log( `Density ${result.chunksPerAudioHour.toFixed(3)} vs CORPUS_CHUNKS_PER_AUDIO_HOUR ` + `${CORPUS_CHUNKS_PER_AUDIO_HOUR} (drift ${drift.toFixed(3)}).`, ); if (drift >= 0.05) { console.log( `!! The constant no longer describes this corpus. Update ` + `CORPUS_CHUNKS_PER_AUDIO_HOUR (and the band table beside it) in ` + `controller/digestPlan.ts, or the sweep projection is priced on a mix ` + `that no longer exists.`, ); } console.log(""); } async function main(): Promise { if (flags.census === "true") { await census(); return; } const plan = await buildDigestSweepPlan({ paths: getPaths(), lane, checkFreshness: flags["no-freshness"] !== "true", channelSlugs: flags.channels ? flags.channels.split(",") : undefined, onLog: (m) => { if (!asJson) console.log(m); }, }); if (asJson) { console.log( JSON.stringify( { ...plan, secondsPerChunk: perChunk, secondsPerAudioHour: legacyRate ?? MEASURED_SECONDS_PER_AUDIO_HOUR, pricedOn: legacyRate !== null ? "audio-hours" : "chunks", generateAudioHours: audioHours(plan.generateSeconds), sharedAudioHours: audioHours(plan.sharedSeconds), generateChunksPerAudioHour: chunksPerAudioHour( plan.generateChunks, plan.generateSeconds, ), sweepDays: days(plan.generateChunks, plan.generateSeconds), }, null, 2, ), ); return; } const eligible = plan.generateSeconds + plan.sharedSeconds + plan.fresh.audioSeconds; console.log(""); console.log( `Digest sweep plan — ${lane} lane, ` + (legacyRate !== null ? `${legacyRate}s per audio-hour (LEGACY basis)` : `${perChunk}s per chunk at ${plan.maxCues} cues/chunk`) + (plan.freshnessChecked ? "" : ", FROM SCRATCH (freshness not checked)"), ); console.log( `Duplicate report: ${plan.clusters.toLocaleString()} cluster(s), ${plan.clusterMembersMapped.toLocaleString()} member(s) mapped.`, ); if (plan.statsSchemaStale) console.log("!! stats cache is stale — run build:stats."); console.log(""); console.log("Remaining work by role (chunks and audio-hours):"); for (const role of DIGEST_PLAN_ROLES) { const t = plan.remaining[role]; console.log( ` ${role.padEnd(17)} ${count(t.chunks).padStart(9)} chunks ` + `${hours(t.audioSeconds).padStart(8)} h ` + `${t.videos.toLocaleString().padStart(7)} videos ` + `${roleMustGenerate(role) ? "GENERATE" : "shared free"}`, ); } console.log( ` ${"already fresh".padEnd(17)} ${count(plan.fresh.chunks).padStart(9)} chunks ` + `${hours(plan.fresh.audioSeconds).padStart(8)} h ` + `${plan.fresh.videos.toLocaleString().padStart(7)} videos skipped`, ); console.log( ` ${"ineligible".padEnd(17)} ${"—".padStart(9)} ${"—".padStart(8)} ` + `${plan.ineligible.toLocaleString().padStart(7)} videos no transcript`, ); if (plan.chunksEstimated > 0) console.log( ` !! ${plan.chunksEstimated.toLocaleString()} video(s) had no cueCount; their chunks were ESTIMATED from duration.`, ); console.log(""); console.log( `TO GENERATE : ${count(plan.generateChunks)} chunks / ${hours(plan.generateSeconds)} audio-hours ` + `(${chunksPerAudioHour(plan.generateChunks, plan.generateSeconds).toFixed(2)} chunks/audio-h) ` + `→ ${days(plan.generateChunks, plan.generateSeconds).toFixed(1)} sweep days`, ); console.log( `SAVED by sharing: ${count(plan.sharedChunks)} chunks / ${hours(plan.sharedSeconds)} audio-hours ` + `(${pct(plan.sharedSeconds, eligible)} of eligible) → ` + `${days(plan.sharedChunks, plan.sharedSeconds).toFixed(1)} days avoided`, ); console.log(""); const listed = top > 0 ? plan.channels.slice(0, top) : plan.channels; // Chunk density per channel is the column that explains why the sweep's rate // changes as the queue drains — it runs heaviest-first, and the heaviest // channels are the CHEAPEST per audio-hour. console.log(`Heaviest channels (the sweep's work queue order):`); for (const c of listed) { if (c.generateChunks <= 0) continue; console.log( ` ${c.channelSlug.padEnd(28)} ${count(c.generateChunks).padStart(8)} chunks ` + `${hours(c.generateSeconds).padStart(7)} h ` + `${chunksPerAudioHour(c.generateChunks, c.generateSeconds).toFixed(2).padStart(5)} c/h ` + `${days(c.generateChunks, c.generateSeconds).toFixed(1).padStart(5)} d`, ); } const rest = plan.channels.length - listed.length; if (rest > 0) console.log(` … and ${rest} more channel(s).`); console.log(""); } main().catch((err) => { console.error(err); process.exit(1); });