#!/usr/bin/env tsx // Score digest engine candidates against a FIXED sample, without writing a // single digest to disk. // // WHY THIS IS A SCRIPT AND NOT INFRASTRUCTURE. It answers one question — which // (model x context x timestamp mode) should carry a multi-week sweep — and the // answer is a number in a table, not a feature. It therefore drives digestApps + // digestPrompt + digestParse DIRECTLY and scores in memory. It NEVER calls // digestVideo or writeDigestSection, so a losing candidate cannot leave anything // behind in the corpus, and no freshness record has to be invalidated afterwards // to undo a round. // // THROUGHPUT IS A FIRST-CLASS METRIC, not a footnote. Measured: 46 s per // 8,194-token chunk on qwen2.5:7b, which over the corpus is ~64 days on one // lane. A 14B at half the context roughly doubles the chunk count and halves the // token rate — order 250 days. A candidate can therefore be ruled out on // projected sweep days alone, however good its chapters look, which is why every // run prints days alongside quality. // // BOUNDARY ACCURACY, AND WHY IT WAS ADDED. Every metric this harness scored // originally — zeroYieldRate, chaptersPerHour, rejectionRate, maxGapSeconds, // genericTitleRate, duplicateTitleRate — is a DEFECT COUNTER: it says how // malformed the output is, never whether a boundary landed in the right place. // So the harness could rank candidates by which was least broken and still not // say which segmented better. lib/boundaryScore.ts closes that with an oracle // the model never sees: the UPLOADER's own chapter marks from // metadata.info.json, present on ~12,139 videos that also have a transcript and // >=4 chapters. Boundaries are facts, so scoring against them reproduces // nothing; the uploader's chapter TITLES are expression and are never scored // against, only used to spot boilerplate ("Intro", "Sponsor"). // // Modes: // // --pick scan the stats cache and write the fixed stratified sample // (plans/bakeoff/sample.json). Run ONCE. A moving sample // makes the comparison between rounds meaningless. // --pick-chapters write a sample restricted to videos that carry >=4 // uploader chapters, so boundary accuracy is scorable. Kept // as a SEPARATE file (speaker-sample.json / // chapter-sample.json) so rounds 1-2 stay comparable. // --score-existing score the ai-digest.json files ALREADY on disk against // their uploader chapters. Zero GPU, zero writes — the way // to prove the scorer works before spending a generation // run on it. // (default) run the candidates over that sample and write a JSON + // Markdown report under plans/bakeoff/. // // Examples: // tsx bin/digest-bakeoff.ts --pick // tsx bin/digest-bakeoff.ts --pick-chapters --sample ../plans/bakeoff/chapter-sample.json // tsx bin/digest-bakeoff.ts --score-existing // tsx bin/digest-bakeoff.ts --label round1 --buckets short,medium \ // --candidates 'qwen2.5:7b@16384,qwen3:8b@16384,gemma2:9b@16384' // tsx bin/digest-bakeoff.ts --label round2 --buckets long,verylong \ // --candidates 'qwen2.5:7b@16384' --modes absolute,chunk-local // tsx bin/digest-bakeoff.ts --label speakers-round1 --speakers off,on \ // --sample ../plans/bakeoff/speaker-sample.json --candidates 'qwen2.5:7b@8192' import path from "node:path"; import { mkdir, readdir, writeFile } from "node:fs/promises"; import { readFile } from "node:fs/promises"; import { open } from "../../common/bin/_lmdb"; import { getPaths } from "../../common/lib/paths"; import { parseFlags } from "../../common/bin/_parseFlags"; import type { VideoStat } from "../../common/lib/stats"; import { getDigestApp } from "../../common/lib/digestApps"; import type { DigestAppConfig, DigestTimestampMode } from "../../common/lib/digest"; import { CHAPTER_SYSTEM_PROMPT, DIGEST_MINUTES_PER_CHAPTER, DIGEST_OVERLAP_CUES, buildChapterPrompt, chapterSchema, maxCuesForContext, toHms, } from "../../common/lib/digestPrompt"; import { parseChapters, type DigestChunkOutput } from "../../common/lib/digestParse"; import { chunkCuesForContext } from "../../common/lib/transcriptWindow"; import { transcriptToMarkdown } from "../../common/lib/transcriptToMarkdown"; import { readNormalizedTranscript } from "../../common/controller/normalizeTranscript"; import { CUES_JSON_FILENAME, isRealAudioFile, readUploaderChapters, type UploaderChapter, } from "../../common/lib/videoStatus"; import { BOUNDARY_TOLERANCES_SECONDS, isBoilerplateChapterTitle, scoreBoundaries, type BoundaryReport, } from "../../common/lib/boundaryScore"; import { loadAttribution } from "../../common/lib/attribution-server"; import type { AttributionRecord } from "../../common/lib/attribution"; import { ATTRIBUTION_MAX_SPEAKERS, ATTRIBUTION_MIN_CLUSTER_SHARE, } from "../../common/lib/attributionPrompt"; import { loadDigest } from "../../common/lib/digest-server"; import type { Cue } from "../../common/lib/vtt"; // --------------------------------------------------------------------------- // The sample // --------------------------------------------------------------------------- // Duration strata. Chosen to match the corpus shape recorded in FACTS.md rather // than to be round numbers: the >4 h bucket is 8.2% of videos but 46% of all // transcript tokens, so a sample that under-represents it measures the cheap // half of the sweep and misses the half where chunk-seam bugs live. const BUCKETS = ["short", "medium", "long", "verylong"] as const; type Bucket = (typeof BUCKETS)[number]; const BUCKET_BOUNDS: Record = { short: { min: 5 * 60, max: 30 * 60, want: 3 }, medium: { min: 45 * 60, max: 90 * 60, want: 2 }, long: { min: 3 * 3600, max: 4.5 * 3600, want: 2 }, verylong: { min: 6 * 3600, max: 14 * 3600, want: 1 }, }; type SampleVideo = { slug: string; channelSlug: string; videoId: string; videoDir: string; title: string; bucket: Bucket; durationSeconds: number; cueCount: number; // Non-boilerplate uploader marks counted at pick time. Recorded for the // record only — the scorer re-reads them from disk, because metadata can be // refetched and a frozen copy would let the sample and the corpus disagree. uploaderChapters?: number; }; type Sample = { version: 1; pickedAt: string; // Corpus-wide totals, captured at pick time. The sweep-days projection is // computed from measured seconds-per-audio-hour times THIS number, so the // projection and the sample come from one scan and can't drift apart. corpus: { videosScanned: number; videosWithTranscript: number; audioHours: number; longTailVideos: number; longTailAudioHours: number; }; videos: SampleVideo[]; }; function bucketFor(seconds: number): Bucket | null { for (const b of BUCKETS) { const { min, max } = BUCKET_BOUNDS[b]; if (seconds >= min && seconds <= max) return b; } return null; } // Deterministic pick, so re-running --pick on an unchanged corpus reproduces the // same sample. No Math.random: a sample that moves between rounds is not a // sample, it is noise. Videos are ordered by a stable hash of the slug and the // first N per bucket are taken, spreading the pick across channels instead of // clustering on whichever channel sorts first. function stableHash(s: string): number { let h = 2166136261; for (let i = 0; i < s.length; i++) { h ^= s.charCodeAt(i); h = Math.imul(h, 16777619); } return h >>> 0; } async function pickSample(outPath: string): Promise { const paths = getPaths(); const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true }); const statsByPath = root.openDB< { metaMs: number; stat: VideoStat }, [string, string] >({ name: "statsByPath", encoding: "msgpack" }); const byBucket = new Map(); for (const b of BUCKETS) byBucket.set(b, []); let videosScanned = 0; let videosWithTranscript = 0; let totalSeconds = 0; let longTailVideos = 0; let longTailSeconds = 0; for (const { key, value } of statsByPath.getRange()) { const stat = value.stat; videosScanned++; if (!stat.hasTranscript || !(stat.duration > 0)) continue; videosWithTranscript++; totalSeconds += stat.duration; if (stat.duration > 4 * 3600) { longTailVideos++; longTailSeconds += stat.duration; } const bucket = bucketFor(stat.duration); if (!bucket) continue; // A digest needs cues; a transcript flagged present but empty is useless // here and would silently shrink a stratum. if (!stat.cueCount || stat.cueCount < 30) continue; byBucket.get(bucket)!.push({ slug: stat.slug, channelSlug: stat.channelSlug, videoId: stat.id, videoDir: (key as [string, string])[1], title: stat.title, bucket, durationSeconds: Math.round(stat.duration), cueCount: stat.cueCount, }); } await root.close(); const videos: SampleVideo[] = []; for (const b of BUCKETS) { const pool = byBucket.get(b)!; pool.sort((a, c) => stableHash(a.slug) - stableHash(c.slug)); // One per channel first, so a stratum can't come entirely from one // creator's house style — a model that happens to suit one show would // otherwise look like a model that suits the corpus. const seenChannels = new Set(); const spread: SampleVideo[] = []; for (const v of pool) { if (seenChannels.has(v.channelSlug)) continue; seenChannels.add(v.channelSlug); spread.push(v); } const want = BUCKET_BOUNDS[b].want; const taken = (spread.length >= want ? spread : pool).slice(0, want); if (taken.length < want) { console.warn( `Warning: bucket ${b} wanted ${want} videos but only ${taken.length} qualify.`, ); } videos.push(...taken); } const sample: Sample = { version: 1, pickedAt: new Date().toISOString(), corpus: { videosScanned, videosWithTranscript, audioHours: Math.round(totalSeconds / 3600), longTailVideos, longTailAudioHours: Math.round(longTailSeconds / 3600), }, videos, }; await mkdir(path.dirname(outPath), { recursive: true }); await writeFile(outPath, `${JSON.stringify(sample, null, 2)}\n`); console.log( `Scanned ${videosScanned} videos (${videosWithTranscript} with transcripts, ` + `${sample.corpus.audioHours} audio-hours; ${longTailVideos} over 4 h holding ` + `${sample.corpus.longTailAudioHours} h).`, ); for (const v of videos) { console.log( ` ${v.bucket.padEnd(8)} ${toHms(v.durationSeconds)} ${v.cueCount .toString() .padStart(5)} cues ${v.slug} ${v.title.slice(0, 60)}`, ); } console.log(`Wrote ${outPath}`); } // --------------------------------------------------------------------------- // Scoring // --------------------------------------------------------------------------- // Titles that carry no information about what was actually said. A cheap proxy // for title quality: a model that segments correctly but names every section // "Discussion" has produced a table of contents nobody can navigate. const GENERIC_TITLE_RE = /^(the\s+)?(intro(duction)?|outro|conclusion|discussion|continued|continuation|overview|summary|recap|closing( remarks)?|opening( remarks)?|final thoughts|misc(ellaneous)?|other|general|topics?|segment|section|chapter|part)\b/i; const GENERIC_TITLE_TAIL_RE = /\b(part|section|segment|chapter)\s+(\d+|one|two|three|four|five|six|seven|eight|nine|ten)$/i; function isGenericTitle(title: string): boolean { const t = title.trim(); return GENERIC_TITLE_RE.test(t) || GENERIC_TITLE_TAIL_RE.test(t); } function normalizeTitle(title: string): string { return title.toLowerCase().replace(/[^\p{L}\p{N}]+/gu, " ").trim(); } type Candidate = { key: string; model: string; // Absent means "the engine's default" — the same thing an unset // settings.digest.apps[id].numCtx means, so the default row measures exactly // what a default-configured sweep would do. numCtx?: number; maxCues: number; timestampMode: DigestTimestampMode; think?: boolean; // Render speaker labels into the transcript from attribution.json. // // FALSE MUST BE BYTE-IDENTICAL TO PRODUCTION. The speakers-off arm is not a // control unless it renders exactly what the sweep renders, which is why the // prefix hook is a no-op that returns null rather than a second renderer. speakers: boolean; }; type VideoScore = { slug: string; bucket: Bucket; durationSeconds: number; chunks: number; chunksFailed: number; zeroYieldChunks: number; kept: number; maxGapSeconds: number; engineSeconds: number; inputTokens: number; outputTokens: number; warningsByCode: Record; genericTitles: number; duplicateTitles: number; // Kept so a table can be sanity-checked against real output by hand, which is // the only way to catch a model that scores well and reads badly. sampleTitles: string[]; // Boundary accuracy against the uploader's own chapter marks. Null when the // video carries none, which is the normal case for ~85% of the corpus. boundary: BoundaryReport | null; // How many named speakers the render actually carried, and how many kept // titles mention one of them. // // NAME UPTAKE IS THE SPEAKER HYPOTHESIS' OWN METRIC. speakers-off cannot // produce it by construction (the roster is never in the prompt), so a // non-zero value on the off arm means a name leaked in some other way — a // useful tripwire, not a score. speakersRendered: number; titlesWithSpeakerName: number; }; type CandidateScore = { candidate: Candidate; videos: VideoScore[]; totals: { videos: number; audioHours: number; chunks: number; chunksFailed: number; zeroYieldChunks: number; zeroYieldRate: number; kept: number; chaptersPerHour: number; maxGapSeconds: number; meanGapSeconds: number; genericTitleRate: number; duplicateTitleRate: number; warningsByCode: Record; rejectionRate: number; engineSeconds: number; tokensPerSecond: number; secondsPerAudioHour: number; projectedSweepDays: number; // Boundary accuracy, pooled over the videos that had an oracle. // // POOLED, NOT AVERAGED PER VIDEO. A per-video mean lets a 4-chapter video // weigh as much as a 90-chapter one, so a candidate could win by doing well // on the shortest lists. Pooling counts matched/generated/reference across // the whole sample and derives precision and recall from the totals. boundary: { videosScored: number; referenceBoundaries: number; medianOffsetSeconds: number | null; withinThirtySecondsRate: number; byTolerance: { toleranceSeconds: number; matched: number; precision: number; recall: number; f1: number; }[]; }; speakerNameTitleRate: number; }; }; // --------------------------------------------------------------------------- // Speaker rendering (the speakers-on arm) // --------------------------------------------------------------------------- // A label per cue, or null where no NAMED speaker covers it. // // CONSUMES attribution.json, NEVER diarization.json. A bare cluster index // rendered as "Speaker 7:" is prompt cost with no semantic content, and it // invites chapter titles like "Speaker 7 responds". lib/diarization.ts already // argues a cluster index is not an identity; this honours that. type SpeakerContext = { // Indexed by position in the FULL cue array, so a chunk can slice it. labelByCueIndex: (string | null)[]; roster: string[]; // Cues that fell inside a named turn. Reported so a report can say how much // of the transcript the labels actually reached. labelledCues: number; }; // Speakers worth rendering: the heaviest by attributed speech, capped and // floored exactly as attributionPrompt caps the naming call itself. // // The cap is doing real work here. Diarization over-splits (median 35 clusters // per video on this corpus, max 325), and 8 of the 9 attribution records that // existed when this was written were the REJECTED text-only pilot, one of them // carrying 346 "speakers" that were raw transcript fragments. Rendering that // unfiltered would bury the transcript in noise and measure the noise. function namedSpeakers(record: AttributionRecord): Map { const seconds = new Map(); for (const seg of record.segments) { const d = Math.max(0, seg.end - seg.start); seconds.set(seg.speaker, (seconds.get(seg.speaker) ?? 0) + d); } const total = Array.from(seconds.values()).reduce((a, b) => a + b, 0); if (total <= 0) return new Map(); const ranked = record.speakers .map((s) => ({ index: s.index, label: s.label?.trim() ?? "", share: (seconds.get(s.index) ?? 0) / total })) .filter((s) => s.label.length > 0 && s.share >= ATTRIBUTION_MIN_CLUSTER_SHARE) .sort((a, b) => b.share - a.share) .slice(0, ATTRIBUTION_MAX_SPEAKERS); return new Map(ranked.map((s) => [s.index, s.label])); } // Map cues onto turns by MAXIMUM OVERLAP, not by containment. // // Cue boundaries do not align to speaker turns — measured on the diarized set, // only 64.4% of cues fall wholly inside a named turn. Containment would leave a // third of the transcript unlabelled for a reason that has nothing to do with // speaker identity; max-overlap assigns each cue to whoever does most of the // talking during it, and still yields null when no named turn touches it. function buildSpeakerContext( cues: Cue[], record: AttributionRecord, ): SpeakerContext { const named = namedSpeakers(record); const segments = record.segments .filter((s) => named.has(s.speaker)) .sort((a, b) => a.start - b.start); const labelByCueIndex: (string | null)[] = new Array(cues.length).fill(null); const rosterSeen = new Set(); const roster: string[] = []; let labelledCues = 0; let cursor = 0; for (let i = 0; i < cues.length; i++) { const cue = cues[i]; const cueEnd = cue.end > cue.start ? cue.end : cue.start + 1; // Segments are sorted, so the scan only ever moves forward. while (cursor < segments.length && segments[cursor].end <= cue.start) cursor++; let best: { label: string; overlap: number } | null = null; for (let j = cursor; j < segments.length; j++) { const seg = segments[j]; if (seg.start >= cueEnd) break; const overlap = Math.min(cueEnd, seg.end) - Math.max(cue.start, seg.start); if (overlap > 0 && (!best || overlap > best.overlap)) { best = { label: named.get(seg.speaker)!, overlap }; } } if (best) { labelByCueIndex[i] = best.label; labelledCues++; if (!rosterSeen.has(best.label)) { rosterSeen.add(best.label); roster.push(best.label); } } } return { labelByCueIndex, roster, labelledCues }; } // Label on speaker CHANGE only, never per line. // // MEASURED COST. Per-line labels inflate a rendered chunk by ~12% of characters // at the median and up to ~30%, which on the ~10-tokens-per-cue budget that // sizes the 600-cue chunk to an 8k window is enough to start truncating — and // ollama truncates SILENTLY. Change-only lands at ~3.4% median. Since the // speaker changes on only ~26% of cues, the two carry the same information. // // A cue with no named speaker gets NO prefix and does not count as a change, so // a coverage gap reads as "the previous speaker continues" rather than as a // fake new person. function speakerPrefixer( context: SpeakerContext, chunkStartIndex: number, ): (cue: Cue, index: number) => string | null { let previous: string | null = null; return (_cue, index) => { const label = context.labelByCueIndex[chunkStartIndex + index] ?? null; if (!label) return null; if (label === previous) return null; previous = label; return `${label}: `; }; } function renderChunk( meta: { id: string; title: string; channel?: string; duration?: number }, cues: Cue[], offsetSeconds: number, prefixForCue?: (cue: Cue, index: number) => string | null, ): string { return transcriptToMarkdown( { ...meta, cues }, { timestamps: true, includeDescription: false, includeTags: false, stampForCue: (_clock, seconds) => toHms(Math.max(0, seconds - offsetSeconds)), // Omitted entirely on the speakers-off arm, so that arm's bytes are the // sweep's bytes. ...(prefixForCue ? { prefixForCue } : {}), }, ); } // The cast list for ONE chunk, not for the video. // // A 12-name roster on a chunk where two people speak is misleading and wastes // context. The preamble also has to explain the change-only convention, or the // model reads an unlabelled line as an unknown speaker. function speakerPreamble(names: string[]): string | undefined { if (names.length === 0) return undefined; return [ "Speakers in this section (a name before a line means that speaker begins", "there; unlabelled lines continue the previous speaker):", ...names.map((n) => `- ${n}`), ].join("\n"); } // Where a sample video's sidecars live. One definition, because three call // sites now need it (scoring, the oracle read, attribution). function videoDirFor(video: SampleVideo): string { const paths = getPaths(); return path.join(paths.channelsDir, video.channelSlug, "data", video.videoDir); } // The oracle, read FRESH at score time rather than frozen into the sample. // // A video's uploader chapters can change when metadata is refetched. Freezing // them would let the sample and the disk disagree silently; reading them here // means a report always scored against what the uploader currently says. // // Boilerplate marks are dropped. "Intro"/"Sponsor"/"Outro" are boundaries in the // video's FURNITURE, not in its subject, and crediting a model for finding the // sponsor read measures the wrong thing — the same class of junk the // `boilerplate` context field exists to remove. async function uploaderBoundaries( video: SampleVideo, ): Promise<{ starts: number[]; chapters: UploaderChapter[] } | null> { const chapters = await readUploaderChapters(videoDirFor(video)); if (!chapters) return null; const kept = chapters.filter((c) => !isBoilerplateChapterTitle(c.title)); if (kept.length === 0) return null; return { starts: kept.map((c) => c.start), chapters: kept }; } async function scoreVideo( video: SampleVideo, candidate: Candidate, log: (m: string) => void, ): Promise { const videoDir = videoDirFor(video); const cuesPath = path.join(videoDir, CUES_JSON_FILENAME); const transcript = await readNormalizedTranscript(cuesPath); if (!transcript || !transcript.cues?.length) { log(` ${video.slug}: no transcript on disk, skipped`); return null; } const cues = transcript.cues; const chunks = chunkCuesForContext(cues, { maxCues: candidate.maxCues, overlapCues: DIGEST_OVERLAP_CUES, }); // Speakers are loaded per VIDEO, once, even though they are rendered per // chunk: the roster has to be sliced to the chunk but the cue->turn mapping // is a whole-video computation. let speakerContext: SpeakerContext | null = null; if (candidate.speakers) { const record = await loadAttribution(videoDir); if (record) { speakerContext = buildSpeakerContext(cues, record); log( ` ${video.slug}: ${speakerContext.roster.length} named speaker(s), ` + `${pct(cues.length > 0 ? speakerContext.labelledCues / cues.length : 0)} of cues labelled`, ); } else { // NOT an error and NOT a skip. A video with no attribution renders // exactly as the off arm renders it, which is the behaviour any shipped // version would need for the ~98% of the corpus that can never be // diarized. Silently degrading is the feature. log(` ${video.slug}: no attribution on disk, rendering without speakers`); } } // Chunk i starts at this index in the full cue array. chunkCuesForContext // overlaps by DIGEST_OVERLAP_CUES, so this is not i * maxCues. const chunkStartIndices: number[] = []; { let cursor = 0; for (const chunk of chunks) { const first = chunk[0]; // Cues are unique by identity here, so indexOf from the last position is // both correct and linear overall. const found = cues.indexOf(first, Math.max(0, cursor - chunk.length)); chunkStartIndices.push(found >= 0 ? found : cursor); cursor = (found >= 0 ? found : cursor) + chunk.length; } } const app = getDigestApp("ollama-direct"); const config: DigestAppConfig = { model: candidate.model, numCtx: candidate.numCtx, temperature: 0, timeoutMs: 20 * 60_000, ...(candidate.think !== undefined ? { think: candidate.think } : {}), }; const outputs: DigestChunkOutput[] = []; const score: VideoScore = { slug: video.slug, bucket: video.bucket, durationSeconds: video.durationSeconds, chunks: chunks.length, chunksFailed: 0, zeroYieldChunks: 0, kept: 0, maxGapSeconds: 0, engineSeconds: 0, inputTokens: 0, outputTokens: 0, warningsByCode: {}, genericTitles: 0, duplicateTitles: 0, sampleTitles: [], boundary: null, speakersRendered: speakerContext?.roster.length ?? 0, titlesWithSpeakerName: 0, }; for (let i = 0; i < chunks.length; i++) { const chunk = chunks[i]; const startSeconds = Math.max(0, Math.floor(chunk[0].start)); const endSeconds = Math.max( startSeconds, Math.ceil(chunk[chunk.length - 1].end || chunk[chunk.length - 1].start), ); const offset = candidate.timestampMode === "chunk-local" ? startSeconds : 0; // The roster is per-CHUNK: the names that actually appear in THIS slice. // A 12-name cast list over a chunk where two people speak is misleading and // spends context for nothing. let prefixForCue: ((cue: Cue, index: number) => string | null) | undefined; let speakerRoster: string | undefined; if (speakerContext) { const startIndex = chunkStartIndices[i]; const namesHere: string[] = []; const seenHere = new Set(); for (let k = 0; k < chunk.length; k++) { const label = speakerContext.labelByCueIndex[startIndex + k]; if (label && !seenHere.has(label)) { seenHere.add(label); namesHere.push(label); } } if (namesHere.length > 0) { prefixForCue = speakerPrefixer(speakerContext, startIndex); speakerRoster = speakerPreamble(namesHere); } } const promptInput = { title: transcript.title || video.videoId, channel: transcript.channel || video.channelSlug, startSeconds, endSeconds, transcript: renderChunk( { id: transcript.id, title: transcript.title, channel: transcript.channel, duration: transcript.duration, }, chunk, offset, prefixForCue, ), timestampMode: candidate.timestampMode, ...(speakerRoster ? { speakerRoster } : {}), }; try { const result = await app.run({ system: CHAPTER_SYSTEM_PROMPT, prompt: buildChapterPrompt(promptInput), schema: chapterSchema(endSeconds - startSeconds), config, }); score.engineSeconds += result.durationMs / 1000; score.inputTokens += result.inputTokens ?? 0; score.outputTokens += result.outputTokens ?? 0; outputs.push({ index: i, startSeconds, endSeconds, data: result.data, timestampMode: candidate.timestampMode, }); } catch (err) { score.chunksFailed++; score.warningsByCode["chunk-failed"] = (score.warningsByCode["chunk-failed"] ?? 0) + 1; log(` ${video.slug} chunk ${i + 1}/${chunks.length} failed: ${(err as Error).message}`); } } // ZERO-YIELD CHUNKS — the headline defect metric. Computed by parsing each // chunk ALONE, because the merged parse cannot attribute a kept chapter back // to the chunk that produced it, and "chunk 3 produced nothing" is precisely // the failure this whole stage exists to fix. A chunk the engine never // answered counts too: from the corpus's point of view the outcome is the // same, an interval of the video with no chapters in it. for (const out of outputs) { if (parseChapters([out], cues).chapters.length === 0) score.zeroYieldChunks++; } score.zeroYieldChunks += score.chunksFailed; const parsed = parseChapters(outputs, cues); score.kept = parsed.chapters.length; for (const w of parsed.warnings) { score.warningsByCode[w.code] = (score.warningsByCode[w.code] ?? 0) + 1; } // MAX COVERAGE GAP — catches "summarised the tail, skipped the head". The // leading gap (0 -> first chapter) and the trailing one (last chapter -> end) // are included deliberately: a video whose chapters all sit in the last 20 // minutes has a coverage failure that consecutive-gap-only scoring hides. const starts = parsed.chapters.map((c) => c.start); const bounds = [0, ...starts, video.durationSeconds]; for (let i = 1; i < bounds.length; i++) { score.maxGapSeconds = Math.max(score.maxGapSeconds, bounds[i] - bounds[i - 1]); } const seen = new Set(); for (const c of parsed.chapters) { if (isGenericTitle(c.title)) score.genericTitles++; const n = normalizeTitle(c.title); if (seen.has(n)) score.duplicateTitles++; seen.add(n); } score.sampleTitles = parsed.chapters.slice(0, 8).map((c) => `${c.clock} ${c.title}`); // BOUNDARY ACCURACY against the uploader. The one metric here that measures // whether the segmentation is RIGHT rather than well-formed. const oracle = await uploaderBoundaries(video); if (oracle) { score.boundary = scoreBoundaries( oracle.starts, parsed.chapters.map((c) => c.start), ); } // SPEAKER NAME UPTAKE. Counted against the roster the render actually used, // so the off arm can be checked for leakage: a non-zero value there means a // name reached the titles by some route other than the labels. if (speakerContext && speakerContext.roster.length > 0) { // WORD-BOUNDARY MATCHING, not substring. normalizeTitle collapses to // space-separated words, so a substring test would count "ghost stories" as // containing the speaker "Host" — and role labels like "Host" and "Caller" // are exactly what the diarized lane produces when the transcript does not // support a real name, so the false positives would not be rare. const needles = speakerContext.roster .map((n) => normalizeTitle(n)) .filter((n) => n.length >= 3) .map((n) => ` ${n} `); for (const c of parsed.chapters) { const t = ` ${normalizeTitle(c.title)} `; if (needles.some((n) => t.includes(n))) score.titlesWithSpeakerName++; } } log( ` ${video.slug} [${video.bucket}] ${chunks.length} chunk(s) → ${score.kept} chapter(s), ` + `${score.zeroYieldChunks} zero-yield, ${Math.round(score.engineSeconds)}s engine` + (score.boundary ? `, boundary F1@30 ${pct(score.boundary.scores[0]?.f1 ?? 0)} ` + `(${score.boundary.referenceCount} uploader mark(s))` : ", no uploader chapters"), ); return score; } function aggregate(candidate: Candidate, videos: VideoScore[]): CandidateScore { const sum = (f: (v: VideoScore) => number): number => videos.reduce((a, v) => a + f(v), 0); const audioHours = sum((v) => v.durationSeconds) / 3600; const chunks = sum((v) => v.chunks); const kept = sum((v) => v.kept); const engineSeconds = sum((v) => v.engineSeconds); const tokens = sum((v) => v.inputTokens + v.outputTokens); const warningsByCode: Record = {}; for (const v of videos) { for (const [code, n] of Object.entries(v.warningsByCode)) { warningsByCode[code] = (warningsByCode[code] ?? 0) + n; } } const rejections = Object.entries(warningsByCode) .filter(([code]) => code !== "seam-duplicate") .reduce((a, [, n]) => a + n, 0); const secondsPerAudioHour = audioHours > 0 ? engineSeconds / audioHours : 0; // Pooled boundary accuracy. See the comment on CandidateScore.totals.boundary // for why this is not a mean of per-video F1s. const scored = videos.filter((v) => v.boundary); const referenceBoundaries = scored.reduce((a, v) => a + v.boundary!.referenceCount, 0); const generatedBoundaries = scored.reduce((a, v) => a + v.boundary!.generatedCount, 0); const within30 = scored.reduce((a, v) => a + v.boundary!.withinThirtySeconds, 0); const allOffsets: number[] = []; for (const v of scored) { // medianOffsetSeconds is per video; pooling the medians is not a median, so // the report quotes the median OF the per-video medians and says so. if (v.boundary!.medianOffsetSeconds !== null) { allOffsets.push(v.boundary!.medianOffsetSeconds); } } allOffsets.sort((a, b) => a - b); const byTolerance = BOUNDARY_TOLERANCES_SECONDS.map((tol) => { const matched = scored.reduce( (a, v) => a + (v.boundary!.scores.find((s) => s.toleranceSeconds === tol)?.matched ?? 0), 0, ); const precision = generatedBoundaries > 0 ? matched / generatedBoundaries : 0; const recall = referenceBoundaries > 0 ? matched / referenceBoundaries : 0; return { toleranceSeconds: tol, matched, precision: round(precision, 4), recall: round(recall, 4), f1: round(precision + recall > 0 ? (2 * precision * recall) / (precision + recall) : 0, 4), }; }); return { candidate, videos, totals: { videos: videos.length, audioHours: round(audioHours, 2), chunks, chunksFailed: sum((v) => v.chunksFailed), zeroYieldChunks: sum((v) => v.zeroYieldChunks), zeroYieldRate: chunks > 0 ? round(sum((v) => v.zeroYieldChunks) / chunks, 4) : 0, kept, chaptersPerHour: audioHours > 0 ? round(kept / audioHours, 2) : 0, maxGapSeconds: videos.reduce((a, v) => Math.max(a, v.maxGapSeconds), 0), meanGapSeconds: videos.length > 0 ? Math.round(sum((v) => v.maxGapSeconds) / videos.length) : 0, genericTitleRate: kept > 0 ? round(sum((v) => v.genericTitles) / kept, 4) : 0, duplicateTitleRate: kept > 0 ? round(sum((v) => v.duplicateTitles) / kept, 4) : 0, warningsByCode, // Rejections per kept chapter — the ratio that says how much of what the // model produced the guards had to throw away. rejectionRate: kept + rejections > 0 ? round(rejections / (kept + rejections), 4) : 0, engineSeconds: Math.round(engineSeconds), tokensPerSecond: engineSeconds > 0 ? round(tokens / engineSeconds, 1) : 0, secondsPerAudioHour: Math.round(secondsPerAudioHour), projectedSweepDays: 0, // filled in once corpus hours are known boundary: { videosScored: scored.length, referenceBoundaries, medianOffsetSeconds: allOffsets.length > 0 ? allOffsets[Math.floor(allOffsets.length / 2)] : null, withinThirtySecondsRate: referenceBoundaries > 0 ? round(within30 / referenceBoundaries, 4) : 0, byTolerance, }, speakerNameTitleRate: kept > 0 ? round(sum((v) => v.titlesWithSpeakerName) / kept, 4) : 0, }, }; } function round(n: number, places: number): number { const f = 10 ** places; return Math.round(n * f) / f; } // --------------------------------------------------------------------------- // Report // --------------------------------------------------------------------------- function markdownReport( label: string, sample: Sample, buckets: Bucket[], scores: CandidateScore[], ): string { const lines: string[] = []; lines.push(`# Digest bake-off — ${label}`); lines.push(""); lines.push( `Sample: ${scores[0]?.totals.videos ?? 0} video(s) from \`plans/bakeoff/sample.json\`` + ` (buckets: ${buckets.join(", ")}), ${scores[0]?.totals.audioHours ?? 0} audio-hours.`, ); lines.push( `Sweep days are projected as measured seconds-per-audio-hour x ` + `${sample.corpus.audioHours} corpus audio-hours, one lane, no parallelism.`, ); lines.push(""); lines.push( "| Candidate | Zero-yield chunks | Chapters/h | Rejection rate | Max gap | Generic | Dup | tok/s | s per audio-h | **Sweep days** |", ); lines.push( "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |", ); for (const s of scores) { const t = s.totals; lines.push( `| \`${s.candidate.key}\` | ${t.zeroYieldChunks}/${t.chunks} (${pct(t.zeroYieldRate)}) | ` + `${t.chaptersPerHour} | ${pct(t.rejectionRate)} | ${toHms(t.maxGapSeconds)} | ` + `${pct(t.genericTitleRate)} | ${pct(t.duplicateTitleRate)} | ${t.tokensPerSecond} | ` + `${t.secondsPerAudioHour} | **${t.projectedSweepDays}** |`, ); } lines.push(""); lines.push("## Boundary accuracy vs the uploader's own chapters"); lines.push(""); lines.push( "The oracle is `metadata.info.json.chapters` — marks a human authored while", "watching, who never saw our prompt. Boilerplate marks (\"Intro\", \"Sponsor\")", "and the boundary at 00:00 are dropped before scoring: neither carries", "segmentation information, and the origin would be a free hit for every", "candidate. Matching is one-to-one and closest-pair-first, so a cluster of", "boundaries around one uploader mark scores one match, not many.", ); lines.push(""); lines.push( "| Candidate | Videos scored | Uploader marks | Median offset | Within 30s | P@30 | R@30 | **F1@30** | F1@60 | Name-in-title |", ); lines.push("| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |"); for (const s of scores) { const b = s.totals.boundary; const t30 = b.byTolerance.find((x) => x.toleranceSeconds === 30); const t60 = b.byTolerance.find((x) => x.toleranceSeconds === 60); lines.push( `| \`${s.candidate.key}\` | ${b.videosScored} | ${b.referenceBoundaries} | ` + `${b.medianOffsetSeconds === null ? "—" : `${b.medianOffsetSeconds}s`} | ` + `${pct(b.withinThirtySecondsRate)} | ${pct(t30?.precision ?? 0)} | ${pct(t30?.recall ?? 0)} | ` + `**${pct(t30?.f1 ?? 0)}** | ${pct(t60?.f1 ?? 0)} | ${pct(s.totals.speakerNameTitleRate)} |`, ); } lines.push(""); lines.push("## Rejections by guard"); lines.push(""); const codes = Array.from( new Set(scores.flatMap((s) => Object.keys(s.totals.warningsByCode))), ).sort(); lines.push(`| Candidate | ${codes.join(" | ")} |`); lines.push(`| --- | ${codes.map(() => "---").join(" | ")} |`); for (const s of scores) { lines.push( `| \`${s.candidate.key}\` | ${codes .map((c) => s.totals.warningsByCode[c] ?? 0) .join(" | ")} |`, ); } lines.push(""); lines.push("## Per-video"); lines.push(""); lines.push( "| Candidate | Video | Bucket | Chunks | Zero-yield | Chapters | Max gap | Engine s | Marks | F1@30 | Speakers |", ); lines.push("| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |"); for (const s of scores) { for (const v of s.videos) { const f1 = v.boundary?.scores.find((x) => x.toleranceSeconds === 30)?.f1; lines.push( `| \`${s.candidate.key}\` | \`${v.slug}\` | ${v.bucket} | ${v.chunks} | ` + `${v.zeroYieldChunks} | ${v.kept} | ${toHms(v.maxGapSeconds)} | ${Math.round(v.engineSeconds)} | ` + `${v.boundary?.referenceCount ?? "—"} | ${f1 === undefined ? "—" : pct(f1)} | ` + `${v.speakersRendered || "—"} |`, ); } } lines.push(""); lines.push("## Sample output (first chapters per video)"); lines.push(""); for (const s of scores) { lines.push(`### \`${s.candidate.key}\``); lines.push(""); for (const v of s.videos) { lines.push(`**${v.slug}** (${v.bucket}, ${toHms(v.durationSeconds)})`); lines.push(""); for (const t of v.sampleTitles) lines.push(`- ${t}`); if (v.sampleTitles.length === 0) lines.push("- _(nothing survived the guards)_"); lines.push(""); } } return `${lines.join("\n")}\n`; } function pct(v: number): string { return `${Math.round(v * 1000) / 10}%`; } // --------------------------------------------------------------------------- // Main // --------------------------------------------------------------------------- // "model@ctx" or "model@ctx:think" / "model@ctx:nothink". // // maxCues is DERIVED from the context rather than configured: the 1200-cue // default is sized for a 16k window (~10 tokens/cue -> ~12k tokens of transcript // plus room for prompt and response), so halving the context must halve the // slice or every call silently truncates — the exact failure that made the first // smoke test summarize a fragment. function parseCandidate( spec: string, mode: DigestTimestampMode, speakers: boolean, maxCuesOverride: number | null, ): Candidate { const [modelPart, rest] = spec.split("@"); const [ctxPart, thinkPart] = (rest ?? "").split(":"); const numCtx = ctxPart ? Number(ctxPart) : undefined; // The SAME derivation production uses, not a parallel copy — otherwise the // bake-off scores a chunk size the sweep would never actually run. // // --max-cues breaks that tie deliberately, for one reason: a paired A/B needs // HEADROOM. Speaker labels add ~10% to the prompt, and on a pool of long VODs // the un-labelled chunk is already near the window — so at the derived size // the speakers-on arm would truncate where speakers-off did not, and the // measurement would be of truncation. Raising numCtx while pinning maxCues // gives both arms the same cue count with room to spare. BOTH ARMS ALWAYS GET // THE SAME VALUE; the report records it. const maxCues = maxCuesOverride ?? maxCuesForContext(numCtx); return { // The speaker axis and a cue override only show in the key when set, so // existing round labels keep reading the way rounds 1-2 wrote them. key: `${modelPart}@${numCtx ?? "default"}/${mode}` + `${maxCuesOverride ? `/${maxCuesOverride}cues` : ""}${speakers ? "/speakers" : ""}`, model: modelPart, ...(numCtx ? { numCtx } : {}), maxCues, timestampMode: mode, speakers, ...(thinkPart === "think" ? { think: true } : thinkPart === "nothink" ? { think: false } : {}), }; } // --------------------------------------------------------------------------- // The chapter-oracle sample // --------------------------------------------------------------------------- // Pick a sample restricted to videos that can actually be SCORED — i.e. that // carry >=`minChapters` non-boilerplate uploader marks. // // WHY IT DOES NOT READ EVERY METADATA FILE. metadata.info.json runs to ~100 KB // and only ~19% of videos carry chapters, so parsing all 77k to find them would // read several GB to answer a question a stable-hash walk answers in a few // hundred reads: candidates are visited in the same deterministic order // pickSample uses, and the walk stops as soon as every stratum is full. // // KEPT IN A SEPARATE FILE from sample.json, deliberately. Rounds 1 and 2 are // scored against that sample; repointing it would silently invalidate the // comparison this harness exists to protect. async function pickChapterSample( outPath: string, opts: { minChapters: number; perBucket: number | null; requireAudio: boolean }, ): Promise { const paths = getPaths(); const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true }); const statsByPath = root.openDB< { metaMs: number; stat: VideoStat }, [string, string] >({ name: "statsByPath", encoding: "msgpack" }); const byBucket = new Map(); for (const b of BUCKETS) byBucket.set(b, []); let videosScanned = 0; let videosWithTranscript = 0; let totalSeconds = 0; let longTailVideos = 0; let longTailSeconds = 0; for (const { key, value } of statsByPath.getRange()) { const stat = value.stat; videosScanned++; if (!stat.hasTranscript || !(stat.duration > 0)) continue; videosWithTranscript++; totalSeconds += stat.duration; if (stat.duration > 4 * 3600) { longTailVideos++; longTailSeconds += stat.duration; } const bucket = bucketFor(stat.duration); if (!bucket) continue; if (!stat.cueCount || stat.cueCount < 30) continue; byBucket.get(bucket)!.push({ slug: stat.slug, channelSlug: stat.channelSlug, videoId: stat.id, videoDir: (key as [string, string])[1], title: stat.title, bucket, durationSeconds: Math.round(stat.duration), cueCount: stat.cueCount, }); } await root.close(); const videos: SampleVideo[] = []; for (const b of BUCKETS) { const pool = byBucket.get(b)!; pool.sort((a, c) => stableHash(a.slug) - stableHash(c.slug)); const want = opts.perBucket ?? BUCKET_BOUNDS[b].want; const seenChannels = new Set(); const taken: SampleVideo[] = []; let probed = 0; for (const v of pool) { if (taken.length >= want) break; // One per channel first, for the same reason pickSample does it: a // stratum drawn from one creator measures a house style. if (seenChannels.has(v.channelSlug)) continue; probed++; const oracle = await uploaderBoundaries(v); if (!oracle || oracle.starts.length < opts.minChapters) continue; if (opts.requireAudio) { // isRealAudioFile, not an extension test of my own: it is the same // predicate the cleanup and diarization lanes use, so "has audio" means // here exactly what it means to the job that would diarize it. const files = await readdir(videoDirFor(v)).catch(() => [] as string[]); if (!files.some((f) => isRealAudioFile(f))) continue; } seenChannels.add(v.channelSlug); taken.push({ ...v, uploaderChapters: oracle.starts.length }); } if (taken.length < want) { console.warn( `Warning: bucket ${b} wanted ${want} scorable videos but only ${taken.length} qualify ` + `(probed ${probed} candidates).`, ); } videos.push(...taken); } const sample: Sample = { version: 1, pickedAt: new Date().toISOString(), corpus: { videosScanned, videosWithTranscript, audioHours: Math.round(totalSeconds / 3600), longTailVideos, longTailAudioHours: Math.round(longTailSeconds / 3600), }, videos, }; await mkdir(path.dirname(outPath), { recursive: true }); await writeFile(outPath, `${JSON.stringify(sample, null, 2)}\n`); for (const v of videos) { console.log( ` ${v.bucket.padEnd(8)} ${toHms(v.durationSeconds)} ${String(v.uploaderChapters).padStart(3)} marks ` + `${v.slug} ${v.title.slice(0, 55)}`, ); } console.log(`Wrote ${outPath} (${videos.length} scorable video(s))`); } // --------------------------------------------------------------------------- // Free scorer smoke test // --------------------------------------------------------------------------- // Score the digests ALREADY on disk against their uploader chapters. // // This is the cheapest honest check available: no model call, no write, and it // answers "does the metric move, and is the plumbing right" before a generation // run is spent finding out. If this prints nonsense — every F1 at 0 or 1 — the // scorer is wrong and no A/B built on it would mean anything. async function scoreExisting(minChapters: number, limit: number): Promise { const paths = getPaths(); const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true }); const statsByPath = root.openDB< { metaMs: number; stat: VideoStat }, [string, string] >({ name: "statsByPath", encoding: "msgpack" }); const candidates: SampleVideo[] = []; for (const { key, value } of statsByPath.getRange()) { const stat = value.stat; if (!stat.hasTranscript || !(stat.duration > 0)) continue; candidates.push({ slug: stat.slug, channelSlug: stat.channelSlug, videoId: stat.id, videoDir: (key as [string, string])[1], title: stat.title, bucket: bucketFor(stat.duration) ?? "short", durationSeconds: Math.round(stat.duration), cueCount: stat.cueCount ?? 0, }); } await root.close(); const rows: { slug: string; marks: number; generated: number; medianOffset: number | null; precision30: number; recall30: number; recall60: number; f1at30: number; f1at60: number; }[] = []; for (const v of candidates) { if (rows.length >= limit) break; const dir = videoDirFor(v); const digest = await loadDigest(dir); const items = digest?.sections?.chapters?.items ?? []; if (items.length === 0) continue; const oracle = await uploaderBoundaries(v); if (!oracle || oracle.starts.length < minChapters) continue; const report = scoreBoundaries(oracle.starts, items.map((c) => c.start)); if (report.referenceCount === 0) continue; const s30 = report.scores.find((s) => s.toleranceSeconds === 30); const s60 = report.scores.find((s) => s.toleranceSeconds === 60); rows.push({ slug: v.slug, marks: report.referenceCount, generated: report.generatedCount, medianOffset: report.medianOffsetSeconds, precision30: s30?.precision ?? 0, recall30: s30?.recall ?? 0, recall60: s60?.recall ?? 0, f1at30: s30?.f1 ?? 0, f1at60: s60?.f1 ?? 0, }); } if (rows.length === 0) { console.log( `No video on disk has both a digest and >=${minChapters} non-boilerplate uploader chapters.`, ); return; } console.log( `Scored ${rows.length} existing digest(s) against uploader chapters (no model calls, no writes).\n`, ); console.log( " slug marks gen medOff P@30 R@30 R@60 F1@30 F1@60", ); for (const r of rows) { console.log( ` ${r.slug.padEnd(36).slice(0, 36)} ${String(r.marks).padStart(5)} ` + `${String(r.generated).padStart(3)} ${String(r.medianOffset ?? "—").padStart(6)} ` + `${pct(r.precision30).padStart(5)} ${pct(r.recall30).padStart(5)} ` + `${pct(r.recall60).padStart(5)} ${pct(r.f1at30).padStart(5)} ${pct(r.f1at60).padStart(5)}`, ); } const sumOf = (f: (r: (typeof rows)[number]) => number): number => rows.reduce((a, r) => a + f(r), 0); // POOLED, not a mean of per-video rates: a 3-mark video must not weigh the // same as a 29-mark one. const marks = sumOf((r) => r.marks); const generated = sumOf((r) => r.generated); const matched30 = sumOf((r) => r.recall30 * r.marks); const matched60 = sumOf((r) => r.recall60 * r.marks); const p30 = generated > 0 ? matched30 / generated : 0; const r30 = marks > 0 ? matched30 / marks : 0; console.log( `\n POOLED: ${marks} uploader mark(s), ${generated} generated. ` + `P@30 ${pct(p30)}, R@30 ${pct(r30)}, R@60 ${pct(marks > 0 ? matched60 / marks : 0)}, ` + `F1@30 ${pct(p30 + r30 > 0 ? (2 * p30 * r30) / (p30 + r30) : 0)}`, ); console.log( ` NOTE: the digest targets one chapter per ${DIGEST_MINUTES_PER_CHAPTER} minutes and so is\n` + ` deliberately DENSER than uploader chaptering (${generated} vs ${marks} here). Precision is\n` + ` therefore partly a density artifact; recall is the half that answers "did it find the\n` + ` human's boundaries". Compare candidates on recall AND on chapters/h together.`, ); } async function main(): Promise { const flags = parseFlags(process.argv.slice(2)); const outDir = flags.outDir ?? path.join(process.cwd(), "..", "plans", "bakeoff"); const samplePath = flags.sample ?? path.join(outDir, "sample.json"); const minChapters = flags["min-chapters"] ? Number(flags["min-chapters"]) : 4; if (flags.pick === "true") { await pickSample(samplePath); return; } if (flags["pick-chapters"] === "true") { await pickChapterSample(samplePath, { minChapters, perBucket: flags["per-bucket"] ? Number(flags["per-bucket"]) : null, requireAudio: flags["require-audio"] === "true", }); return; } if (flags["score-existing"] === "true") { await scoreExisting(minChapters, flags.limit ? Number(flags.limit) : 200); return; } const sample = JSON.parse(await readFile(samplePath, "utf8")) as Sample; const label = flags.label ?? "round"; const buckets = (flags.buckets ?? BUCKETS.join(",")) .split(",") .map((b) => b.trim()) .filter((b): b is Bucket => (BUCKETS as readonly string[]).includes(b)); const modes = (flags.modes ?? "absolute") .split(",") .map((m) => m.trim()) .filter((m): m is DigestTimestampMode => m === "absolute" || m === "chunk-local"); const specs = (flags.candidates ?? "qwen2.5:7b@16384") .split(",") .map((s) => s.trim()) .filter(Boolean); // The paired A/B axis. Default "off" — the same rendering every earlier round // used, so omitting the flag reproduces them. const speakerArms = (flags.speakers ?? "off") .split(",") .map((s) => s.trim()) .filter((s) => s === "on" || s === "off") .map((s) => s === "on"); const videos = sample.videos.filter((v) => buckets.includes(v.bucket)); if (videos.length === 0) throw new Error(`No sample videos in buckets ${buckets.join(",")}`); const maxCuesOverride = flags["max-cues"] ? Number(flags["max-cues"]) : null; const candidates: Candidate[] = []; for (const spec of specs) { for (const mode of modes) { for (const speakers of speakerArms) { candidates.push(parseCandidate(spec, mode, speakers, maxCuesOverride)); } } } console.log( `Bake-off ${label}: ${candidates.length} candidate(s) x ${videos.length} video(s) ` + `(${Math.round(videos.reduce((a, v) => a + v.durationSeconds, 0) / 3600)} audio-hours each).`, ); // Up front, not just before the final write, so the per-video checkpoint below // has somewhere to land from the very first video. await mkdir(outDir, { recursive: true }); const scores: CandidateScore[] = []; for (const candidate of candidates) { console.log(`\n=== ${candidate.key} (maxCues ${candidate.maxCues}) ===`); const startedAt = Date.now(); const perVideo: VideoScore[] = []; for (const video of videos) { const s = await scoreVideo(video, candidate, (m) => console.log(m)); if (s) perVideo.push(s); // CHECKPOINT AFTER EVERY VIDEO, because the reports are only written when // the whole run finishes and a long run does not reliably get there. A // 12-video run was OOM-killed on its last video after 26 minutes of engine // time and left NOTHING behind — no error, no partial, just a dead process // (the kill is a SIGKILL, so no handler can save it either). On a box that // swaps, "it completed 11 of 12" has to survive. await writeFile( path.join(outDir, `${label}.partial.json`), `${JSON.stringify({ label, candidate: candidate.key, videos: perVideo }, null, 2)}\n`, ).catch(() => {}); } const agg = aggregate(candidate, perVideo); // Days, from measured seconds-per-audio-hour against the corpus total // captured in the same scan that picked the sample. agg.totals.projectedSweepDays = round( (agg.totals.secondsPerAudioHour * sample.corpus.audioHours) / 86400, 1, ); scores.push(agg); console.log( `--- ${candidate.key}: ${agg.totals.kept} chapters, ` + `${agg.totals.zeroYieldChunks}/${agg.totals.chunks} zero-yield, ` + `${agg.totals.tokensPerSecond} tok/s, ` + `projected ${agg.totals.projectedSweepDays} sweep days ` + `(wall ${Math.round((Date.now() - startedAt) / 60000)} min)`, ); } scores.sort((a, b) => a.totals.zeroYieldRate - b.totals.zeroYieldRate); await mkdir(outDir, { recursive: true }); const jsonPath = path.join(outDir, `${label}.json`); const mdPath = path.join(outDir, `${label}.md`); await writeFile( jsonPath, `${JSON.stringify({ label, sample: samplePath, corpus: sample.corpus, buckets, scores }, null, 2)}\n`, ); await writeFile(mdPath, markdownReport(label, sample, buckets, scores)); console.log(`\nWrote ${jsonPath}\nWrote ${mdPath}`); } main().catch((err) => { console.error(err); process.exit(1); });