// Does a generated chapter boundary land where a human put one? // // WHY THIS EXISTS. Every metric digest-bakeoff.ts scored before this one — // zeroYieldRate, chaptersPerHour, rejectionRate, maxGapSeconds, // genericTitleRate, duplicateTitleRate — is a DEFECT COUNTER. Each answers "how // broken is the output", none answers "is the segmentation right". So the // bake-off could rank candidates by which was least malformed and still could // not say which produced better chapters. That is the gap this closes. // // THE ORACLE IS THE UPLOADER'S OWN CHAPTERS, and the choice is deliberate on two // grounds. First, independence: scoring "chapter starts align with speaker // changes" against a variant that was GIVEN the speaker changes proves nothing, // so the ground truth has to come from outside the experiment. Uploader chapters // are authored by a human who watched the video and never saw our prompt. // Second, posture: a boundary is a FACT about where a subject changes, not // expression. Scoring against timestamps reproduces nothing, which is why this // is safe where admitting the uploader's chapter TITLES into the corpus would // not be (see PLAN.md on why the archive publishes new expression about works). // // Pure — no I/O, no clock — so it is unit-testable and can be reused by any // future digest change, which is worth more than the one experiment that // prompted it. // The reference boundary at t=0 is dropped before scoring, always. // // Nearly every uploader chapter list opens with a chapter at 00:00 ("Intro"), // and every segmentation trivially has a boundary at the start of the video. // Counting it would hand both arms of any A/B a free true-positive that carries // no information about segmentation skill, inflating precision, recall and F1 by // roughly 1/n on short chapter lists. A metric that cannot be lost is not a // measurement. export const BOUNDARY_ORIGIN_EPSILON_SECONDS = 5; // The tolerances every report quotes. 30 s is "the model found the same moment"; // 60 s is "the model found the same transition, a little late". Both are // reported because a variant that trades tightness for recall should be visible // as exactly that, not averaged into one number. export const BOUNDARY_TOLERANCES_SECONDS = [30, 60] as const; export type BoundaryMatch = { reference: number; generated: number; distance: number; }; export type BoundaryScore = { toleranceSeconds: number; referenceCount: number; generatedCount: number; matched: number; // matched / generatedCount — of the boundaries we emitted, how many a human // agrees with. Low precision means over-segmentation. precision: number; // matched / referenceCount — of the boundaries a human drew, how many we // found. Low recall means we missed real transitions. recall: number; f1: number; matches: BoundaryMatch[]; }; export type BoundaryReport = { referenceCount: number; generatedCount: number; // Per reference boundary, the distance to the NEAREST generated boundary, // independent of any matching. This is the "how far off were we" view, and it // is reported alongside F1 because the two fail differently: a model that // emits one boundary 5 s from every reference scores a perfect median offset // and a terrible precision. medianOffsetSeconds: number | null; meanOffsetSeconds: number | null; withinThirtySeconds: number; scores: BoundaryScore[]; }; function median(values: number[]): number | null { if (values.length === 0) return null; const sorted = [...values].sort((a, b) => a - b); const mid = Math.floor(sorted.length / 2); return sorted.length % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid]; } // Drop the origin boundary and any duplicate/unsorted noise, so callers can pass // raw starts from either side without pre-cleaning them. export function normalizeBoundaries( starts: readonly number[], originEpsilon = BOUNDARY_ORIGIN_EPSILON_SECONDS, ): number[] { const seen = new Set(); const out: number[] = []; for (const raw of starts) { if (!Number.isFinite(raw)) continue; const s = Math.round(raw); if (s <= originEpsilon) continue; if (seen.has(s)) continue; seen.add(s); out.push(s); } return out.sort((a, b) => a - b); } // Greedy one-to-one matching, closest pair first. // // ONE-TO-ONE IS THE POINT. A model that emits ten boundaries clustered inside // one 30 s window must not be credited with ten matches against a single // uploader boundary — that is the exact over-segmentation failure precision is // supposed to punish, and a nearest-neighbour count would reward it instead. // Closest-first (rather than left-to-right) keeps the pairing stable when two // references sit within one tolerance of the same generated boundary. export function matchBoundaries( reference: readonly number[], generated: readonly number[], toleranceSeconds: number, ): BoundaryMatch[] { const pairs: BoundaryMatch[] = []; for (const r of reference) { for (const g of generated) { const distance = Math.abs(r - g); if (distance <= toleranceSeconds) pairs.push({ reference: r, generated: g, distance }); } } // Ties broken by position so the result is deterministic across runs. pairs.sort( (a, b) => a.distance - b.distance || a.reference - b.reference || a.generated - b.generated, ); const usedReference = new Set(); const usedGenerated = new Set(); const matches: BoundaryMatch[] = []; for (const p of pairs) { if (usedReference.has(p.reference) || usedGenerated.has(p.generated)) continue; usedReference.add(p.reference); usedGenerated.add(p.generated); matches.push(p); } return matches.sort((a, b) => a.reference - b.reference); } export function scoreBoundariesAt( reference: readonly number[], generated: readonly number[], toleranceSeconds: number, ): BoundaryScore { const matches = matchBoundaries(reference, generated, toleranceSeconds); const matched = matches.length; const precision = generated.length > 0 ? matched / generated.length : 0; const recall = reference.length > 0 ? matched / reference.length : 0; const f1 = precision + recall > 0 ? (2 * precision * recall) / (precision + recall) : 0; return { toleranceSeconds, referenceCount: reference.length, generatedCount: generated.length, matched, precision, recall, f1, matches, }; } // The whole picture for one video. `reference` and `generated` are raw starts in // seconds; normalization (origin drop, de-dup, sort) happens here so every // caller gets the same treatment. export function scoreBoundaries( referenceRaw: readonly number[], generatedRaw: readonly number[], tolerances: readonly number[] = BOUNDARY_TOLERANCES_SECONDS, ): BoundaryReport { const reference = normalizeBoundaries(referenceRaw); const generated = normalizeBoundaries(generatedRaw); const offsets: number[] = []; for (const r of reference) { let best = Infinity; for (const g of generated) best = Math.min(best, Math.abs(r - g)); if (Number.isFinite(best)) offsets.push(best); } return { referenceCount: reference.length, generatedCount: generated.length, medianOffsetSeconds: median(offsets), meanOffsetSeconds: offsets.length > 0 ? offsets.reduce((a, b) => a + b, 0) / offsets.length : null, withinThirtySeconds: offsets.filter((d) => d <= 30).length, scores: tolerances.map((t) => scoreBoundariesAt(reference, generated, t)), }; } // --------------------------------------------------------------------------- // Uploader chapter titles: usable as boilerplate, never as a quality reference // --------------------------------------------------------------------------- // Uploader chapter lists are dominated by structural marks — the modal first // chapter across this corpus is literally "Intro". They are therefore NOT a // reference for title quality (scoring our titles against them would score us // against boilerplate), but the marks themselves are worth recognizing because a // boundary at "Sponsor" is a boundary in the AD READ, not in the subject, and // crediting a model for finding it measures the wrong thing. const BOILERPLATE_TITLE_RE = /^(intro(duction)?|outro|start|beginning|end(ing)?|sponsor(ed)?( segment| message)?|ad( read| break)?|advert(isement)?|thanks?( for watching)?|subscribe|like and subscribe|patreon|merch|announcements?|housekeeping|stream starts?( soon)?|waiting( screen)?|brb|be right back|credits|outro music|music)\b/i; export function isBoilerplateChapterTitle(title: string): boolean { return BOILERPLATE_TITLE_RE.test(title.trim()); }