// How an attribution record is SCORED. Pure — no I/O, no model — so it // unit-tests, and so every harness that measures a variant measures it the same // way. // // WHY THIS IS ITS OWN MODULE. These metrics were written inside // plans/tools/attribution-pilot.ts, which meant (a) the headline number that rejected // the text-only lane was code nobody had ever run against a known answer, and // (b) a second harness comparing a new prompt against the pilot's baseline would // have had to copy them. plans/FACTS.md records the digest lane already making // exactly that bet — a regex copy-pasted into digest-validate.ts "so the two are // comparable" — and calls it a bet that has to keep being won. One module, one // definition, and a test file with hand-built fixtures so the definition is // pinned to an answer worked out by hand rather than to whatever it printed the // first time. // // THE METRICS, and what each is FOR: // // newLabelsPerChunk THE HEADLINE. Cross-chunk identity is the one // property the text-only lane exists for. // dominantSpeakerShare The degenerate-case guard. One label at ~100% of // talk time scores a PERFECT 0 on the headline while // being worthless, and the headline alone cannot tell // those apart. Never read one without the other. // transcriptTextLabelRate What round 2 exists to drive to zero. See below. // truncatedLabelRate The direct evidence for the same failure. // // NULL, NOT ZERO, whenever a property could not be measured. A single-chunk // video has no seam to cross; a chunk-count mismatch means the reconstruction // disagrees with the runner. Reporting 0 for "not measured" is how a harness // reports a perfect score for something that never ran. import { attributionSpeechSeconds, type AttributionRecord, } from "./attribution"; import { ATTRIBUTION_LABEL_MAX } from "./attributionPrompt"; import { DIGEST_OVERLAP_CUES } from "./digestPrompt"; import { chunkCuesForContext } from "./transcriptWindow"; import type { Cue } from "./vtt"; // Roles a label can carry that name nobody. isUselessSpeakerLabel already blocks // "Speaker 1" before the write, so the record cannot contain one; this measures // the WIDER set the runner permits. Anchored whole-string, not a prefix match: // "Host" is generic, "Host Jane Doe" is not. export const GENERIC_LABEL_RE = /^(the\s+)?(co-?host|host|guest|caller|narrator|interviewer|moderator|panelist|announcer|commentator|reporter|audience|crowd|man|woman|male|female|person|voice|speaker)(\s*\d+)?$/i; // A label with at least this many words is not a name or a role — it is prose. // // Calibrated against the measured failure, not guessed. In the 2026-08-08 pilot // the labels that were transcript text ran 5-12 words ("the general consensus // were religion is subscribed to like"), while every label that named somebody // was 1-3 ("Bronca", "Nick the Dick", "Stephen Colbert"). Five is the first // value above every real name observed and below every prose fragment observed. export const TRANSCRIPT_TEXT_MIN_WORDS = 5; // Chunk boundaries EXACTLY as findTurns computes them — same chunker, same // overlap, same floor/ceil. Every chunk-indexed metric rests on this agreeing // with the runner, which is why it takes maxCues from the resolved config rather // than a literal 600, and why scoreRecord asserts its count against // provenance.chunks and reports null (never 0) on a mismatch. export function chunkRangesFor( cues: Cue[], maxCues: number, ): { start: number; end: number }[] { return chunkCuesForContext(cues, { maxCues, overlapCues: DIGEST_OVERLAP_CUES }).map( (chunk) => { const start = Math.max(0, Math.floor(chunk[0].start)); const last = chunk[chunk.length - 1]; return { start, end: Math.max(start, Math.ceil(last.end || last.start)) }; }, ); } function normalizeForCompare(label: string): string { return label .trim() .replace(/\s+/g, " ") .replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, "") .toLowerCase(); } // Labels the runner's deliberately conservative normalizeLabel keeps apart but // that are obviously one person ("Jane" / "Jane Doe"). The runner is right to // refuse the merge — a merged speaker silently attributes words to the wrong // person — but a SCORER may flag, and this rate is the direct read on whether // that conservatism is set right. export function nearDuplicate(a: string, b: string): boolean { const ta = new Set(normalizeForCompare(a).split(" ").filter(Boolean)); const tb = new Set(normalizeForCompare(b).split(" ").filter(Boolean)); if (ta.size === 0 || tb.size === 0) return false; const [small, big] = ta.size <= tb.size ? [ta, tb] : [tb, ta]; let shared = 0; for (const t of small) if (big.has(t)) shared++; if (shared === small.size) return true; // proper token-subset for (const t of small) if (t.length >= 4 && big.has(t)) return true; return false; } // The whole transcript as one whitespace-collapsed, case-folded string, for // substring testing. Collapsing is what makes the comparison work at all: the // model is shown `[HH:MM:SS] text` LINES, and the labels it echoed back in the // pilot carry the cue's own internal newlines ("...atheist would\ncare about // this right") which appear nowhere in the cue text as stored. export function transcriptHaystack(cues: Cue[]): string { return cues .map((c) => c.text) .join(" ") .replace(/\s+/g, " ") .toLowerCase(); } // Is this "label" actually a chunk of the transcript the model was reading? // // THE FAILURE THIS MEASURES. `speaker` ships as a free string (minLength 2, // maxLength 60) and nothing makes transcript text unrepresentable in it, so on a // long prompt the model stops answering the question and continues the passage // into the label field instead. The 2026-08-08 pilot produced 346 such labels on // one 13-chunk video. isUselessSpeakerLabel does not catch them — it only blocks // "Speaker N". // // Two independent tests, either of which is sufficient: // // 1. The label contains a LINE BREAK. A speaker's name never does; a // transcript cue rendered over two lines always can. // 2. The label is >= TRANSCRIPT_TEXT_MIN_WORDS words AND appears verbatim in // the transcript. Both halves are needed: prose alone would flag a genuine // long role, and substring alone would flag "Nick the Dick", a real name // the transcript states outright. // // This is a LOWER BOUND. A fragment truncated mid-way across a cue join, or one // the model paraphrased, does not match and is not counted. It is the right // direction to be wrong in for a metric whose target is zero. export function isTranscriptTextLabel(label: string, haystack: string): boolean { if (/[\r\n]/.test(label)) return true; const normalized = label.replace(/\s+/g, " ").trim().toLowerCase(); const words = normalized.split(" ").filter(Boolean); if (words.length < TRANSCRIPT_TEXT_MIN_WORDS) return false; return haystack.includes(normalized); } function median(values: number[]): number { if (values.length === 0) return 0; const s = [...values].sort((a, b) => a - b); const mid = Math.floor(s.length / 2); return s.length % 2 ? s[mid] : (s[mid - 1] + s[mid]) / 2; } // What a record is scored AGAINST. Deliberately not the pilot's PricedVideo: // a scorer that names a harness's internal type cannot be shared with a second // harness, which is the whole reason this module exists. export type ScoreTarget = { // As chunkRangesFor computed them, for the SAME maxCues the run used. chunkRanges: { start: number; end: number }[]; durationSeconds: number; // Needed only by transcriptTextLabelRate, which reports null without them. cues?: Cue[]; }; export type VideoMetrics = { labels: string[]; labelsPerVideo: number; newLabelsPerChunk: number | null; singletonLabelRate: number | null; dominantSpeakerShare: number; talkShares: number[]; nearDuplicateLabelRate: number; nearDuplicatePairs: string[]; genericLabelRate: number; genericSecondsShare: number; nonAsciiLabelRate: number; // Share of labels that are transcript text. Null when no cues were supplied — // never 0, which would read as a clean sheet. transcriptTextLabelRate: number | null; transcriptTextLabels: string[]; // Share of labels sitting at exactly ATTRIBUTION_LABEL_MAX characters. A name // landing on the cap is possible; a POPULATION of them means the decoder was // mid-sentence when the schema's maxLength cut it off, which is the same // failure transcriptTextLabelRate names, measured without needing the cues. truncatedLabelRate: number; coverageRate: number; maxGapSeconds: number; medianGapSeconds: number; markFreeChunks: number | null; segments: number; chunkReconstructionOk: boolean; labelFirstChunk: (number | null)[]; labelChunkCount: (number | null)[]; }; export function scoreRecord( record: AttributionRecord, target: ScoreTarget, ): VideoMetrics { const ranges = target.chunkRanges; // The runner records what it actually chunked; if our reconstruction differs, // every chunk-indexed number below is meaningless and is reported as null. const recordedChunks = record.provenance.chunks; const reconstructionOk = recordedChunks === undefined || recordedChunks === ranges.length; const multiChunk = reconstructionOk && ranges.length > 1; const chunkOf = (t: number): number => { // The 40-cue overlap means a seam second sits in two chunks; first match is // the earlier one, which is the chunk that ESTABLISHED the label. for (let i = 0; i < ranges.length; i++) { if (t >= ranges[i].start && t <= ranges[i].end) return i; } return t < ranges[0].start ? 0 : ranges.length - 1; }; const chunksByLabel = new Map>(); for (const s of record.segments) { const set = chunksByLabel.get(s.speaker) ?? new Set(); set.add(chunkOf(s.start)); chunksByLabel.set(s.speaker, set); } const labels = record.speakers.map((s) => s.label); const seconds = record.speakers.map((s) => s.seconds ?? 0); const totalSeconds = seconds.reduce((a, b) => a + b, 0); const labelFirstChunk = record.speakers.map((s) => { const set = chunksByLabel.get(s.index); if (!reconstructionOk || !set || set.size === 0) return null; return Math.min(...set); }); const labelChunkCount = record.speakers.map((s) => { const set = chunksByLabel.get(s.index); if (!reconstructionOk || !set) return null; return set.size; }); // THE HEADLINE. Labels first appearing after chunk 0, per later chunk. Near 0 // means the cast stabilised; >= 1 means the model reinvented it every chunk // and the lane is not delivering the one property it exists for. const newLabelsPerChunk = multiChunk ? labelFirstChunk.filter((c) => c !== null && c > 0).length / (ranges.length - 1) : null; const singletonLabelRate = multiChunk && labels.length > 0 ? labelChunkCount.filter((c) => c === 1).length / labels.length : null; // THE DEGENERATE-CASE GUARD. One label at ~100% of talk time scores a perfect // 0 on the headline while being worthless, and the headline alone cannot tell // those apart. const dominantSpeakerShare = totalSeconds > 0 ? Math.max(...seconds) / totalSeconds : 0; const dupPairs: string[] = []; const involved = new Set(); for (let i = 0; i < labels.length; i++) { for (let j = i + 1; j < labels.length; j++) { if (!nearDuplicate(labels[i], labels[j])) continue; dupPairs.push(`${labels[i]} ~ ${labels[j]}`); involved.add(i); involved.add(j); } } const genericIdx = labels .map((l, i) => (GENERIC_LABEL_RE.test(l.trim()) ? i : -1)) .filter((i) => i >= 0); const genericSeconds = genericIdx.reduce((a, i) => a + seconds[i], 0); const haystack = target.cues ? transcriptHaystack(target.cues) : null; const transcriptTextLabels = haystack === null ? [] : labels.filter((l) => isTranscriptTextLabel(l, haystack)); // COVERAGE, and what it is worth. Text-only marks are TILED — each runs to the // next change, the last to the end — so this is ~1.0 by construction. It is a // structural check (anything else is a bug), not a quality signal. The honest // coverage metric for this lane is markFreeChunks. const duration = target.durationSeconds || 0; const covered = attributionSpeechSeconds(record); const sorted = [...record.segments].sort((a, b) => a.start - b.start); const gaps: number[] = []; let cursor = 0; for (const s of sorted) { if (s.start > cursor) gaps.push(s.start - cursor); cursor = Math.max(cursor, s.end); } if (duration > cursor) gaps.push(duration - cursor); const markFreeChunks = reconstructionOk ? ranges.filter( (r) => !record.segments.some((s) => s.start >= r.start && s.start <= r.end), ).length : null; return { labels, labelsPerVideo: labels.length, newLabelsPerChunk, singletonLabelRate, dominantSpeakerShare, talkShares: totalSeconds > 0 ? seconds.map((s) => s / totalSeconds) : seconds.map(() => 0), nearDuplicateLabelRate: labels.length > 0 ? involved.size / labels.length : 0, nearDuplicatePairs: dupPairs, genericLabelRate: labels.length > 0 ? genericIdx.length / labels.length : 0, genericSecondsShare: totalSeconds > 0 ? genericSeconds / totalSeconds : 0, nonAsciiLabelRate: labels.length > 0 ? labels.filter((l) => /[^\x20-\x7E]/.test(l)).length / labels.length : 0, transcriptTextLabelRate: haystack === null || labels.length === 0 ? null : transcriptTextLabels.length / labels.length, transcriptTextLabels, truncatedLabelRate: labels.length > 0 ? labels.filter((l) => l.length === ATTRIBUTION_LABEL_MAX).length / labels.length : 0, coverageRate: duration > 0 ? covered / duration : 0, maxGapSeconds: gaps.length ? Math.max(...gaps) : 0, medianGapSeconds: median(gaps), markFreeChunks, segments: record.segments.length, chunkReconstructionOk: reconstructionOk, labelFirstChunk, labelChunkCount, }; }