// The attribution metrics, pinned to answers worked out BY HAND. // // WHY THIS FILE EXISTS. The number that rejected the text-only lane — // 14.70 new labels per chunk — came out of code that had never been run against // a known answer. A scorer is not neutral infrastructure: if newLabelsPerChunk // were off by one chunk, or dominantSpeakerShare divided by the wrong total, a // lane would have been killed (or shipped) on an artifact. Each fixture below is // small enough that the expected value is obvious by inspection, which is the // only kind of test that can catch a wrong metric — comparing against whatever // it printed the first time cannot. // // The three shapes the plan names, plus the guards that were only ever asserted // in a comment: // perfect identity one cast, established in chunk 0, never added to // fresh cast per chunk the measured failure // degenerate single label a PERFECT headline score that is worthless import { strict as assert } from "node:assert"; import test from "node:test"; import { GENERIC_LABEL_RE, chunkRangesFor, isTranscriptTextLabel, nearDuplicate, scoreRecord, transcriptHaystack, type ScoreTarget, } from "./attributionScore"; import type { AttributionRecord, AttributionSegment } from "./attribution"; import type { Cue } from "./vtt"; // Four chunks, and they OVERLAP — because the real ones do. At maxCues 600 with // DIGEST_OVERLAP_CUES 40 the measured ranges on the probe videos run // 0.1-21.0 / 19.5-40.9 / 39.3-60.8 minutes: each chunk starts before the // previous one ends. A fixture with touching boundaries would be testing a // chunker that does not exist, and would put every segment that starts on a // boundary into the earlier chunk by accident rather than by the rule. const RANGES = [ { start: 0, end: 100 }, { start: 90, end: 200 }, { start: 190, end: 300 }, { start: 290, end: 400 }, ]; function record( speakers: { label: string; seconds: number }[], segments: AttributionSegment[], chunks?: number, ): AttributionRecord { return { videoId: "v", generatedAt: "2026-08-08T00:00:00.000Z", speakers: speakers.map((s, index) => ({ index, label: s.label, seconds: s.seconds })), segments, provenance: { method: "text-only", appId: "ollama", model: "qwen2.5:7b", promptVersion: 1, generatedAt: "2026-08-08T00:00:00.000Z", ...(chunks !== undefined ? { chunks, chunksOk: chunks } : {}), }, }; } const target = (over: Partial = {}): ScoreTarget => ({ chunkRanges: RANGES, durationSeconds: 400, ...over, }); test("perfect identity: a cast fixed in chunk 0 scores 0 new labels per chunk", () => { // Two speakers alternating, both present in all four chunks. Both first appear // in chunk 0, so NOTHING is new after it: 0 / (4 - 1) = 0. Every segment start // sits strictly inside one chunk, away from the overlap. const m = scoreRecord( record( [ { label: "Jane Doe", seconds: 160 }, { label: "John Smith", seconds: 160 }, ], [ { start: 10, end: 50, speaker: 0 }, { start: 50, end: 90, speaker: 1 }, { start: 110, end: 150, speaker: 0 }, { start: 150, end: 190, speaker: 1 }, { start: 210, end: 250, speaker: 0 }, { start: 250, end: 290, speaker: 1 }, { start: 310, end: 350, speaker: 0 }, { start: 350, end: 390, speaker: 1 }, ], 4, ), target(), ); assert.equal(m.newLabelsPerChunk, 0); assert.deepEqual(m.labelFirstChunk, [0, 0]); assert.deepEqual(m.labelChunkCount, [4, 4]); assert.equal(m.singletonLabelRate, 0); // 160 / 320 — read WITH the headline: this is what a good record looks like, // and it is what tells the degenerate case below apart from this one. assert.equal(m.dominantSpeakerShare, 0.5); assert.equal(m.markFreeChunks, 0); assert.equal(m.chunkReconstructionOk, true); }); test("fresh cast per chunk: three new labels over three later chunks scores 1.00", () => { // One label per chunk, each first seen in its own chunk. Three of the four are // new after chunk 0, over 3 later chunks: 3 / 3 = 1. const m = scoreRecord( record( [ { label: "Alice", seconds: 80 }, { label: "Bob", seconds: 80 }, { label: "Carol", seconds: 80 }, { label: "Dave", seconds: 80 }, ], [ { start: 10, end: 90, speaker: 0 }, { start: 110, end: 190, speaker: 1 }, { start: 210, end: 290, speaker: 2 }, { start: 310, end: 390, speaker: 3 }, ], 4, ), target(), ); assert.equal(m.newLabelsPerChunk, 1); assert.deepEqual(m.labelFirstChunk, [0, 1, 2, 3]); // Every label appears in exactly one chunk — the signature of a reinvented // cast, and why singletonLabelRate is read beside the headline. assert.equal(m.singletonLabelRate, 1); assert.equal(m.dominantSpeakerShare, 0.25); }); test("a label established in the overlap counts to the EARLIER chunk", () => { // Seconds 90-100 belong to both chunk 0 and chunk 1. The chunk that // ESTABLISHED a label is the earlier one — otherwise a cast that never changed // would score as freshly invented every time a speaker happened to resume // inside a seam, and the headline would indict the model for the chunker's // overlap. const m = scoreRecord( record( [ { label: "Jane Doe", seconds: 100 }, { label: "John Smith", seconds: 100 }, ], [ { start: 10, end: 90, speaker: 0 }, { start: 95, end: 190, speaker: 1 }, // starts INSIDE the 90-100 overlap ], 4, ), target(), ); assert.deepEqual(m.labelFirstChunk, [0, 0]); assert.equal(m.newLabelsPerChunk, 0); }); test("degenerate single label scores a PERFECT headline and is worthless", () => { // THE TRAP the headline cannot see on its own. One label covering the whole // video: nothing is ever "new", so newLabelsPerChunk is a flawless 0 — and // dominantSpeakerShare is 1.0, which is the only thing that gives it away. const m = scoreRecord( record([{ label: "Host", seconds: 400 }], [{ start: 0, end: 400, speaker: 0 }], 4), target(), ); assert.equal(m.newLabelsPerChunk, 0); assert.equal(m.dominantSpeakerShare, 1); assert.equal(m.labelsPerVideo, 1); // "Host" is generic — the third signal that this record names nobody. assert.equal(m.genericLabelRate, 1); assert.equal(m.genericSecondsShare, 1); }); test("a single-chunk video reports null, never 0, for cross-chunk metrics", () => { // There is no seam to cross. Reporting 0 here would be a harness reporting a // perfect score for something that never ran — and three of the pilot's eight // videos were single-chunk. const m = scoreRecord( record([{ label: "Jeremy", seconds: 100 }], [{ start: 0, end: 100, speaker: 0 }], 1), target({ chunkRanges: [{ start: 0, end: 100 }], durationSeconds: 100 }), ); assert.equal(m.newLabelsPerChunk, null); assert.equal(m.singletonLabelRate, null); }); test("a chunk-count mismatch invalidates every chunk-indexed number", () => { // The record says it chunked 7 ways; our reconstruction says 4. Any // chunk-indexed figure would be comparing different things. const m = scoreRecord( record([{ label: "Jane", seconds: 400 }], [{ start: 0, end: 400, speaker: 0 }], 7), target(), ); assert.equal(m.chunkReconstructionOk, false); assert.equal(m.newLabelsPerChunk, null); assert.equal(m.singletonLabelRate, null); assert.equal(m.markFreeChunks, null); assert.deepEqual(m.labelFirstChunk, [null]); }); // --------------------------------------------------------------------------- // transcriptTextLabelRate — the metric round 2 exists to drive to zero // --------------------------------------------------------------------------- const CUES: Cue[] = [ { start: 0, end: 10, text: "the general consensus were religion is subscribed to like" }, { start: 10, end: 20, text: "Nick the Dick says, in what way can the state bring charges" }, { start: 20, end: 30, text: "I really do not know about that" }, ]; test("transcript text is flagged, and a real name the transcript states is not", () => { const haystack = transcriptHaystack(CUES); // VERBATIM PROSE from the transcript: 9 words, appears in the text. Flagged. assert.equal( isTranscriptTextLabel("the general consensus were religion is subscribed to like", haystack), true, ); // THE DISCRIMINATION THAT MATTERS. "Nick the Dick" is a real name and the // transcript states it outright, so substring matching ALONE would flag it. // The word-count floor is what saves it. assert.equal(isTranscriptTextLabel("Nick the Dick", haystack), false); assert.equal(isTranscriptTextLabel("Bronca", haystack), false); assert.equal(isTranscriptTextLabel("Host", haystack), false); // A LINE BREAK is sufficient on its own — a person's name never contains one, // and the pilot's labels carried the cue's own internal newlines. assert.equal(isTranscriptTextLabel("I don't think 99% of atheist would\ncare", haystack), true); // Long, but NOT from this transcript: not transcript text. This is the half // that keeps the metric from being "is the label long". assert.equal( isTranscriptTextLabel("a sentence that never appears anywhere in it", haystack), false, ); }); test("newlines are collapsed on both sides before matching", () => { // The model is shown `[HH:MM:SS] text` LINES and echoes the line break back; // the cue text as stored has none. Without collapsing both sides this match // fails and the metric under-reports the failure it exists to measure. const haystack = transcriptHaystack(CUES); assert.equal( isTranscriptTextLabel("the general consensus were\nreligion is subscribed to like", haystack), true, ); }); test("transcriptTextLabelRate is null without cues, and a rate with them", () => { const rec = record( [ { label: "Nick the Dick", seconds: 100 }, { label: "the general consensus were religion is subscribed to like", seconds: 100 }, ], [ { start: 0, end: 100, speaker: 0 }, { start: 100, end: 200, speaker: 1 }, ], 4, ); // No cues -> null. NOT 0, which would read as a clean sheet. assert.equal(scoreRecord(rec, target()).transcriptTextLabelRate, null); const withCues = scoreRecord(rec, target({ cues: CUES })); assert.equal(withCues.transcriptTextLabelRate, 0.5); assert.deepEqual(withCues.transcriptTextLabels, [ "the general consensus were religion is subscribed to like", ]); }); test("truncatedLabelRate catches the maxLength cut without needing the cues", () => { // Exactly 60 characters — ATTRIBUTION_LABEL_MAX. One is a coincidence; a // POPULATION of them means the decoder was mid-sentence when the schema cut // it off. This is the evidence that survives when no cues are to hand. const sixty = "x".repeat(60); assert.equal(sixty.length, 60); const m = scoreRecord( record( [ { label: sixty, seconds: 100 }, { label: "Jane", seconds: 100 }, ], [ { start: 0, end: 100, speaker: 0 }, { start: 100, end: 200, speaker: 1 }, ], 4, ), target(), ); assert.equal(m.truncatedLabelRate, 0.5); }); // --------------------------------------------------------------------------- // The helpers the metrics are built from // --------------------------------------------------------------------------- test("nearDuplicate flags one person split in two, not two distinct people", () => { assert.equal(nearDuplicate("Jane", "Jane Doe"), true); // proper token-subset assert.equal(nearDuplicate("The Host", "host"), true); // case + article assert.equal(nearDuplicate("Jane Doe", "John Smith"), false); assert.equal(nearDuplicate("Host", "Guest"), false); }); test("GENERIC_LABEL_RE is anchored whole-string", () => { assert.equal(GENERIC_LABEL_RE.test("Host"), true); assert.equal(GENERIC_LABEL_RE.test("Caller 2"), true); assert.equal(GENERIC_LABEL_RE.test("the host"), true); // "Host" is generic; "Host Jane Doe" names somebody. assert.equal(GENERIC_LABEL_RE.test("Host Jane Doe"), false); }); test("chunkRangesFor reproduces the chunker the runner uses", () => { // 250 cues at maxCues 100 / overlap 40 -> step 60: [0,100) [60,160) [120,220) // [180,250]. Four chunks, and the ranges must be the CUE times, floored and // ceiled the way the runner does it. const cues: Cue[] = Array.from({ length: 250 }, (_, i) => ({ start: i * 10 + 0.4, end: i * 10 + 9.6, text: `cue ${i}`, })); const ranges = chunkRangesFor(cues, 100); assert.equal(ranges.length, 4); assert.deepEqual(ranges[0], { start: 0, end: Math.ceil(99 * 10 + 9.6) }); assert.deepEqual(ranges[1], { start: Math.floor(60 * 10 + 0.4), end: Math.ceil(159 * 10 + 9.6) }); // A transcript that fits in one chunk is one range, not zero. assert.equal(chunkRangesFor(cues.slice(0, 50), 100).length, 1); assert.equal(chunkRangesFor([], 100).length, 0); });