// Turn a stream of per-chunk speaker MARKS into the speakers + segments an // AttributionRecord carries. Pure — no I/O, no model — so it unit-tests, and so // the shipped runner and any bake-off harness assemble a record the SAME way. // // WHY THIS IS ITS OWN MODULE. This logic used to live inside findTurns in // controller/attributeOne.ts, where it was reachable only by running the shipped // lane against a real corpus and a real model. A harness that wants to score a // PROMPT VARIANT without writing sidecars must assemble a record too — and if it // re-implemented the roster, the mark-collapse or the seconds denormalization, // the two would drift and every cross-harness comparison would be measuring the // assembler as much as the prompt. plans/FACTS.md already records one such bet // (a regex copy-pasted between digest-validate and the parser "so the two are // comparable") that has to keep being won; this one is won by construction. // // The three steps, in order, and each is load-bearing: // // intern — a normalized label -> a stable speaker index. Insertion order IS // speaker order, so speaker 0 is the first voice heard. // marks -> segments. Each mark runs until the speaker next CHANGES. // seconds — denormalized onto the speakers, so a caller can report talk-time // shares without walking every segment. import type { AttributionSegment, AttributionSpeaker } from "./attribution"; // One "the speaker is now X, from this second" observation, before it is known // how long X speaks for. `at` is seconds from the start of the video. export type SpeakerMark = { at: number; speaker: number }; // "The Host:" and "the host" are the same person. Deliberately conservative — // it folds case, surrounding punctuation and internal whitespace, and nothing // else. Fuzzier matching ("Jane" == "Jane Doe") would merge two people on a // guess, and the asymmetry runs the other way: a split speaker is a visible // quality problem, a merged one silently attributes words to the wrong person. export function normalizeLabel(label: string): string { return label .trim() .replace(/\s+/g, " ") .replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, "") .toLowerCase(); } // The roster carried across chunk seams. `labels` is what gets shown back to the // model as knownSpeakers, so it holds the ORIGINAL casing while lookup is by the // normalized key. export type SpeakerRoster = { // Original-cased labels, in first-seen order. Index into this IS the speaker // index used by every mark and segment. readonly labels: string[]; // The index for `label`, adding it if this is the first time it is seen. intern(label: string): number; // The index for `label` if already known, else undefined. Used by the // closed-cast variant, which must NOT mint an index for a label outside the // cast it declared. lookup(label: string): number | undefined; }; export function createSpeakerRoster(initial: string[] = []): SpeakerRoster { const byKey = new Map(); const labels: string[] = []; const intern = (label: string): number => { const key = normalizeLabel(label); const existing = byKey.get(key); if (existing !== undefined) return existing; const index = labels.length; byKey.set(key, index); labels.push(label.trim()); return index; }; for (const label of initial) intern(label); return { labels, intern, lookup: (label: string) => byKey.get(normalizeLabel(label)), }; } // Where the speaker next changes after mark `i`, or the end of the video. function nextChange( marks: SpeakerMark[], i: number, speaker: number, videoEnd: number, ): number { for (let j = i + 1; j < marks.length; j++) { if (marks[j].speaker !== speaker) return marks[j].at; } return videoEnd; } // Marks -> segments. Each mark runs until the next one; the last runs to the // end of the transcript. Duplicates at a seam (the 40-cue overlap means two // chunks see the same stretch) collapse because a mark that does not CHANGE the // speaker is not a boundary. // // Sorts a COPY: the caller's array is often the accumulator it is still logging // about, and mutating an argument in a pure module is the kind of surprise that // only shows up in the second caller. export function marksToSegments( marks: SpeakerMark[], videoEnd: number, ): AttributionSegment[] { const sorted = [...marks].sort((a, b) => a.at - b.at || a.speaker - b.speaker); const segments: AttributionSegment[] = []; for (let i = 0; i < sorted.length; i++) { const m = sorted[i]; if (segments.length > 0 && segments[segments.length - 1].speaker === m.speaker) { continue; } const end = nextChange(sorted, i, m.speaker, videoEnd); // A zero-length segment is a model emitting two marks on the same second; // it is noise, not a turn. if (end > m.at) segments.push({ start: m.at, end, speaker: m.speaker }); } return segments; } // The speakers array, with `seconds` denormalized off the segments. // // NO confidence field. The text-only lane has nothing honest to put here: the // model was not asked for one (it is answering "where does the speaker change", // not "how sure are you who this is"), and inventing a number would be exactly // the overclaiming PLAN.md warns against. A consumer that wants to filter on // confidence should filter on `method` instead. export function speakersFromSegments( labels: string[], segments: AttributionSegment[], ): AttributionSpeaker[] { const seconds = new Map(); for (const s of segments) { seconds.set(s.speaker, (seconds.get(s.speaker) ?? 0) + (s.end - s.start)); } return labels.map((label, index) => ({ index, label, seconds: Math.round(seconds.get(index) ?? 0), })); } // The whole assembly, in one call — what both the runner and the bake-off use, // so neither can drift from the other. // // `videoEnd` is the end of the transcript, extended to cover any mark that lands // past it: a segment that starts after the video ends would otherwise be // dropped for being zero-length, silently losing the last speaker. export function assembleTurns(args: { labels: string[]; marks: SpeakerMark[]; transcriptEnd: number; }): { speakers: AttributionSpeaker[]; segments: AttributionSegment[] } { if (args.labels.length === 0) return { speakers: [], segments: [] }; const lastMark = args.marks.reduce((max, m) => Math.max(max, m.at), 0); const videoEnd = Math.max(args.transcriptEnd, lastMark); const segments = marksToSegments(args.marks, videoEnd); return { speakers: speakersFromSegments(args.labels, segments), segments }; }