// Marks -> segments -> speakers, pinned by hand. // // This logic was extracted OUT of findTurns in controller/attributeOne.ts, where // the only way to exercise it was to run the shipped lane against a real corpus // and a real model. The extraction is only safe if the behaviour is unchanged, // so these assert the specific properties the original comments claimed and // nothing ever checked: the seam collapse, the zero-length drop, and the // insertion-order-is-speaker-order rule. import { strict as assert } from "node:assert"; import test from "node:test"; import { assembleTurns, createSpeakerRoster, marksToSegments, normalizeLabel, speakersFromSegments, } from "./attributionTurns"; test("the roster folds case, punctuation and whitespace — and nothing more", () => { assert.equal(normalizeLabel(" The Host: "), "the host"); assert.equal(normalizeLabel("The Host"), "the host"); assert.equal(normalizeLabel("«Jane Doe»"), "jane doe"); // Deliberately NOT merged: fuzzier matching would attribute one person's words // to another on a guess, and a merge is the silent failure. assert.notEqual(normalizeLabel("Jane"), normalizeLabel("Jane Doe")); }); test("insertion order IS speaker order, and re-interning is stable", () => { const roster = createSpeakerRoster(); assert.equal(roster.intern("Jane Doe"), 0); assert.equal(roster.intern("John Smith"), 1); // Same person, different casing and trailing punctuation -> same index. assert.equal(roster.intern("jane doe"), 0); assert.equal(roster.intern("The Host"), 2); // The label list keeps the ORIGINAL casing, because it is what gets shown // back to the model as knownSpeakers. assert.deepEqual(roster.labels, ["Jane Doe", "John Smith", "The Host"]); }); test("lookup does not mint an index for an unknown label", () => { // The closed-cast variant needs this: a label outside the declared cast must // not silently become a new speaker, which is the whole failure it exists to // make unrepresentable. const roster = createSpeakerRoster(["Jane Doe", "John Smith"]); assert.equal(roster.lookup("JANE DOE"), 0); assert.equal(roster.lookup("Nobody"), undefined); assert.deepEqual(roster.labels, ["Jane Doe", "John Smith"]); }); test("a mark that does not CHANGE the speaker is not a boundary", () => { // The 40-cue overlap means two chunks see the same stretch and both mark it. // The duplicate must collapse rather than cutting the segment in two. const segments = marksToSegments( [ { at: 0, speaker: 0 }, { at: 10, speaker: 0 }, // seam duplicate { at: 20, speaker: 1 }, { at: 30, speaker: 1 }, // seam duplicate { at: 40, speaker: 0 }, ], 100, ); assert.deepEqual(segments, [ { start: 0, end: 20, speaker: 0 }, { start: 20, end: 40, speaker: 1 }, { start: 40, end: 100, speaker: 0 }, ]); }); test("marks are sorted, and the caller's array is not mutated", () => { const marks = [ { at: 30, speaker: 1 }, { at: 0, speaker: 0 }, ]; const segments = marksToSegments(marks, 60); assert.deepEqual(segments, [ { start: 0, end: 30, speaker: 0 }, { start: 30, end: 60, speaker: 1 }, ]); // A pure module that reorders its argument is a surprise that only shows up // in the second caller. assert.deepEqual(marks, [ { at: 30, speaker: 1 }, { at: 0, speaker: 0 }, ]); }); test("a zero-length segment is dropped as noise", () => { // Two marks on the same second is a model stuttering, not a turn. const segments = marksToSegments( [ { at: 50, speaker: 0 }, { at: 50, speaker: 1 }, ], 50, ); // Speaker 0 runs 50 -> 50 (dropped); speaker 1 runs 50 -> videoEnd 50 // (dropped too). Nothing survives, which is correct: there is no duration. assert.deepEqual(segments, []); }); test("seconds are denormalized onto the speakers", () => { const speakers = speakersFromSegments( ["Jane", "John"], [ { start: 0, end: 30, speaker: 0 }, { start: 30, end: 45, speaker: 1 }, { start: 45, end: 60, speaker: 0 }, ], ); assert.deepEqual(speakers, [ { index: 0, label: "Jane", seconds: 45 }, { index: 1, label: "John", seconds: 15 }, ]); // NO confidence field on either — the text lane was never asked for one, and // inventing a number would be overclaiming. assert.equal("confidence" in speakers[0], false); }); test("a label with no surviving segment still gets a speaker, at 0 seconds", () => { // It was interned, so it is part of the cast the model proposed; recording it // at 0 is how the harness can SEE a label that contributed nothing. const speakers = speakersFromSegments(["Jane", "Ghost"], [{ start: 0, end: 10, speaker: 0 }]); assert.deepEqual(speakers[1], { index: 1, label: "Ghost", seconds: 0 }); }); test("assembleTurns extends the end past a mark that lands beyond the transcript", () => { // Otherwise the last speaker's segment is zero-length and is silently dropped. const { speakers, segments } = assembleTurns({ labels: ["Jane", "John"], marks: [ { at: 0, speaker: 0 }, { at: 120, speaker: 1 }, ], transcriptEnd: 100, }); assert.deepEqual(segments, [ { start: 0, end: 120, speaker: 0 }, { start: 120, end: 120, speaker: 1 }, ].slice(0, 1)); assert.equal(speakers.length, 2); assert.equal(speakers[1].seconds, 0); }); test("no labels means no record at all", () => { assert.deepEqual(assembleTurns({ labels: [], marks: [], transcriptEnd: 100 }), { speakers: [], segments: [], }); });