import { test } from "node:test"; import assert from "node:assert/strict"; import { ATTRIBUTION_PROMPT_VERSION, attributionTarget, isAttributionDowngrade, isAttributionFresh, attributionSpeechSeconds, transcriptSourceOf, type AttributionRecord, } from "./attribution"; import { attributionStatus } from "./attributionStatus"; import { isUselessSpeakerLabel, selectClusterSamples, speakerSchema, turnSchema, } from "./attributionPrompt"; import { defaultAttribution, sanitizeAttribution, } from "./settings"; import type { DiarizationTurn } from "./diarization"; import type { Cue } from "./vtt"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/attribution.test.ts const CURRENT = { appId: "ollama-direct", model: "qwen2.5:7b", promptVersion: ATTRIBUTION_PROMPT_VERSION, }; function record( over: Partial = {}, ): AttributionRecord { return { videoId: "vid1", generatedAt: "2026-08-07T00:00:00.000Z", speakers: [{ index: 0, label: "Host" }], segments: [{ start: 0, end: 10, speaker: 0 }], provenance: { method: "diarized", appId: CURRENT.appId, model: "qwen2.5:7b-instruct", modelRequested: CURRENT.model, promptVersion: ATTRIBUTION_PROMPT_VERSION, generatedAt: "2026-08-07T00:00:00.000Z", ...over, }, }; } // --------------------------------------------------------------------------- // Freshness // --------------------------------------------------------------------------- test("fresh: the record matches the identity we would produce now", () => { assert.equal( isAttributionFresh(record(), attributionTarget(CURRENT, "diarized")), true, ); }); test("the REQUESTED model is compared, not the one the engine reported", () => { // "qwen2.5:7b" resolving to "qwen2.5:7b-instruct" is not a model change, and // treating it as one would regenerate the whole corpus on every run. Same rule // as DigestProvenance.modelRequested. assert.equal( isAttributionFresh(record(), attributionTarget(CURRENT, "diarized")), true, ); assert.equal( isAttributionFresh( record(), attributionTarget({ ...CURRENT, model: "llama3:8b" }, "diarized"), ), false, ); }); test("stale: a different app or prompt generation is work", () => { assert.equal( isAttributionFresh( record(), attributionTarget({ ...CURRENT, appId: "claude-code" }, "diarized"), ), false, ); assert.equal( isAttributionFresh( record(), attributionTarget( { ...CURRENT, promptVersion: ATTRIBUTION_PROMPT_VERSION + 1 }, "diarized", ), ), false, ); }); // THE COMPATIBILITY RULE. Without it, adding a field to the provenance would // mark every record on disk stale at once — the trap lib/digest.ts's sameVariant // and lib/diarization.ts's sameThreshold both exist to avoid. test("an absent recorded promptVersion compares equal to today's default", () => { const old = record(); delete (old.provenance as { promptVersion?: number }).promptVersion; assert.equal( isAttributionFresh(old, attributionTarget(CURRENT, "diarized")), true, ); // ...equal to the DEFAULT specifically, not to anything: with a bumped // generation configured, the same record is stale. assert.equal( isAttributionFresh( old, attributionTarget( { ...CURRENT, promptVersion: ATTRIBUTION_PROMPT_VERSION + 1 }, "diarized", ), ), false, ); }); test("a record from the other lane is never fresh for this one", () => { assert.equal( isAttributionFresh( record({ method: "text-only" }), attributionTarget(CURRENT, "diarized"), ), false, ); assert.equal( isAttributionFresh(record(), attributionTarget(CURRENT, "text-only")), false, ); }); // A cluster index means nothing except relative to the diarization run that // produced it: re-run at a different threshold and cluster 3 is a different // person, or nobody. Without this the names would silently point at new clusters. test("re-diarizing invalidates the names that pointed at the old clusters", () => { const named = record({ diarizationGeneratedAt: "2026-08-01T00:00:00.000Z" }); assert.equal( isAttributionFresh(named, { ...attributionTarget(CURRENT, "diarized"), diarizationGeneratedAt: "2026-08-01T00:00:00.000Z", }), true, ); assert.equal( isAttributionFresh(named, { ...attributionTarget(CURRENT, "diarized"), diarizationGeneratedAt: "2026-08-09T00:00:00.000Z", }), false, ); // The text-only lane never reads diarization, so its target carries no such // value and the comparison must not fire at all. assert.equal( attributionTarget(CURRENT, "text-only").diarizationGeneratedAt, undefined, ); }); // The names are an assertion ABOUT a text. Replace the text — our own whisper // transcript over YouTube's auto-captions, which the backfill's auto-transcribe // hand-off now makes routine — and the assertion is unverified: different // wording, different timings, possibly different speakers named. test("our own transcript replacing the auto-captions invalidates the names made from them", () => { const fromVtt = record({ transcriptSource: "vtt" }); assert.equal( isAttributionFresh(fromVtt, { ...attributionTarget(CURRENT, "diarized"), transcriptSource: "whisper", }), false, ); // Same source: nothing changed, no regeneration. assert.equal( isAttributionFresh(record({ transcriptSource: "whisper" }), { ...attributionTarget(CURRENT, "diarized"), transcriptSource: "whisper", }), true, ); }); test("a record written before transcriptSource existed is not invalidated by it", () => { // The INVERSE of the diarization guard, and deliberately so: that one fires on // the target alone, this one requires the record to carry the field too. A // record on disk knows nothing about its transcript's source, and reading that // silence as "different" would invalidate every attribution in the corpus the // moment the field shipped. Same rule as sameVersion's absent-means-default. assert.equal( isAttributionFresh(record(), { ...attributionTarget(CURRENT, "diarized"), transcriptSource: "whisper", }), true, ); }); test("a target that does not know the source does not compare it", () => { // A caller that has not read the video's files cannot assert anything. assert.equal( attributionTarget(CURRENT, "text-only").transcriptSource, undefined, ); assert.equal( isAttributionFresh( record({ transcriptSource: "vtt" }), attributionTarget(CURRENT, "diarized"), ), true, ); }); test("transcriptSourceOf asserts nothing about a source it does not know", () => { assert.equal(transcriptSourceOf("whisper"), "whisper"); assert.equal(transcriptSourceOf("vtt"), "vtt"); // cues.json's source is pickIndexTranscript's kind, so live_chat never // reaches attribution — but "unknown means do not assert" is the rule, not an // accident of which values happen to occur. assert.equal(transcriptSourceOf("live_chat"), undefined); assert.equal(transcriptSourceOf(null), undefined); assert.equal(transcriptSourceOf(undefined), undefined); }); test("no record is never fresh", () => { assert.equal( isAttributionFresh(null, attributionTarget(CURRENT, "diarized")), false, ); }); // --------------------------------------------------------------------------- // The ordering rule — the whole safety of two kinds writing one file // --------------------------------------------------------------------------- test("diarized may overwrite text-only; text-only may never overwrite diarized", () => { const textOnly = record({ method: "text-only" }); const diarized = record(); // The UPGRADE. This is the point of the second kind existing. assert.equal(isAttributionDowngrade(textOnly, "diarized"), false); // The DOWNGRADE. ~30 calls of guessed identity must not replace one call // grounded in acoustic clustering. assert.equal(isAttributionDowngrade(diarized, "text-only"), true); // An EQUAL method is a regeneration (a prompt bump, a model change), not a // downgrade — otherwise nothing could ever be redone. assert.equal(isAttributionDowngrade(diarized, "diarized"), false); assert.equal(isAttributionDowngrade(textOnly, "text-only"), false); // Nothing on disk blocks nothing. assert.equal(isAttributionDowngrade(null, "text-only"), false); }); // --------------------------------------------------------------------------- // Status // --------------------------------------------------------------------------- test("status keeps the two lanes apart, and only claims stale when it can know", () => { assert.equal(attributionStatus(null), "none"); assert.equal(attributionStatus(record()), "diarized"); assert.equal(attributionStatus(record({ method: "text-only" })), "text-only"); // With a target, staleness is knowable. assert.equal( attributionStatus( record(), attributionTarget({ ...CURRENT, model: "llama3:8b" }, "diarized"), ), "stale", ); // WITHOUT one — an exported viewer, which has no settings and no engine — // staleness is unknowable, so the honest answer is the method recorded. assert.equal(attributionStatus(record()), "diarized"); }); // --------------------------------------------------------------------------- // Cluster sampling — what the diarized lane's one call actually costs // --------------------------------------------------------------------------- function cue(start: number, end: number, text: string): Cue { return { start, end, text } as Cue; } test("clusters are ranked by talk time, and the noise floor drops fragments", () => { const turns: DiarizationTurn[] = [ { start: 0, end: 100, speaker: 1 }, { start: 100, end: 160, speaker: 2 }, // 0.5s of crosstalk out of 160.5 — under the 1% floor, and naming it would // spend prompt on nobody. Diarization over-splits; this is the direction // that biases toward dropping on uncertainty. { start: 160, end: 160.5, speaker: 3 }, ]; const cues = [ cue(0, 50, "welcome back to the show"), cue(50, 100, "today we are talking about the filing"), cue(100, 160, "thanks for having me on"), cue(160, 160.5, "mm"), ]; const picked = selectClusterSamples(turns, cues); assert.deepEqual( picked.map((c) => c.cluster), [1, 2], ); assert.equal(picked[0].seconds, 100); assert.ok(picked[0].share > picked[1].share); assert.ok(picked[0].samples.length > 0); assert.ok(picked[0].samples[0].text.includes("welcome back")); }); test("the speaker cap holds however badly the diarizer over-split", () => { // Measured on a real file: 29 clusters before the threshold change, 13 after. // The cap is what keeps one call one call. const turns: DiarizationTurn[] = Array.from({ length: 40 }, (_, i) => ({ start: i * 10, end: i * 10 + 10, speaker: i, })); const cues = turns.map((t, i) => cue(t.start, t.end, `line ${i}`)); assert.equal(selectClusterSamples(turns, cues).length, 12); }); test("a diarization with no speech yields nothing to name", () => { assert.deepEqual(selectClusterSamples([], []), []); assert.deepEqual( selectClusterSamples([{ start: 5, end: 5, speaker: 0 }], []), [], ); }); // --------------------------------------------------------------------------- // Schemas + label guards // --------------------------------------------------------------------------- test("the cluster field is an enum of the clusters that actually exist", () => { // An ENUM, not an integer range: numeric minimum/maximum is widely dropped by // schema->grammar conversion, and a hallucinated cluster index would attribute // speech to a speaker who does not exist. const schema = speakerSchema([0, 3, 7]) as Record; const items = (schema.properties as never as Record).speakers as never as { items: { properties: { cluster: { enum: number[] } } }; maxItems: number; }; assert.deepEqual(items.items.properties.cluster.enum, [0, 3, 7]); assert.equal(items.maxItems, 3); }); test("the turn schema pins timestamps with the same regex the digest measured", () => { const schema = turnSchema() as Record; const turns = (schema.properties as never as Record).turns as never as { items: { properties: { start: { pattern: string } } }; }; // Two digits per field, so ":00:27" and "1:2:3" are rejected by the decoder. assert.equal(turns.items.properties.start.pattern, "^[0-9][0-9]:[0-9][0-9]:[0-9][0-9]$"); }); test('"Speaker 1" is rejected: the number is already known', () => { assert.equal(isUselessSpeakerLabel("Speaker 1"), true); assert.equal(isUselessSpeakerLabel("speaker"), true); assert.equal(isUselessSpeakerLabel("Unknown"), true); assert.equal(isUselessSpeakerLabel("Voice 2"), true); assert.equal(isUselessSpeakerLabel("a"), true); assert.equal(isUselessSpeakerLabel("Host"), false); assert.equal(isUselessSpeakerLabel("Jane Doe"), false); // Not a false positive: a real name that merely contains the word. assert.equal(isUselessSpeakerLabel("Speaker of the House"), false); }); test("speech seconds sum the attributed segments", () => { assert.equal(attributionSpeechSeconds(record()), 10); }); // --------------------------------------------------------------------------- // Settings // --------------------------------------------------------------------------- test("nothing is armed by default", () => { const d = defaultAttribution(); assert.equal(d.enabled, false); // Both lanes off UNDER the master switch, so turning the feature on to look at // it cannot start a ~194,000-call corpus sweep. assert.equal(d.diarizedEnabled, false); assert.equal(d.textOnlyEnabled, false); }); test("promptVersion is floored at the shipped constant, never pinned below it", () => { // Pinning freshness to a superseded generation freezes that generation's // output into the corpus, indistinguishable from the current prompt's — the // exact trap digestPrompt.ts's version 1 -> 2 note documents. assert.equal( sanitizeAttribution({ promptVersion: 0 }).promptVersion, ATTRIBUTION_PROMPT_VERSION, ); assert.equal( sanitizeAttribution({ promptVersion: -5 }).promptVersion, ATTRIBUTION_PROMPT_VERSION, ); // Raising it IS allowed: that is how an operator forces a corpus-wide redo // after a prompt tweak, without a code change. assert.equal( sanitizeAttribution({ promptVersion: ATTRIBUTION_PROMPT_VERSION + 3 }) .promptVersion, ATTRIBUTION_PROMPT_VERSION + 3, ); assert.equal( sanitizeAttribution({ promptVersion: "nonsense" }).promptVersion, ATTRIBUTION_PROMPT_VERSION, ); }); test("an empty model override survives sanitization", () => { // Empty is meaningful — "use the app's own model" — so it must not revert to // a default that only looks the same. assert.equal(sanitizeAttribution({ model: " " }).model, ""); assert.equal(sanitizeAttribution({ model: " llama3:8b " }).model, "llama3:8b"); assert.equal(sanitizeAttribution(null).model, ""); });