// The prompts + JSON schemas the two attribution lanes are driven with. Pure — // no I/O — so it unit-tests and can be imported anywhere. Mirrors // digestPrompt.ts, and reuses its timestamp machinery rather than restating it: // HMS_PATTERN, HMS_RE, toHms and hmsToSeconds are the SAME contract, and a // second copy of a regex that a measured run proved fixes malformed output is // exactly the sort of drift that makes one lane quietly worse than the other. // // Bump ATTRIBUTION_PROMPT_VERSION (in attribution.ts) for any change here. // // THE TWO LANES ARE PRICED IN DIFFERENT UNITS, and the prompts are why: // // diarized — ONE call per video. diarization.json already carries globally // consistent cluster indices, so the model is only asked to put a // name to each cluster given samples of what it said. // text-only — ONE call per CHUNK, because finding the speaker changes at all // requires reading the transcript. On this corpus that is ~194,000 // calls, the same order as the digest sweep. // // Do not price either in seconds-per-audio-hour. Chunk density varies fourfold // across this corpus, which is documented in plans/STATE.md as the unit error // behind a retracted throughput headline. import { HMS_PATTERN, toHms } from "./digestPrompt"; import type { Cue } from "./vtt"; import type { DiarizationTurn } from "./diarization"; // How many distinct speakers we will ask a model to name for one video. // // Not a schema nicety — diarization OVER-SPLITS. Measured on this corpus: a // known two-person interview clustered to 6 at the shipped 0.9 threshold, and a // full real file produced 13 speakers with the host at 73% of talk time. Naming // all of them wastes prompt on clusters that are seconds of crosstalk, and a // long enum makes a constrained decoder's job harder for no gain. So: the // heaviest N by speech time get named, the rest are simply left unattributed — // which is the direction that biases toward dropping on uncertainty. export const ATTRIBUTION_MAX_SPEAKERS = 12; // Clusters under this share of attributed speech are not worth a name; they are // overwhelmingly crosstalk or an over-split fragment of a speaker already named. export const ATTRIBUTION_MIN_CLUSTER_SHARE = 0.01; // Utterances shown per cluster, and how much of each. Six short samples spread // across the video identifies a recurring speaker far better than one long // passage, and keeps the whole naming call inside a default 8k window even at // the 12-cluster cap. export const ATTRIBUTION_SAMPLES_PER_SPEAKER = 6; export const ATTRIBUTION_SAMPLE_MAX_CHARS = 220; // A label the model may not use. "Speaker 1" is what an unhelpful model falls // back to, and it is indistinguishable from the cluster index we already have. export const ATTRIBUTION_LABEL_MIN = 2; export const ATTRIBUTION_LABEL_MAX = 60; // --------------------------------------------------------------------------- // Diarized lane — name N clusters, one call // --------------------------------------------------------------------------- export type ClusterSample = { cluster: number; // Seconds of speech attributed to this cluster by the diarizer. seconds: number; // Share of all attributed speech, 0..1. Stated in the prompt because talk-time // is the single strongest cue for "this is the host". share: number; // Representative utterances, in video order. samples: { clock: string; text: string }[]; }; export function speakerSchema(clusters: number[]): Record { return { type: "object", properties: { speakers: { type: "array", minItems: 1, maxItems: Math.max(1, clusters.length), items: { type: "object", properties: { // An ENUM of the actual cluster indices, not an integer with a // range. Both engines convert a schema to a grammar, and an // enumerated literal alternation is the constraint that converts // reliably everywhere — numeric `minimum`/`maximum` is widely // ignored by grammar conversion, which would put us back to // validating a hallucinated cluster after the fact. cluster: { enum: clusters }, label: { type: "string", minLength: ATTRIBUTION_LABEL_MIN, maxLength: ATTRIBUTION_LABEL_MAX, }, confidence: { type: "number" }, }, required: ["cluster", "label", "confidence"], }, }, }, required: ["speakers"], }; } export const SPEAKER_SYSTEM_PROMPT = [ "You identify who is speaking in a transcript.", "You reply with JSON only, matching the provided schema exactly.", "Every label you write is in ENGLISH.", "You never invent a name: a label is either a name the transcript itself", "states, or a role you can justify from what that speaker says.", ].join(" "); export type SpeakerPromptInput = { title: string; channel: string; clusters: ClusterSample[]; contextNote?: string; }; export function buildSpeakerPrompt(input: SpeakerPromptInput): string { const lines: string[] = []; lines.push( `The transcript of "${input.title}" (${input.channel}) has been split by an audio`, `diarizer into ${input.clusters.length} speaker cluster(s). Each cluster is one voice.`, ); lines.push(""); lines.push( "Give each cluster a label. Prefer a real name when the transcript states one —", 'someone is introduced, greeted, or names themselves. When it does not, use a role', 'you can justify from the material ("Host", "Interviewer", "Caller"). Do NOT use', '"Speaker 1"-style labels: the number is already known and adds nothing.', ); lines.push(""); lines.push( "Set `confidence` between 0 and 1 for each label: 0.9 when the transcript names", "that person outright, 0.5 for a role you inferred, 0.2 or less when you are", "guessing. An honest low number is more useful than a confident wrong name.", ); if (input.contextNote?.trim()) { lines.push(""); lines.push("Context for this channel (use it for names and recurring people):"); lines.push(input.contextNote.trim()); } for (const c of input.clusters) { lines.push(""); lines.push( `Cluster ${c.cluster} — ${Math.round(c.share * 100)}% of the speech (${toHms(c.seconds)} total):`, ); for (const s of c.samples) lines.push(` [${s.clock}] ${s.text}`); } return lines.join("\n"); } // Pick the clusters worth naming, and representative samples for each. // // Pure and separate from the runner so the selection can be tested without a // model: it is the whole of what the diarized lane costs, and getting it wrong // (all samples from one minute, or every three-second fragment named) is the // difference between one good call and one useless one. export function selectClusterSamples( turns: DiarizationTurn[], cues: Cue[], opts: { maxSpeakers?: number; samplesPerSpeaker?: number; minShare?: number; } = {}, ): ClusterSample[] { const maxSpeakers = opts.maxSpeakers ?? ATTRIBUTION_MAX_SPEAKERS; const samplesPer = opts.samplesPerSpeaker ?? ATTRIBUTION_SAMPLES_PER_SPEAKER; const minShare = opts.minShare ?? ATTRIBUTION_MIN_CLUSTER_SHARE; const seconds = new Map(); for (const t of turns) { const d = Math.max(0, t.end - t.start); seconds.set(t.speaker, (seconds.get(t.speaker) ?? 0) + d); } const total = [...seconds.values()].reduce((a, b) => a + b, 0); if (total <= 0) return []; const ranked = [...seconds.entries()] .map(([cluster, s]) => ({ cluster, seconds: s, share: s / total })) .filter((c) => c.share >= minShare) .sort((a, b) => b.seconds - a.seconds || a.cluster - b.cluster) .slice(0, maxSpeakers); return ranked.map((c) => { // The cluster's own turns, longest first — a long turn carries enough words // to identify a voice, while the two-second ones are backchannel ("right", // "mm-hm") that identifies nobody. const mine = turns .filter((t) => t.speaker === c.cluster) .sort((a, b) => b.end - b.start - (a.end - a.start)); const picked: { clock: string; text: string }[] = []; const used = new Set(); for (const turn of mine) { if (picked.length >= samplesPer) break; const text = cuesInRange(cues, turn.start, turn.end, used); if (!text) continue; picked.push({ clock: toHms(turn.start), text }); } // Back in video order: a model reading samples chronologically can follow a // conversation, and an introduction almost always comes first. picked.sort((a, b) => a.clock.localeCompare(b.clock)); return { cluster: c.cluster, seconds: Math.round(c.seconds), share: c.share, samples: picked, }; }); } // The cue text overlapping [start, end], collapsed and capped. `used` stops two // turns of the same cluster from quoting the same cues back. function cuesInRange( cues: Cue[], start: number, end: number, used: Set, ): string { const parts: string[] = []; for (let i = 0; i < cues.length; i++) { const c = cues[i]; if (c.start >= end) break; if ((c.end || c.start) <= start) continue; if (used.has(i)) continue; const text = c.text.trim().replace(/\s+/g, " "); if (!text) continue; used.add(i); parts.push(text); if (parts.join(" ").length >= ATTRIBUTION_SAMPLE_MAX_CHARS) break; } return parts.join(" ").slice(0, ATTRIBUTION_SAMPLE_MAX_CHARS); } // --------------------------------------------------------------------------- // Text-only lane — find the speaker changes, one call per chunk // --------------------------------------------------------------------------- export function turnSchema(): Record { return { type: "object", properties: { turns: { type: "array", minItems: 1, items: { type: "object", properties: { // The SAME pin digestPrompt measured: two digits per field, so // ":00:27" and "1:2:3" are rejected by the decoder itself rather // than by a parser afterwards. start: { type: "string", pattern: HMS_PATTERN }, speaker: { type: "string", minLength: ATTRIBUTION_LABEL_MIN, maxLength: ATTRIBUTION_LABEL_MAX, }, }, required: ["start", "speaker"], }, }, }, required: ["turns"], }; } export const TURN_SYSTEM_PROMPT = [ "You mark where the speaker changes in a transcript.", "You reply with JSON only, matching the provided schema exactly.", "Every speaker label you write is in ENGLISH.", "You never invent a timestamp: every start you emit is copied from a", "[HH:MM:SS] marker that appears in the transcript you were given.", "You reuse a speaker label you have already been given whenever the same", "person is speaking, because the same person must have the same label", "everywhere in the video.", ].join(" "); export type TurnPromptInput = { title: string; channel: string; // The chunk's own range, in seconds. Stated in the prompt — one of the three // changes that eliminated out-of-range output for the digest lane. startSeconds: number; endSeconds: number; // The transcript slice, rendered as `[HH:MM:SS] text` lines. transcript: string; // Labels established by EARLIER chunks of this same video. // // THIS IS THE CROSS-CHUNK IDENTITY MECHANISM, and it is the whole reason the // text-only lane is hard. Each call sees one slice, so without carrying the // roster forward the model would invent a fresh cast every chunk and speaker 0 // in chunk 1 would have nothing to do with speaker 0 in chunk 30 — the one // property attribution needs above all others. Carrying it does not GUARANTEE // consistency (a model can still split one person across two labels), which is // why the diarized lane, where the clustering solved this acoustically, is // both cheaper and better. knownSpeakers?: string[]; contextNote?: string; }; export function buildTurnPrompt(input: TurnPromptInput): string { const from = toHms(input.startSeconds); const to = toHms(input.endSeconds); const lines: string[] = []; lines.push( `Below is one section of the transcript of "${input.title}" (${input.channel}).`, ); lines.push(""); lines.push(`This section covers ${from} to ${to}.`); lines.push( `Mark every point where the SPEAKER changes. EVERY start you emit MUST be between`, `${from} and ${to} inclusive, and MUST be copied from a [HH:MM:SS] marker in the`, "transcript below — do not compute or estimate one.", ); lines.push(""); lines.push( "Emit one entry for the first speaker of the section and one for each change", "after that. Do not emit an entry per line: a person speaking continuously for", "five minutes is ONE entry.", ); if (input.knownSpeakers && input.knownSpeakers.length > 0) { lines.push(""); lines.push( "These labels were already used earlier in THIS video. Reuse the exact label", "whenever the same person is speaking — the same person must not get two", "different labels:", ); for (const s of input.knownSpeakers) lines.push(` - ${s}`); lines.push( "Introduce a new label only when it is genuinely someone who has not spoken yet.", ); } lines.push(""); lines.push( "Label people by name when the transcript states one, otherwise by a role you can", 'justify from what they say ("Host", "Guest", "Caller"). Do NOT use "Speaker 1".', ); if (input.contextNote?.trim()) { lines.push(""); lines.push("Context for this channel (use it for names and recurring people):"); lines.push(input.contextNote.trim()); } lines.push(""); lines.push("Transcript section:"); lines.push(""); lines.push(input.transcript); return lines.join("\n"); } // --------------------------------------------------------------------------- // CLOSED-CAST variant — split the text lane in two, and make the failure // UNREPRESENTABLE // --------------------------------------------------------------------------- // // THE FAILURE THIS EXISTS TO FIX, stated precisely because the first write-up of // it was wrong. The 2026-08-08 pilot recorded the text lane collapsing "once the // prompt carries a knownSpeakers roster and grows". The per-chunk data says // otherwise: on destiny/5nmDzKB23OU, 15 of 16 labels first appear in CHUNK 0 — // where findTurns passes no roster at all — and on rekietalaw/EsZhaCfc8HQ the // collapse is intermittent (chunk 0 clean, 1 bad, 2 clean, 3 bad, 4-5 clean, // 6 catastrophic at 48 labels, 7 clean) rather than monotonic in prompt length. // The decisive evidence is the labels themselves: dozens sit at EXACTLY 60 // characters, the schema's own maxLength, several carrying the cue's internal // newline. The model was not misusing a roster. It was continuing the transcript // into the `speaker` field, and a free string could not stop it. // // So the fix is not a better instruction — it is a smaller LABEL SPACE: // // pass A cast discovery. ONE call per video over evenly-spaced excerpts: // "who speaks here?" This is the thing the model demonstrably does do // well — the diarized lane named 9 clusters cleanly with calibrated // confidences, and every single-chunk video in the pilot returned a // clean name ("Jeremy", "Donald Trump", "Stephen Colbert"). // pass B turn assignment, per chunk, with `speaker` an ENUM over that cast. // A transcript fragment cannot be emitted because it is not in the // grammar. // // This is the constraint speakerSchema() already proves works, applied to the // other lane: "an enumerated literal alternation is the constraint that converts // reliably everywhere". // // IT TARGETS COST AS WELL AS QUALITY, and that is the point. Engine time was // within 7% of wall time in the pilot, so 66.2 s/chunk is pure decode volume // (31,266 output tokens on one 13-chunk video). An enum token per turn instead // of a 60-character string collapses the decode. One change, both failures. // How many excerpts the cast-discovery call sees. Evenly spaced across the whole // video rather than clustered: a cast list built from the first ten minutes // misses everyone who arrives later, and on this corpus a chunk spans 21-34 // measured minutes so a video runs to several hours. 24 excerpts at // ATTRIBUTION_SAMPLE_MAX_CHARS is ~5k characters — comfortably inside a default // 8k window with the whole prompt. export const ATTRIBUTION_CAST_SAMPLES = 24; // The escape hatch in the closed cast. WITHOUT this the enum would force the // model to name somebody for every turn, and a forced choice among a wrong cast // is worse than no answer — the bias must run toward dropping on uncertainty, // the same direction selectClusterSamples takes when it leaves light clusters // unnamed. Turns landing here are discarded rather than attributed. export const ATTRIBUTION_UNKNOWN_SPEAKER = "Unknown"; // Ceiling on turns from ONE chunk. turnSchema() has no maxItems at all, which is // how a single chunk produced 48 new labels. // // Measured, not guessed: at maxCues 600 a chunk spans 21-34 minutes on the two // probe videos. 60 allows ~2 speaker changes per minute — well above the ~4 // turns/minute the ACOUSTIC diarizer found on a fast reaction video, and far // above the granularity the prompt actually asks for ("a person speaking // continuously for five minutes is ONE entry"). It is a ceiling on runaway // decode, not a target: minItems stays 1. export const ATTRIBUTION_MAX_TURNS_PER_CHUNK = 60; // Evenly-spaced excerpts from a transcript — the text lane's analogue of // selectClusterSamples, which picks samples per acoustic cluster. There are no // clusters here, so the only defensible spread is uniform across the video. // // Pure and separate from the runner for the same reason its diarized twin is: // this selection IS what pass A costs, and getting it wrong (every sample from // one minute) makes a cheap call a useless one. export function selectTranscriptSamples( cues: Cue[], n: number = ATTRIBUTION_CAST_SAMPLES, ): { clock: string; text: string }[] { const wanted = Math.max(1, Math.floor(n)); const usable = cues .map((c, i) => ({ c, i })) .filter(({ c }) => c.text.trim().length > 0); if (usable.length === 0) return []; const out: { clock: string; text: string }[] = []; const used = new Set(); // Anchor positions spread across the transcript. Each anchor then pulls // FORWARD through consecutive cues until the char budget is met, so a sample // is a readable passage rather than one clipped line — the same shape // cuesInRange produces for the diarized lane. for (let s = 0; s < wanted; s++) { const anchor = Math.floor((s * usable.length) / wanted); let at = anchor; while (at < usable.length && used.has(usable[at].i)) at++; if (at >= usable.length) continue; const parts: string[] = []; const startClock = toHms(usable[at].c.start); for (let k = at; k < usable.length; k++) { if (used.has(usable[k].i)) break; const text = usable[k].c.text.trim().replace(/\s+/g, " "); used.add(usable[k].i); parts.push(text); if (parts.join(" ").length >= ATTRIBUTION_SAMPLE_MAX_CHARS) break; } const text = parts.join(" ").slice(0, ATTRIBUTION_SAMPLE_MAX_CHARS); if (text) out.push({ clock: startClock, text }); } return out; } // Pass A's schema. `name` stays a free string — this call is the one place a // free string is CORRECT, because discovering an unknown name is the whole job. // It is also the cheap call: one per video, bounded by ATTRIBUTION_MAX_SPEAKERS. export function castSchema(): Record { return { type: "object", properties: { cast: { type: "array", minItems: 1, maxItems: ATTRIBUTION_MAX_SPEAKERS, items: { type: "object", properties: { name: { type: "string", minLength: ATTRIBUTION_LABEL_MIN, maxLength: ATTRIBUTION_LABEL_MAX, }, confidence: { type: "number" }, }, required: ["name", "confidence"], }, }, }, required: ["cast"], }; } export const CAST_SYSTEM_PROMPT = [ "You identify WHO SPEAKS in a transcript.", "You reply with JSON only, matching the provided schema exactly.", "Every name you write is in ENGLISH.", "You never invent a name: a name is either one the transcript itself states,", "or a role you can justify from what that person says.", "You never copy a sentence of the transcript into a name field. A name is a", "few words at most.", ].join(" "); export type CastPromptInput = { title: string; channel: string; samples: { clock: string; text: string }[]; contextNote?: string; }; export function buildCastPrompt(input: CastPromptInput): string { const lines: string[] = []; lines.push( `Below are excerpts taken at even intervals through the transcript of`, `"${input.title}" (${input.channel}).`, ); lines.push(""); lines.push( "List the people who SPEAK in this video. Prefer a real name when the transcript", 'states one — someone is introduced, greeted, or names themselves. When it does', 'not, use a role you can justify from the material ("Host", "Caller"). Do NOT use', '"Speaker 1"-style names.', ); lines.push(""); lines.push( "A name is a PERSON, not a sentence: at most a few words. Never copy transcript", "text into a name.", ); lines.push(""); lines.push( "List only people who actually speak. Somebody merely talked ABOUT is not in the", "cast. Prefer a short list you are sure of over a long speculative one — anyone", "you leave out is simply left unattributed, which is the safer answer.", ); lines.push(""); lines.push( "Set `confidence` between 0 and 1: 0.9 when the transcript names that person", "outright, 0.5 for a role you inferred, 0.2 or less when you are guessing. An", "honest low number is more useful than a confident wrong name.", ); if (input.contextNote?.trim()) { lines.push(""); lines.push("Context for this channel (use it for names and recurring people):"); lines.push(input.contextNote.trim()); } lines.push(""); lines.push("Excerpts:"); for (const s of input.samples) { lines.push(""); lines.push(` [${s.clock}] ${s.text}`); } return lines.join("\n"); } // Pass B's schema — the closed one. THIS is the change under test. // // Two constraints turnSchema() lacks, and each maps to one measured failure: // speaker: enum a transcript fragment is not in the grammar, so the // 14.70-new-labels-per-chunk failure cannot be expressed. // maxItems a chunk cannot emit 48 turns, so the 66.2 s/chunk decode // volume is bounded. export function turnSchemaClosed( cast: string[], maxTurns: number = ATTRIBUTION_MAX_TURNS_PER_CHUNK, ): Record { // De-duplicated, and with the escape hatch appended exactly once — a repeated // literal in an alternation is at best wasted grammar and at worst a // conversion error. const seen = new Set(); const options: string[] = []; for (const name of [...cast, ATTRIBUTION_UNKNOWN_SPEAKER]) { const trimmed = name.trim(); const key = trimmed.toLowerCase(); if (!trimmed || seen.has(key)) continue; seen.add(key); options.push(trimmed); } return { type: "object", properties: { turns: { type: "array", minItems: 1, maxItems: Math.max(1, Math.floor(maxTurns)), items: { type: "object", properties: { // The SAME pin digestPrompt measured: two digits per field. start: { type: "string", pattern: HMS_PATTERN }, speaker: { enum: options }, }, required: ["start", "speaker"], }, }, }, required: ["turns"], }; } export type ClosedTurnPromptInput = { title: string; channel: string; startSeconds: number; endSeconds: number; transcript: string; // The cast pass A found. NOT a roster that grows chunk by chunk — it is fixed // for the whole video before any chunk runs, which is what makes cross-chunk // identity a property of the SCHEMA rather than of the model's memory. cast: string[]; contextNote?: string; }; export function buildClosedTurnPrompt(input: ClosedTurnPromptInput): string { const from = toHms(input.startSeconds); const to = toHms(input.endSeconds); const lines: string[] = []; lines.push( `Below is one section of the transcript of "${input.title}" (${input.channel}).`, ); lines.push(""); lines.push(`This section covers ${from} to ${to}.`); lines.push( `Mark every point where the SPEAKER changes. EVERY start you emit MUST be between`, `${from} and ${to} inclusive, and MUST be copied from a [HH:MM:SS] marker in the`, "transcript below — do not compute or estimate one.", ); lines.push(""); lines.push( "Emit one entry for the first speaker of the section and one for each change", "after that. Do not emit an entry per line: a person speaking continuously for", "five minutes is ONE entry.", ); lines.push(""); lines.push("`speaker` MUST be one of these exact values, and nothing else:"); for (const c of input.cast) lines.push(` - ${c}`); lines.push(` - ${ATTRIBUTION_UNKNOWN_SPEAKER}`); lines.push(""); lines.push( `Use "${ATTRIBUTION_UNKNOWN_SPEAKER}" whenever you cannot tell which of them is`, "speaking. That is a normal answer, not a failure — an honest Unknown is better", "than a guess, because a wrong name puts words in someone's mouth.", ); if (input.contextNote?.trim()) { lines.push(""); lines.push("Context for this channel (use it for names and recurring people):"); lines.push(input.contextNote.trim()); } lines.push(""); lines.push("Transcript section:"); lines.push(""); lines.push(input.transcript); return lines.join("\n"); } // --------------------------------------------------------------------------- // A label the schema accepted but that carries no information. Rejected by the // parser rather than by the schema because a `not`/pattern-negation constraint // does not survive grammar conversion — the same reason the digest parser // re-checks HMS_RE that the schema already pinned. // // "unknown" is in here, which is what makes ATTRIBUTION_UNKNOWN_SPEAKER work as // an escape hatch for free: a turn the closed schema legitimately marks Unknown // is dropped by the same guard that rejects a useless label, so it never becomes // a speaker. const USELESS_LABEL_RE = /^(speaker|person|voice|unknown)\s*[0-9]*$/i; export function isUselessSpeakerLabel(label: string): boolean { const trimmed = label.trim(); if (trimmed.length < ATTRIBUTION_LABEL_MIN) return true; return USELESS_LABEL_RE.test(trimmed); }