// Client-safe types and constants for the per-video speaker-ATTRIBUTION sidecar // — who is speaking, as opposed to diarization.json's anonymous "someone else is // speaking now". Mirrors diarization.ts exactly: server-only I/O lives in // attribution-server.ts, and the freshness comparator lives here so the runner // and the backfill registry share ONE definition of what is stale. // // TWO LANES PRODUCE THIS FILE, and which one produced a given record is the most // consequential field in it: // // "diarized" — diarization.json already carries globally-consistent cluster // indices, so the model only has to NAME N clusters from samples. // About one call per video, and the cross-chunk identity problem // below does not exist: the engine already solved it acoustically. // "text-only" — no diarization.json. The model has to read the transcript to // find speaker changes at all, which costs roughly the digest // sweep's chunk count (~194,000 calls corpus-wide) and has to // re-establish identity at every chunk seam. // // So the two are not interchangeable outputs of one job — one is cheaper AND // better — and the ordering rule below is what keeps them from undoing each // other. // // WHY THIS IS NOT A DIGEST SECTION. Folding a `speakers` section into the digest // prompt looks like it halves the GPU bill. It does not work: digests are // generated CHUNK-LOCAL (the measured default at digest PROMPT_VERSION 2), so // each chunk is labelled with no knowledge of the others — and the one property // attribution needs above all is that speaker 0 in chunk 1 is the same person in // chunk 30. Cross-chunk identity is the hard part of the text-only lane and the // reason this is its own artifact. Recorded here so the idea is not re-proposed // as an obvious saving. // // WHY THE FILENAME MATTERS: SUB_FILE_RE (lib/videoStatus.ts) claims any // `transcript..` as a subtitle track; sidecar() throws on such a name. // One person, as this record names them. `index` is what segments point at, so // it is stable within the record regardless of which lane produced it. export type AttributionSpeaker = { index: number; // What the model called them: a name when the transcript supports one // ("Jane Doe"), a role when it does not ("Host", "Caller 1"). label: string; // The diarization cluster this speaker IS. Set only by the diarized lane — // text-only has no clusters to point at, which is exactly the difference // between the two lanes. cluster?: number; // 0..1, as the model reported it. Kept because the filtering that eventually // consumes this must bias toward DROPPING on uncertainty, and it cannot do // that without a number to threshold on. Text-only is expected to be lower, // especially on auto-caption channels with no speaker turns at all. confidence?: number; // Seconds of attributed speech. Denormalized from `segments` so callers can // report talk-time shares without walking every segment. seconds?: number; }; // One contiguous stretch attributed to a speaker. Seconds from the start of the // video, the same base diarization.json and the cue stream use. export type AttributionSegment = { start: number; end: number; // Index into speakers[]. speaker: number; }; // A guard rejection, kept for the same reason DigestWarning is: the worst // outcome a generator has must leave a trace somewhere other than a job log // that rotates. export type AttributionWarning = { code: | "chunk-failed" | "bad-timestamp" | "unknown-cluster" | "empty" // The text-only lane emitted a turn whose timestamp lands outside the chunk // that produced it, so the turn was discarded. Counted per chunk, not per // turn, because it is not rare: the bake-off's third round measured 520 of // 1,705 turns (30%) dropped this way, and the shipped findTurns was dropping // them on the same condition with no count and no log line — invisible in // rounds 1 and 2 and in every record on disk. A discard rate that high is // the difference between "the model is doing well" and "the model is // producing four usable turns in five", so it has to leave a trace. | "out-of-range"; chunk?: number; detail?: string; }; export type AttributionMethod = "text-only" | "diarized"; // What produced a given attribution.json — the identity isAttributionFresh // compares against what the current configuration WOULD produce now. export type AttributionProvenance = { method: AttributionMethod; appId: string; // The model that ACTUALLY ran, as the engine reported it. model: string; // What the config asked for. Freshness compares THIS one, so a request for // "qwen2.5" resolving to "qwen2.5:7b" does not read as a model change and // trigger a needless regeneration (DigestProvenance.modelRequested's rule). modelRequested?: string; promptVersion: number; generatedAt: string; // DIARIZED LANE ONLY: the `generatedAt` of the diarization.json whose cluster // indices these speakers name. // // This is load-bearing, not bookkeeping. A cluster index is meaningful only // relative to one diarization run — re-run it at a different threshold and // cluster 3 is a different person, or nobody. Without this field a // re-diarization would silently leave the old names pointing at new clusters, // and nothing on disk could tell. With it, the naming goes stale exactly when // the thing it names is replaced. diarizationGeneratedAt?: string; // Which transcript the names were made from — "whisper" (ours) or "vtt" // (YouTube's, usually its ASR). // // Same kind of field as diarizationGeneratedAt above, for the same kind of // reason: the names are an assertion ABOUT a text, and replacing that text // makes the assertion unverified. The backfill's auto-transcribe hand-off makes // this a routine event rather than a rarity — a video whose only transcript was // YouTube ASR gets our own, with different wording, different timings and // possibly different speakers named. Without this field nothing on disk could // tell that had happened. transcriptSource?: AttributionTranscriptSource; // Text-only lane: how the video was sliced and how much of it survived. Same // pair digestVideo records, and for the same reason — one 500 from ollama // costs that chunk, not the 30-call video, and a partial result has to be // legible as partial afterwards. chunks?: number; chunksOk?: number; // Metered lanes only. costUsd?: number; durationMs?: number; }; // Which transcript the names were found in. // // Narrower than NormalizedTranscript.source on purpose: cues.json's `source` is // pickIndexTranscript's `kind` (normalizeTranscript.ts:133), which is only ever // "whisper" or "vtt" — never "live_chat". export type AttributionTranscriptSource = "whisper" | "vtt"; // Map a recorded/normalized source string onto the two values freshness knows, // or undefined for anything else. UNKNOWN MEANS DO NOT ASSERT: an unrecognised // value must not be turned into a comparison that invalidates a record. export function transcriptSourceOf( source: string | null | undefined, ): AttributionTranscriptSource | undefined { return source === "whisper" || source === "vtt" ? source : undefined; } export type AttributionRecord = { videoId: string; // ISO 8601, set when the sidecar is finalized. generatedAt: string; speakers: AttributionSpeaker[]; segments: AttributionSegment[]; warnings?: AttributionWarning[]; provenance: AttributionProvenance; }; export const ATTRIBUTION_FILENAME = "attribution.json"; // Bump for ANY change to the prompt text or the JSON schema in // attributionPrompt.ts. A record whose recorded promptVersion differs is // regenerated; one that matches is skipped, and that skip is what keeps a re-run // minutes long over a sample instead of weeks over the corpus. // // It lives HERE rather than in attributionPrompt.ts so settings.ts can import it // for the default without depending on the prompt module — the same arrangement // DEFAULT_DIARIZATION_THRESHOLD has, and for the same reason: this module has no // dependencies at all, so anything may import it. export const ATTRIBUTION_PROMPT_VERSION = 1; // THE ORDERING RULE, and the whole safety of two backfill kinds writing one file. // // Both `attribution-text` and `attribution-diarized` write attribution.json. The // diarized lane may overwrite a text-only record — that is an UPGRADE, and it is // the point of the second kind existing. The text lane must never overwrite a // diarized one — that is a DOWNGRADE, and it would silently replace one // call-per-video output grounded in acoustic clustering with a ~30-call // reconstruction that guesses at identity across chunk seams. // // Expressed as a rank so both the registry's state() and the runner's own guard // read from one definition rather than each spelling the comparison out. const METHOD_RANK: Record = { "text-only": 1, diarized: 2, }; // Would writing `candidate` over `existing` lose information? True only for a // strict downgrade — an equal method is a legitimate regeneration (a prompt bump, // a model change), not a downgrade. export function isAttributionDowngrade( existing: AttributionRecord | null, candidate: AttributionMethod, ): boolean { if (!existing) return false; const have = existing.provenance?.method; if (!have) return false; return METHOD_RANK[have] > METHOD_RANK[candidate]; } // What we WOULD produce for this video now, as an identity. The mirror of // DigestFreshnessTarget and DiarizationFreshnessTarget. export type AttributionFreshnessTarget = { method: AttributionMethod; appId: string; model: string; promptVersion: number; // Diarized lane only: the generatedAt of the diarization.json on disk right // now. Absent means "do not compare" — which is what the text-only lane wants, // since it never reads diarization at all. diarizationGeneratedAt?: string; // The source of the transcript on disk right now. Absent means "do not // compare" — a caller that has not read the video's files cannot assert it. transcriptSource?: AttributionTranscriptSource; }; // The identity the current configuration would produce, from a structurally // typed config rather than SiteSettings — so this module stays free of a // settings import and settings.ts can keep importing the version constant above. // // ONE definition, and the callers are the runner's short-circuit and the backfill // registry's state(). A comparator and the writer it guards disagreeing about // the identity is how a corpus ends up either regenerating forever or never. export function attributionTarget( cfg: { appId: string; model: string; promptVersion?: number; }, method: AttributionMethod, diarizationGeneratedAt?: string, ): AttributionFreshnessTarget { return { method, appId: cfg.appId, model: cfg.model, promptVersion: cfg.promptVersion ?? ATTRIBUTION_PROMPT_VERSION, ...(method === "diarized" && diarizationGeneratedAt ? { diarizationGeneratedAt } : {}), }; } // An absent recorded field is read as TODAY'S DEFAULT. Copied from digest.ts's // sameVariant and diarization.ts's sameThreshold for the reason both exist: the // compatibility rule has to be explicit, or adding a field later silently // invalidates every record on disk. function sameVersion(a: number | undefined, b: number | undefined): boolean { return ( (a ?? ATTRIBUTION_PROMPT_VERSION) === (b ?? ATTRIBUTION_PROMPT_VERSION) ); } // Is this sidecar what the current configuration would produce, FOR THIS LANE? // // `method` is compared like everything else, which has a consequence worth being // explicit about: a diarized record is not "fresh" against a text-only target, // and a text-only record is not "fresh" against a diarized one. That is correct // for the diarized kind (a text-only record is an upgrade opportunity — work) // but would be wrong for the text kind, where a diarized record means there is // nothing to do. The text kind therefore checks isAttributionDowngrade FIRST and // only asks this question when it is genuinely its own record it is looking at. // See lib/operations.ts. export function isAttributionFresh( record: AttributionRecord | null, target: AttributionFreshnessTarget, ): boolean { if (!record) return false; const p = record.provenance; if (!p) return false; if (p.method !== target.method) return false; if ( target.diarizationGeneratedAt !== undefined && p.diarizationGeneratedAt !== target.diarizationGeneratedAt ) { // The clusters these names point at have been replaced. See // AttributionProvenance.diarizationGeneratedAt. return false; } if ( target.transcriptSource !== undefined && p.transcriptSource !== undefined && p.transcriptSource !== target.transcriptSource ) { // The text these names were found in has been replaced — our own transcript // over YouTube's auto-captions, typically. // // Note the extra condition, which is the INVERSE of the diarization guard // just above: the RECORD must carry the field too. That is sameVersion's // rationale applied to an optional string rather than a number — a record // written before this field existed knows nothing about its transcript's // source, and reading that silence as "different" would invalidate every // attribution on disk the moment this shipped. Records written from now on // carry it and are compared. return false; } return ( p.appId === target.appId && (p.modelRequested ?? p.model) === target.model && sameVersion(p.promptVersion, target.promptVersion) ); } // Total attributed speech, in seconds. Unlike diarizationSpeechSeconds this does // NOT expect overlap — a segment stream that assigns one speaker per stretch is // what both lanes produce — but it is still tolerant of it, because a model that // emits overlapping ranges is a quality problem, not a crash. export function attributionSpeechSeconds(record: AttributionRecord): number { return record.segments.reduce( (sum, s) => sum + Math.max(0, s.end - s.start), 0, ); }