// Client-safe types and constants for the per-video speaker-diarization // sidecar. Mirrors transcribeOutcome.ts: server-only I/O lives in // diarization-server.ts. // // WHY A SIDECAR AND NOT A FIELD ON `Cue`: speaker ranges are deliberately kept // out of the cue stream. Putting them inside cues would bump SCHEMA_VERSION // (controller/buildIndex.ts), invalidate the IndexedDB transcript cache // (components/transcriptStore.ts), force a re-emit of every transcripts page // (the SHA-1 skip in buildIndex.ts would miss everywhere), push against the // 25 MB archive cap, and inflate the digest token budget (lib/digestPrompt.ts // assumes ~10 tokens/cue). A sidecar costs none of that, and attribution can be // recomputed from it at any time. // // WHY THE FILENAME MATTERS: never name a sidecar `transcript..` — // SUB_FILE_RE in lib/videoStatus.ts claims it as a subtitle track, and sidecar() // throws on it. The constant lives beside the other video-dir filenames there. // One contiguous stretch of audio attributed to a single (anonymous) speaker // cluster. Times are seconds from the start of the audio. `speaker` is a // cluster INDEX, not an identity — naming clusters is a later, redoable pass. export type DiarizationTurn = { start: number; end: number; speaker: number; }; // What produced a given diarization.json. Recorded so a later attribution pass // can tell whether a file is worth re-running: swapping the engine, the // segmentation model, or the clustering threshold all change the output, and // only an explicit record makes that legible after the audio is gone. export type DiarizationEngine = { // Wrapper/engine id, e.g. "sherpa-onnx". engine: string; // Model identifiers, free-form per engine (basename is enough — the full path // is machine-specific and would make records non-portable across shards). segmentationModel?: string; embeddingModel?: string; // The single-model engines' model, kept separate from the segmentation/ // embedding PAIR above rather than overloading it. Sortformer is one GGUF and // has no embedding model at all, and writing it into `segmentationModel` would // make a sortformer record compare equal to a sherpa one that happened to share // a basename. model?: string; // Engine/library version string, when the engine reports one. version?: string; // Clustering threshold used. The single most consequential knob: it decides // how many speakers come out, and re-running with a different one is the most // likely reason to regenerate. threshold?: number; }; export type DiarizationRecord = { videoId: string; // ISO 8601, set when the sidecar is finalized. generatedAt: string; // Audio duration in seconds as the engine saw it. audioSeconds?: number; // Wall-clock diarization time in milliseconds. durationMs?: number; // Distinct speaker clusters found. Denormalized from `turns` so callers can // bucket/report without walking every turn. speakers: number; turns: DiarizationTurn[]; engine: DiarizationEngine; // Present only when the engine processed the audio in windows rather than // whole (long recordings — see scripts/diarize-sherpa.py for why). // // DELIBERATELY OUTSIDE `engine`, AND DELIBERATELY NOT COMPARED. The freshness // target below is built from settings alone and cannot know a given video's // duration, so a window field in the identity would mark every sidecar on disk // stale the day windowing shipped — for work that is unchanged. Keeping it out // here, rather than merely forgetting to compare it, is what stops a later // edit to isDiarizationFresh from sweeping it in by accident. Recorded because // it genuinely describes how the answer was produced, and a re-clustering pass // may want to know a seam existed. windowing?: { windows: number; windowSeconds: number; overlapSeconds: number; }; }; export const DIARIZATION_FILENAME = "diarization.json"; // The engine id scripts/diarize.mjs records when it runs its own default // (sherpa-onnx) rather than a `--engine` replacement. Named here so the // freshness comparator below and the wrapper agree on one spelling. export const DEFAULT_DIARIZATION_ENGINE = "sherpa-onnx"; // The ggml/Sortformer engine (scripts/build-sortformer.sh, scripts/ // diarize-sortformer.mjs). End-to-end rather than clustered: no segmentation + // embedding pair, no threshold, a hard ceiling of 4 speakers, and — because its // state is a fixed-size speaker cache — memory that is O(1) in recording length // rather than O(n^2) in segment count. export const SORTFORMER_DIARIZATION_ENGINE = "sortformer"; export type DiarizationEngineId = | typeof DEFAULT_DIARIZATION_ENGINE | typeof SORTFORMER_DIARIZATION_ENGINE; // Compute device for engines that have a choice. Only sortformer does; sherpa is // ONNX/CPU here (no Vulkan compute path on Linux for onnxruntime). export type DiarizationBackend = "vulkan" | "cpu"; export const DIARIZATION_ENGINE_IDS: readonly DiarizationEngineId[] = [ DEFAULT_DIARIZATION_ENGINE, SORTFORMER_DIARIZATION_ENGINE, ]; export const DIARIZATION_BACKENDS: readonly DiarizationBackend[] = [ "vulkan", "cpu", ]; // The clustering-threshold default, measured on this corpus (see // DiarizationSettings.threshold for the sweep that produced it). It lives here // rather than only in settings.ts because isDiarizationFresh needs it to // normalize an ABSENT recorded threshold, and this module must stay importable // from anywhere. settings.ts imports it, so there is still one source of truth. export const DEFAULT_DIARIZATION_THRESHOLD = 0.9; // What we WOULD produce for this video now, as an identity. The mirror of // DigestFreshnessTarget in lib/digest.ts, and deliberately the same shape of // idea: a sidecar is stale when its recorded provenance differs from this, not // when it is old. export type DiarizationFreshnessTarget = { // OPTIONAL, and compared only when present — which now means "only when the // configured engine is not the default". // // The asymmetry is deliberate and load-bearing. scripts/diarize.mjs records // whatever actually ran, including the basename of a `--engine` / // DIARIZE_ENGINE_CMD replacement, so asserting a hardcoded "sherpa-onnx" here // would mark every sidecar from any other wrapper permanently stale — an // infinite regeneration loop at ~500-680 s/audio-hour, and one the e2e fake // engine trips on its first run. So the default engine still asserts nothing // and compares exactly as it always did. // // Selecting `sortformer` IS a configuration statement, and there the engine is // asserted: sherpa sidecars become stale and the backfill lane offers to redo // them, which is the point — the two engines disagree about how many speakers // exist, and a corpus half-diarized by each is not one corpus. Switching back // needs no special case: a sortformer record carries neither segmentation nor // embedding model, so it fails the sherpa comparison on the models anyway. engine?: string; // Basenames, as DiarizationEngine records them — full paths are // machine-specific and would make every record stale on another shard. segmentationModel?: string; embeddingModel?: string; // Single-model engines. See DiarizationEngine.model. model?: string; threshold?: number; }; // Basename without importing node:path — this module is client-safe and has no // dependencies, which is what lets both the engine wrapper's consumers and the // backfill registry share the comparator below. function baseName(p: string): string { const parts = p.split(/[/\\]/); return parts[parts.length - 1] ?? p; } // The identity the CURRENT configuration would produce. Structurally typed // rather than taking DiarizationSettings, so this module stays free of a // settings import (settings.ts imports the threshold default FROM here). // // One definition, two callers — controller/diarizeOne.ts's short-circuit and // lib/operations.ts's state() — because a comparator and the writer it // guards disagreeing about the identity is how a corpus ends up either // regenerating forever or never. export function diarizationTarget(cfg: { engine?: DiarizationEngineId; segModel?: string; embModel?: string; sortformerModel?: string; threshold?: number; }): DiarizationFreshnessTarget { // Sortformer is end-to-end: one model, no segmentation/embedding pair, and no // clustering threshold. Comparing sherpa's knobs against it would mark every // sortformer sidecar stale on a threshold edit that could not have changed a // single one of its turns. if (cfg.engine === SORTFORMER_DIARIZATION_ENGINE) { return { engine: SORTFORMER_DIARIZATION_ENGINE, ...(cfg.sortformerModel ? { model: baseName(cfg.sortformerModel) } : {}), }; } return { // `engine` is deliberately NOT set for the default engine — see // DiarizationFreshnessTarget. Settings cannot know which binary a `--engine` // override will run, so claiming to compare it would mark every sidecar from // any other wrapper permanently stale. ...(cfg.segModel ? { segmentationModel: baseName(cfg.segModel) } : {}), ...(cfg.embModel ? { embeddingModel: baseName(cfg.embModel) } : {}), threshold: cfg.threshold ?? DEFAULT_DIARIZATION_THRESHOLD, }; } // Absent and "" are the same thing (nothing configured), so a record written by // an engine that reports no model name compares equal to a target that has none // either. Copied from digest.ts's sameVariant for the same reason it exists // there: the compatibility rule has to be explicit or adding a field silently // invalidates the whole corpus. function sameModel(a: string | undefined, b: string | undefined): boolean { return (a ?? "") === (b ?? ""); } // An absent recorded threshold is read as TODAY'S DEFAULT. That is the trick // that stops a newly-recorded field from invalidating everything written before // it existed: a sidecar from before the field was written compares equal as long // as the current setting is still the default. function sameThreshold(a: number | undefined, b: number | undefined): boolean { return ( (a ?? DEFAULT_DIARIZATION_THRESHOLD) === (b ?? DEFAULT_DIARIZATION_THRESHOLD) ); } // Is this sidecar what the current configuration would produce? // // Before this, diarizeOne short-circuited on mere EXISTENCE, which meant a // threshold or model change — the two most likely reasons to re-run at all — // left the whole corpus looking done. Three fields are compared and one is // deliberately not: `version` is excluded, because an engine point-release must // not invalidate ~500-680 s/audio-hour of captured work when the models and the // threshold it was clustered at are unchanged. export function isDiarizationFresh( record: DiarizationRecord | null, target: DiarizationFreshnessTarget, ): boolean { if (!record) return false; const e = record.engine; if (!e) return false; return ( (target.engine === undefined || e.engine === target.engine) && sameModel(e.segmentationModel, target.segmentationModel) && sameModel(e.embeddingModel, target.embeddingModel) && sameModel(e.model, target.model) && sameThreshold(e.threshold, target.threshold) ); } // Total attributed speech, in seconds. Turns may overlap (two people talking at // once), so this can exceed the audio duration — that is a real signal, not a // bug, and callers that want wall-time coverage should merge ranges first. export function diarizationSpeechSeconds(record: DiarizationRecord): number { return record.turns.reduce((sum, t) => sum + Math.max(0, t.end - t.start), 0); }