#!/usr/bin/env node // Standalone speaker diarization: audio in, diarization.json out. // // Wraps a diarization engine the same way scripts/parakeet-stitch.mjs wraps // parakeet-cli, and for the same reason: the app only ever knows a PATH // (paths.diarizeBin / DIARIZE_BIN), so the engine behind it can be swapped // without touching a caller. Today that engine is sherpa-onnx via // scripts/diarize-sherpa.py (ONNX, CPU); pyannote would be a drop-in // replacement for the --engine command. // // TWO ENGINES, PICKED BY --engine-kind: // sherpa-onnx (default) scripts/diarize-sherpa.py — ONNX segmentation + // embeddings, agglomerative clustering, CPU only. // sortformer scripts/diarize-sortformer.mjs — NVIDIA's streaming // Sortformer as ggml, end-to-end, Vulkan or CPU. // // THE GPU NOTE THIS FILE USED TO CARRY IS NOW OUT OF DATE, and the correction is // worth keeping: it said no diarization model had been ported to ggml, so the // RX 6600 XT could not run one while parakeet.cpp happily did ASR on it. That was // true of PyTorch and ONNX runtimes, and it is no longer true — openresearchtools // ported Sortformer to ggml, which inherits the same Vulkan backend parakeet // uses. Measured on this box: 894 s/audio-hour on Vulkan against 1305 for the // same engine on tuned CPU threads, on ONE core instead of 2.2. // // It is still not the fastest lane: sherpa-onnx does the same file in // 492 s/audio-hour across ~4 cores. Sortformer is chosen for QUALITY (4 speakers // where sherpa splits into 13, and 4 where sherpa splits into 35), and the GPU is // chosen because it is the cheapest way to run it. // // Dual-use, by design: // * In the app: controller/diarizeOne.ts runs this with cwd == the video dir // and an explicit --output, then the caller treats the file as the artifact. // * On the CLI: run it directly. With no --output it prints JSON to stdout. // // Output is a DiarizationRecord (see common/lib/diarization.ts) — speaker- // labelled time ranges plus engine provenance, so a later attribution pass can // tell what produced a given file and re-run selectively. // // Usage: // diarize.mjs [options] // // Options (env fallback in parens): // --output write the record here (default: stdout) // --video-id recorded in the sidecar (default: basename of cwd) // --engine-kind sherpa-onnx | sortformer (DIARIZE_ENGINE_KIND, sherpa-onnx) // --engine engine command (DIARIZE_ENGINE_CMD) // --sortformer-bin

engine binary (SORTFORMER_BIN) [sortformer] // --sortformer-model

.gguf model (SORTFORMER_MODEL) [sortformer] // --backend vulkan | cpu (default vulkan) [sortformer] // --python python for the default engine (DIARIZE_PYTHON, "python3") // --seg segmentation model (DIARIZE_SEG_MODEL) [required] // --emb speaker-embedding model (DIARIZE_EMB_MODEL) [required] // --threshold clustering threshold (DIARIZE_THRESHOLD, 0.5) // --threads engine threads (DIARIZE_THREADS, 4) // --ffmpeg ffmpeg binary (FFMPEG_BIN, "ffmpeg") // --ffprobe ffprobe binary (FFPROBE_BIN, "ffprobe") // --window-minutes window length, 0 = never window // (DIARIZE_WINDOW_MINUTES, 45) // --window-after-minutes only window files longer than this // (DIARIZE_WINDOW_AFTER_MINUTES, 90) // -h, --help // // WINDOWING IS RECORDED BUT IS NOT PART OF THE FRESHNESS IDENTITY. It lands on // the record as a top-level `windowing` field, deliberately OUTSIDE `engine` — // isDiarizationFresh compares engine/models/threshold, and the target it // compares against is built from settings alone, which cannot know a given // video's duration. A window field in that identity would therefore mark every // sidecar on disk stale on the day windowing shipped, for work that is // unchanged. Same treatment `version` already gets, and for the same reason. // See common/lib/diarization.ts, which says so in as many words. import { spawn } from "node:child_process"; import { rename, writeFile } from "node:fs/promises"; import path from "node:path"; const HELP = `diarize.mjs [options] Runs speaker diarization and writes a DiarizationRecord JSON sidecar. See the header of this file for the full option list.`; function fail(msg, code = 2) { process.stderr.write(`diarize: ${msg}\n`); process.exit(code); } const argv = process.argv.slice(2); if (argv.includes("-h") || argv.includes("--help")) { process.stdout.write(HELP + "\n"); process.exit(0); } function arg(flag, fallback) { const i = argv.indexOf(flag); return i < 0 || i === argv.length - 1 ? fallback : argv[i + 1]; } const positional = []; for (let i = 0; i < argv.length; i++) { if (argv[i].startsWith("-")) { i++; // every flag here takes a value continue; } positional.push(argv[i]); } const audio = positional[0]; if (!audio) fail("missing "); const output = arg("--output", arg("-o", undefined)); const videoId = arg("--video-id", path.basename(process.cwd())); const python = arg("--python", process.env.DIARIZE_PYTHON ?? "python3"); const seg = arg("--seg", process.env.DIARIZE_SEG_MODEL ?? ""); const emb = arg("--emb", process.env.DIARIZE_EMB_MODEL ?? ""); const threshold = Number( arg("--threshold", process.env.DIARIZE_THRESHOLD ?? "0.5"), ); const threads = Number(arg("--threads", process.env.DIARIZE_THREADS ?? "4")); const ffmpeg = arg("--ffmpeg", process.env.FFMPEG_BIN ?? "ffmpeg"); const ffprobe = arg("--ffprobe", process.env.FFPROBE_BIN ?? "ffprobe"); const windowMinutes = arg( "--window-minutes", process.env.DIARIZE_WINDOW_MINUTES ?? "45", ); const windowAfterMinutes = arg( "--window-after-minutes", process.env.DIARIZE_WINDOW_AFTER_MINUTES ?? "90", ); const engineCmd = arg("--engine", process.env.DIARIZE_ENGINE_CMD ?? ""); const engineKind = arg( "--engine-kind", process.env.DIARIZE_ENGINE_KIND ?? "sherpa-onnx", ); const sortformerBin = arg("--sortformer-bin", process.env.SORTFORMER_BIN ?? ""); const sortformerModel = arg( "--sortformer-model", process.env.SORTFORMER_MODEL ?? "", ); const backend = arg("--backend", "vulkan"); if (!engineCmd && engineKind !== "sherpa-onnx" && engineKind !== "sortformer") { fail(`--engine-kind must be 'sherpa-onnx' or 'sortformer' (got '${engineKind}')`); } // `--engine` still wins over `--engine-kind`. It is the raw escape hatch (and // what the e2e fake engine uses), so a kind must never quietly override it. const usingSortformer = !engineCmd && engineKind === "sortformer"; let cmd; let args; if (engineCmd) { const parts = engineCmd.split(" ").filter(Boolean); cmd = parts[0]; args = [...parts.slice(1), audio]; } else if (usingSortformer) { if (!sortformerBin) fail("missing --sortformer-bin (or SORTFORMER_BIN)"); if (!sortformerModel) fail("missing --sortformer-model (or SORTFORMER_MODEL)"); // process.execPath, not a bare "node": the app may run under a node that is not // the one on PATH, and the engine has to be the same runtime as its wrapper. cmd = process.execPath; args = [ path.join(import.meta.dirname, "diarize-sortformer.mjs"), "--bin", sortformerBin, "--model", sortformerModel, "--backend", backend, "--threads", String(threads), "--ffmpeg", ffmpeg, audio, ]; } else { if (!seg) fail("missing --seg (or DIARIZE_SEG_MODEL)"); if (!emb) fail("missing --emb (or DIARIZE_EMB_MODEL)"); cmd = python; args = [ path.join(import.meta.dirname, "diarize-sherpa.py"), "--seg", seg, "--emb", emb, "--threshold", String(threshold), "--threads", String(threads), "--ffmpeg", ffmpeg, "--ffprobe", ffprobe, "--window-minutes", String(windowMinutes), "--window-after-minutes", String(windowAfterMinutes), audio, ]; } const started = Date.now(); const child = spawn(cmd, args, { stdio: ["ignore", "pipe", "inherit"] }); let stdout = ""; child.stdout.setEncoding("utf8"); child.stdout.on("data", (d) => { stdout += d; }); child.on("error", (err) => fail(`failed to run ${cmd}: ${err.message}`, 3)); child.on("close", async (code) => { if (code !== 0) fail(`engine exited ${code}`, code ?? 1); let raw; try { raw = JSON.parse(stdout); } catch { fail("engine produced no parseable JSON on stdout"); } if (!Array.isArray(raw?.turns)) fail("engine output has no `turns` array"); const turns = raw.turns.map((t) => ({ start: Number(t.start), end: Number(t.end), speaker: Number(t.speaker), })); const record = { videoId, generatedAt: new Date().toISOString(), ...(Number.isFinite(raw.audioSeconds) ? { audioSeconds: raw.audioSeconds } : {}), durationMs: Date.now() - started, // Outside `engine` on purpose — see the header. Absent for a whole-file run, // which is what every sidecar written before windowing existed looks like. ...(raw.windowing && typeof raw.windowing === "object" ? { windowing: raw.windowing } : {}), speakers: new Set(turns.map((t) => t.speaker)).size, turns, // Provenance is per-engine, and the fields a given engine CANNOT have are // left off rather than written empty. Recording sherpa's threshold on a // sortformer record would be a lie that isDiarizationFresh then acts on: it // has no clustering step and no threshold, so a threshold edit must not mark // its sidecars stale. engine: usingSortformer ? { engine: "sortformer", model: path.basename(sortformerModel), // The driver reports its ggml backend here ("sortformer/Vulkan0"), so // the record says which device produced it. Excluded from the freshness // identity along with every other `version` — CPU and Vulkan produce // byte-identical turns, so re-running one on the other buys nothing. ...(raw.version ? { version: String(raw.version) } : {}), } : { engine: engineCmd ? path.basename(cmd) : "sherpa-onnx", // basename only: absolute paths are machine-specific and the corpus is // rsynced between shards, so a full path would make records non-portable. ...(seg ? { segmentationModel: path.basename(seg) } : {}), ...(emb ? { embeddingModel: path.basename(emb) } : {}), ...(raw.version ? { version: String(raw.version) } : {}), threshold, }, }; const json = JSON.stringify(record) + "\n"; if (!output) { process.stdout.write(json); return; } // Atomic: a half-written sidecar must never be what convinces the cleanup // sweep it is safe to delete the only copy of the audio. const tmp = `${output}.tmp-${process.pid}`; await writeFile(tmp, json); await rename(tmp, output); process.stderr.write( `diarize: ${record.turns.length} turns, ${record.speakers} speakers -> ${output}\n`, ); });