commit a54044ef64cdba2629b925c90c9be901c2311b03
parent 29d99aa8374c22265797a4ab2e8501e0555c22a3
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 10 Aug 2026 15:26:47 -0400
Put the sortformer engine settings on the settings page
The engine shipped reachable only by hand-editing settings.json, because
the Diarization fieldset had no controls for `engine`, `backend`,
`sortformerBin` or `sortformerModel`. A field the form does not render is
a field an operator cannot set.
The engine select comes first, since it decides which of the other fields
matter, and its help text states the trade in the terms the choice is
actually made on: sherpa-onnx is FASTER, sortformer is more accurate, and
switching marks the other engine's captures as work to redo.
The e2e is the point rather than a formality. The diarization branch of
the save action REBUILDS the whole block, so a control added to the
markup and forgotten in the action would look like it saved and silently
revert -- the same hazard the digest prompt-shape test already exists for.
It also asserts the sherpa fields survive selecting the other engine.
Targeted by control NAME, not by label: these labels wrap their whole
explanatory hint, so the accessible name is a paragraph, and getByLabel's
substring matching makes "Engine" both non-exact and ambiguous against
"Engine threads".
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Diffstat:
4 files changed, 132 insertions(+), 1 deletion(-)
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,7 +1,7 @@
# Changelog
## [Unreleased]
-- **Speaker capture can now run on the graphics card, and stops inventing speakers that were never there.** The old engine works by grouping voices it thinks sound alike, and it splits far too eagerly: on a 13-minute reaction video with one host it found **13 speakers**, and on the worst video in the corpus it found **35**. The new engine decides speaker turns directly instead of grouping them afterwards, and returns **4** in both cases — agreeing with the old one about how much of the video the main speaker talks for (73.5% against 73.1%) while collapsing the invented tail. It has no threshold to tune, and it caps at 4 speakers, so a panel of five will merge two of them rather than split one into twenty. Turn it on with `"engine": "sortformer"` after running `scripts/build-sortformer.sh`; the previous engine stays the default and stays supported for machines with no usable GPU.
+- **Speaker capture can now run on the graphics card, and stops inventing speakers that were never there.** The old engine works by grouping voices it thinks sound alike, and it splits far too eagerly: on a 13-minute reaction video with one host it found **13 speakers**, and on the worst video in the corpus it found **35**. The new engine decides speaker turns directly instead of grouping them afterwards, and returns **4** in both cases — agreeing with the old one about how much of the video the main speaker talks for (73.5% against 73.1%) while collapsing the invented tail. It has no threshold to tune, and it caps at 4 speakers, so a panel of five will merge two of them rather than split one into twenty. Run `scripts/build-sortformer.sh`, then pick **Engine → Sortformer** under Settings → Diarization and paste the two paths it prints. The previous engine stays the default and stays supported for machines with no usable GPU.
- **Speaker capture no longer has a length limit to worry about.** The old engine's memory grew with the square of how many turns a recording contains, which is why long streams had to be processed in windows and why a duration cap existed at all. The new engine holds a fixed amount of memory no matter how long the recording is — 558 MB for a 13-minute video or an 8-hour one — so nothing has to be windowed, capped, or decoded to a temporary file first.
- **Switching speaker-capture engines correctly marks the old results as work to redo.** The two engines disagree about how many speakers exist, so a corpus half-captured by each is not one corpus. Changing the engine now shows every recording captured by the other one as outstanding backfill work, rather than leaving it looking finished. Changing unrelated settings — a clustering threshold the new engine does not even have — no longer marks anything stale.
- **Speaker capture on the GPU stands aside for transcription instead of fighting it for video memory.** It needs about 4.4 GB of the 8 GB card that transcription also uses, so it now holds while transcription is working and resumes the moment the card is free — the same behaviour the digest sweep already had. The "run a guaranteed share alongside" setting is ignored while GPU capture is enabled, because a share of video memory is not a slower run, it is a failure.
diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts
@@ -370,6 +370,24 @@ export async function saveSettingsAction(
dDiar.python,
segModel: String(formData.get("diarizationSegModel") ?? "").trim(),
embModel: String(formData.get("diarizationEmbModel") ?? "").trim(),
+ // Read as plain strings and left to sanitizeDiarization to validate:
+ // an unrecognized value there falls back to the DEFAULT engine, which
+ // is the one every sidecar on disk already matches. Narrowing here
+ // instead would mean this form and the sanitizer could disagree about
+ // what a valid engine is, and the corpus pays for that disagreement in
+ // weeks of regeneration.
+ engine: String(
+ formData.get("diarizationEngine") ?? dDiar.engine,
+ ) as typeof dDiar.engine,
+ backend: String(
+ formData.get("diarizationBackend") ?? dDiar.backend,
+ ) as typeof dDiar.backend,
+ sortformerBin: String(
+ formData.get("diarizationSortformerBin") ?? "",
+ ).trim(),
+ sortformerModel: String(
+ formData.get("diarizationSortformerModel") ?? "",
+ ).trim(),
}
: dDiar;
diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx
@@ -763,6 +763,71 @@ export function SettingsForm({ initial, apps, digestApps }: Props) {
</span>
</span>
</label>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Engine</span>
+ <select
+ name="diarizationEngine"
+ defaultValue={initial.diarization.engine}
+ className="rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ <option value="sherpa-onnx">
+ sherpa-onnx — clustered, CPU only (default)
+ </option>
+ <option value="sortformer">
+ Sortformer — end-to-end, GPU or CPU
+ </option>
+ </select>
+ <span className="text-xs text-muted-foreground">
+ <strong>sherpa-onnx</strong> groups voices that sound alike, and it
+ splits far too eagerly: on a 13-minute video with one host it finds
+ 13 speakers, and on the worst file in this corpus it finds 35.{" "}
+ <strong>Sortformer</strong> decides turns directly instead of
+ grouping them afterwards and returns 4 in both cases, agreeing with
+ the other engine about how much of the video the main speaker talks
+ for. It has no threshold, and it caps at 4 speakers — a panel of
+ five will merge two rather than invent twenty.
+ <br />
+ <br />
+ sherpa-onnx is the <strong>faster</strong> of the two (492 against
+ 894 seconds per audio-hour here), so this is a quality choice, not a
+ speed one. Switching also{" "}
+ <strong>marks every recording captured by the other engine as work
+ to redo</strong>, because the two disagree about how many speakers
+ exist — on the audio still held here that is weeks of it.
+ </span>
+ </label>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Sortformer device</span>
+ <select
+ name="diarizationBackend"
+ defaultValue={initial.diarization.backend}
+ className="rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ <option value="vulkan">Vulkan — the graphics card</option>
+ <option value="cpu">CPU</option>
+ </select>
+ <span className="text-xs text-muted-foreground">
+ Ignored unless the engine above is Sortformer. Both produce{" "}
+ <strong>identical</strong> speaker turns, so this only trades one
+ resource for another: the card is ~1.5× faster and uses one core
+ instead of two, but takes about 4.4 GB of the same 8 GB card
+ transcription uses — so capture stands aside while transcription is
+ working, and resumes when the card is free. On CPU it needs 4.84 GB
+ of system memory instead.
+ </span>
+ </label>
+ <Field
+ label="Sortformer engine binary"
+ name="diarizationSortformerBin"
+ defaultValue={initial.diarization.sortformerBin}
+ hint="Absolute path to the diarize-file binary built by scripts/build-sortformer.sh. Empty means the Sortformer engine is not configured, and every run reports a skip rather than a failure."
+ />
+ <Field
+ label="Sortformer model (GGUF)"
+ name="diarizationSortformerModel"
+ defaultValue={initial.diarization.sortformerModel}
+ hint="Absolute path to the .gguf downloaded by scripts/build-sortformer.sh. Its filename is recorded in every sidecar, so changing the model is what marks earlier captures worth redoing."
+ />
<Field
label="Segmentation model (ONNX)"
name="diarizationSegModel"
diff --git a/editor/e2e/settings.spec.ts b/editor/e2e/settings.spec.ts
@@ -255,3 +255,51 @@ test("an unrelated settings save does not reset the digest prompt shape", async
expect(saved.digest?.timestampMode).toBe("absolute");
expect(saved.digest?.promptVariant).toBe("keepme");
});
+
+// THE BUG THIS EXISTS FOR: the sortformer engine shipped with its settings
+// reachable only by hand-editing settings.json, because the form simply had no
+// controls for them. A field the form does not render is a field an operator
+// cannot set — and, worse, the diarization branch rebuilds the whole block on
+// save, so a control added to the markup but forgotten in the action would look
+// like it worked and silently revert.
+test("the sortformer engine settings round-trip through the form", async ({
+ page,
+}) => {
+ await page.goto("/settings");
+
+ // Targeted by NAME, not by label. getByLabel matches by substring and these
+ // labels wrap their whole explanatory hint, so the accessible name is a
+ // paragraph — "Engine" matches nothing exactly and matches "Engine threads"
+ // loosely. The control's name is the thing the action actually reads.
+ await page.locator('select[name="diarizationEngine"]').selectOption("sortformer");
+ await page.locator('select[name="diarizationBackend"]').selectOption("cpu");
+ await page
+ .getByLabel("Sortformer engine binary")
+ .fill("/opt/sortformer/diarize-file");
+ await page
+ .getByLabel("Sortformer model (GGUF)")
+ .fill("/opt/sortformer/model.gguf");
+
+ await page.getByRole("button", { name: /save settings/i }).click();
+ await expect(
+ page.getByRole("status").filter({ hasText: "Saved" }),
+ ).toBeVisible();
+
+ const saved = await readJson<{
+ diarization?: {
+ engine: string;
+ backend: string;
+ sortformerBin: string;
+ sortformerModel: string;
+ segModel: string;
+ threshold: number;
+ };
+ }>("test-settings.json");
+ expect(saved.diarization?.engine).toBe("sortformer");
+ expect(saved.diarization?.backend).toBe("cpu");
+ expect(saved.diarization?.sortformerBin).toBe("/opt/sortformer/diarize-file");
+ expect(saved.diarization?.sortformerModel).toBe("/opt/sortformer/model.gguf");
+ // The sherpa fields are still carried, not clobbered by selecting the other
+ // engine — switching back must not mean re-entering three paths.
+ expect(saved.diarization?.threshold).toBe(0.9);
+});