commit 5b8cb9c3403fe2476f78aed391c2458675df0bbd
parent 33afe87ac54ab81361db57ec41158202087e2915
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 14 Jun 2026 01:38:58 -0400
Transcription workers: auto-disable unreachable remotes (Phase 5)
Use the heartbeat to make failover actually protect the current video, not just
later ones:
- workerPool.markDegraded(id): force a worker degraded immediately (skipped until
re-enabled), distinct from the consecutive-failure threshold.
- remoteTranscribe.pingRemoteHealth(worker): probe /api/worker/health with a
timeout; true only on a healthy 200.
- transcribeWithWorker, on a remote transport failure: probe health and, if the
remote is down, degrade it now so this video's retry (and later videos) skip it
and fail over to a healthy worker — instead of burning all attempts on a dead
remote. A still-reachable remote (transient 5xx) keeps the gentler 3-strike
path. Auto-disable is logged.
- Workers page degraded note covers "unreachable or repeated failures".
e2e: an unreachable remote (dead port) at priority 0 auto-disables and the video
fails over to a local worker; the Workers page shows it degraded. Migration specs
updated to the worker editor (the old transcriptionApp <select> is gone).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Diffstat:
7 files changed, 92 insertions(+), 10 deletions(-)
diff --git a/common/controller/remoteTranscribe.ts b/common/controller/remoteTranscribe.ts
@@ -45,6 +45,28 @@ function base(worker: Worker): string {
return (worker.remote?.baseUrl ?? "").replace(/\/+$/, "");
}
+// Probe a remote worker's /api/worker/health. Returns true only on a 200 with a
+// healthy body. Used to confirm a remote is actually down (vs a transient job
+// error) so the scheduler can degrade it immediately instead of burning retries.
+export async function pingRemoteHealth(
+ worker: Worker,
+ timeoutMs = 3000,
+): Promise<boolean> {
+ const baseUrl = base(worker);
+ if (!baseUrl) return false;
+ try {
+ const res = await fetch(`${baseUrl}/api/worker/health`, {
+ headers: authHeaders(worker),
+ signal: AbortSignal.timeout(timeoutMs),
+ });
+ if (!res.ok) return false;
+ const body = (await res.json()) as { ok?: boolean };
+ return body.ok === true;
+ } catch {
+ return false;
+ }
+}
+
// Delegate a transcription to the remote and poll to completion. Returns the
// transcript.json bytes for the upload path, or null for the shared-fs path
// (the remote wrote the transcript directly onto the shared mount, so the caller
diff --git a/common/controller/transcribeOne.ts b/common/controller/transcribeOne.ts
@@ -8,7 +8,7 @@ import { detectTranscriptFormat } from "../lib/whisper";
import { getWorkerPool } from "../jobs/workerPool";
import type { TaskTracker } from "../jobs/taskHooks";
import { normalizeTranscript } from "./normalizeTranscript";
-import { transcribeViaRemote } from "./remoteTranscribe";
+import { pingRemoteHealth, transcribeViaRemote } from "./remoteTranscribe";
import { TranscribeError } from "./transcribeError";
import { isRealAudioFile } from "../lib/videoStatus";
@@ -255,8 +255,22 @@ export async function transcribeWithWorker(
const failureClass =
err instanceof TranscribeError ? err.failureClass : "transcription";
if (failureClass === "transport") {
- pool.markFailure(worker.id);
lastErr = err;
+ // A remote that fails a health check is genuinely down: degrade it now so
+ // this video's retry AND later videos skip it, rather than burning all
+ // attempts on it. A reachable remote (transient 5xx) or a local worker
+ // uses the gentler consecutive-failure threshold.
+ if (worker.kind === "remote" && !(await pingRemoteHealth(worker))) {
+ if (pool.markDegraded(worker.id)) {
+ log(
+ `Worker ${worker.id} is unreachable — disabled for this run; re-enable it from the Workers page once it's back.`,
+ );
+ }
+ } else if (pool.markFailure(worker.id)) {
+ log(
+ `Worker ${worker.id} auto-disabled after ${MAX_WORKER_ATTEMPTS} consecutive failures; re-enable it from the Workers page.`,
+ );
+ }
log(
`Worker ${worker.id} transport failure on ${opts.videoId}: ${String(err)} — retrying on another worker`,
);
diff --git a/common/jobs/workerPool.ts b/common/jobs/workerPool.ts
@@ -322,6 +322,17 @@ class WorkerPool {
return false;
}
+ // Force a worker degraded immediately (skipped by the scheduler until
+ // re-enabled), regardless of the consecutive-failure count. Used when a remote
+ // is confirmed unreachable by a health ping — no point burning more attempts on
+ // it. Returns true if this transitioned a previously-healthy worker.
+ markDegraded(id: string): boolean {
+ const entry = this.entries.get(id);
+ if (!entry || entry.degraded) return false;
+ entry.degraded = true;
+ return true;
+ }
+
// True when at least one worker is enabled and not degraded (ignoring momentary
// fullness — a full but enabled pool is working at capacity, not paused). Its
// negation is what surfaces the "paused — waiting for a worker" state: a running
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -4,7 +4,7 @@
- **Sites can link to each other.** A site's form gained a **Public URL** field (the absolute URL it's served at, e.g. `https://jeralyzer.com`) and a **Related sites** section. The export footer automatically links to every *other* site that has a Public URL, so filling these in is all that's needed for cross-site links; a site left without a URL is simply omitted from the lists. The **Related sites** editor lets a site pull closely-related siblings to the front under named groups (e.g. Jeralyzer featuring Rekietalyzer under "MTG drama") — add a group, give it an optional heading, and check which sibling sites belong; everything you don't feature falls into a trailing "Other sites" group on its own. Groups reorder with ↑/↓. The picker only lists sites that actually exist, and featured ids for sites that were since deleted are dropped on save (with a heads-up note). It's a subtle, secondary feature — see the matching note in the export changelog for how it renders.
- **The editor refreshes itself on a timer so its data stays live without a manual reload.** Every page now passively re-fetches its own server-rendered data on a configurable interval — so the sidebar badges (active/running job counts, changelog dot), channel reports, and any other on-screen figures keep up to date on their own. It uses Next's `router.refresh()` (the same mechanism the jobs list already used) mounted once globally in the root layout, so it covers every page and the shared sidebar with no per-page wiring. To avoid wasting work when you're not looking, it **pauses entirely while the browser tab is hidden** and does **one immediate refresh the moment you return** to the tab (rather than waiting out the interval); it also skips a tick while a previous refresh is still settling, so refreshes can't pile up. The cadence is set in **Settings → Auto-refresh interval (seconds)**: default **5s** (clamped 1–600), or **0 to disable** passive refresh completely. This replaces the jobs page's old bespoke 2.5s auto-refresh (the `/jobs/active` page keeps its faster 1s progress-bar polling, which animates per-task bars without a full re-render).
- **Transcription is now driven by configurable workers instead of one global engine.** The old single **App** dropdown in **Settings → Transcription** is replaced by a **Transcription workers** list. Each worker is **one processing slot** — one transcription at a time — with its own engine (whisper.cpp / chough / parakeet) and config, and a priority given by its position in the list (top = preferred). To run several in parallel, add more workers; a **Copy** button duplicates one (e.g. point two copies at the same chough `--server` for two togglable server slots). A batch ("Transcribe missing", bucket, bulk, single-video) hands each video — per task — to the highest-priority free worker, so a fast GPU worker and a slower CPU worker (e.g. parakeet on the GPU + chough on the CPU) run side by side instead of one engine doing everything. Total parallelism is the number of enabled workers; the old per-run **Concurrency** control and the global **Parallel transcriptions** setting are gone (add/remove workers, or disable/drain one, to change load). A pre-worker `settings.json` migrates automatically to one worker per slot of the previously-selected app (the old parallel-transcriptions count becomes that many enabled copies), plus a disabled worker for any other engine you had configured, so existing installs keep their parallelism. Scheduling is a single process-wide pool, so two batches can't oversubscribe the same GPU. One-slot-per-worker also means you can disable a single slot to free *some* of a CPU/GPU while the rest keep transcribing.
-- **Remote workers: offload transcription to another instance of this app on your LAN.** Add a **remote** worker in the Settings list with the base URL of another instance (e.g. `http://gpu-box.lan:3001`) and a shared token. When a video is dispatched to it, this instance uploads the audio over HTTP, the remote transcribes it through *its own* worker pool (picking among its local engines), streams progress and log back, and this instance pulls the finished `transcript.json` and normalizes it locally — so the remote needs no knowledge of your channels, just CPU/GPU. The protocol lives under `/api/worker/*` and is **disabled unless `WORKER_TOKEN` is set** in the environment, so an instance is never an open transcription server by accident; every request carries `Authorization: Bearer <token>`, validated with a constant-time compare against the accepting instance's own `WORKER_TOKEN` (never against settings). Uploaded audio and the produced transcript live in a scratch dir that's cleaned up once the result is pulled (or the job is cancelled). If a remote is unreachable or returns a transport error mid-job, the video is automatically retried on another worker; a genuine transcription failure on the remote is not retried. A `GET /api/worker/health` endpoint returns the remote's worker summary for heartbeating.
+- **Remote workers: offload transcription to another instance of this app on your LAN.** Add a **remote** worker in the Settings list with the base URL of another instance (e.g. `http://gpu-box.lan:3001`) and a shared token. When a video is dispatched to it, this instance uploads the audio over HTTP, the remote transcribes it through *its own* worker pool (picking among its local engines), streams progress and log back, and this instance pulls the finished `transcript.json` and normalizes it locally — so the remote needs no knowledge of your channels, just CPU/GPU. The protocol lives under `/api/worker/*` and is **disabled unless `WORKER_TOKEN` is set** in the environment, so an instance is never an open transcription server by accident; every request carries `Authorization: Bearer <token>`, validated with a constant-time compare against the accepting instance's own `WORKER_TOKEN` (never against settings). Uploaded audio and the produced transcript live in a scratch dir that's cleaned up once the result is pulled (or the job is cancelled). If a remote returns a transport error mid-job, the video is automatically retried on another worker; a genuine transcription failure on the remote is not retried. On a transport failure the remote's `GET /api/worker/health` is probed, and a remote confirmed **down** is auto-disabled (shown "degraded" on the Workers page, with **Enable** to retry once it's back) so neither the current video nor later ones keep burning attempts on it — they fail over to a healthy worker. A worker that racks up repeated failures while still reachable is auto-disabled after a few strikes.
- **New Workers page (`/workers`) with live status and runtime controls.** Lists every worker with its state (idle / busy / draining / disabled / degraded) and the video it's currently transcribing with per-task progress. Each worker can be **disabled** (stop taking new work immediately; in-flight transcriptions keep running), **drained** (stop taking new work but let the current video finish — the graceful "free up the GPU when it's done" path), or **enabled** again — without editing settings, so you can hand a CPU/GPU back to other programs and reclaim it later. A **Pause all** button disables every worker at once and remembers each one's state; **Resume all** restores them exactly. These runtime controls are transient (a restart returns workers to their configured enabled state); the Settings list is where the persisted defaults live.
- **Batches pause instead of failing when no worker is available.** If every worker is disabled (or you hit **Pause all**) while a transcription batch is running, the batch parks — it keeps its in-flight video to completion, starts no new ones, and stays **running** on `/jobs/active` rather than failing the remaining videos. Re-enabling any worker (or **Resume all**) immediately resumes it where it left off. A video whose worker fails for a transport reason (e.g. a remote worker that went away) is automatically retried on another worker before being recorded as failed.
- **Git worktrees can run in parallel on non-colliding ports (dev tooling).** Two checkouts of the repo (via `git worktree`) can now run their dev servers and e2e suites at the same time without port clashes. A new helper, `scripts/worktree.mjs` (exposed as `pnpm wt`), assigns each worktree a port block offset by `index * 100` based on its position in `git worktree list` — the main worktree keeps the original defaults (editor 3001, test 3011, export 3010/3000/3020), worktree #1 gets 31xx, and so on. `pnpm dev:editor`, `pnpm dev:export`, `pnpm start:export`, and `pnpm e2e` route through `wt run`, which injects the assigned ports, so they "just work" per worktree; the editor/export `package.json` port flags and the Playwright configs now honor these env vars (previously `pnpm dev:test` hardcoded 3011, so a custom `PORT` only moved the URL Playwright waited on, not the server). E2E specs that hit the editor's test API now derive the base URL from `PLAYWRIGHT_BASE_URL` (centralized in `editor/e2e/baseUrl.ts`) instead of hardcoding `localhost:3011`. `pnpm wt add <branch>` creates a sibling worktree pre-seeded with `settings.json` and prints its ports; `--share-data` links it to the main worktree's downloaded `transcripts/` for read-mostly reuse (with an LMDB concurrent-write caveat). See `WORKTREES.md`.
diff --git a/editor/app/workers/components/WorkersView.tsx b/editor/app/workers/components/WorkersView.tsx
@@ -228,7 +228,7 @@ function WorkerCard({
</div>
{w.degraded && (
<p className="text-xs text-red-700 dark:text-red-300">
- Auto-disabled after repeated failures. Enable to retry.
+ Auto-disabled — unreachable or repeated failures. Enable to retry.
</p>
)}
{w.tasks.length > 0 && (
diff --git a/editor/e2e/transcription-app-migration.spec.ts b/editor/e2e/transcription-app-migration.spec.ts
@@ -2,10 +2,11 @@ import { test, expect } from "@playwright/test";
import { resetData, writeSettings } from "./helpers";
// A pre-multi-app settings.json (transcribeBin/transcribeArgs/transcribeModel,
-// no transcriptionApp key) must migrate onto the app registry when read, so the
-// Settings page shows a selected app rather than crashing on missing fields.
+// no transcriptionApp/workers keys) must migrate onto the worker model when
+// read, so the Settings page shows a worker on the matching engine rather than
+// crashing on missing fields.
-test("legacy whisper settings migrate onto the whisper-cpp app", async ({
+test("legacy whisper settings migrate onto a whisper-cpp worker", async ({
page,
}) => {
await resetData(null);
@@ -18,10 +19,10 @@ test("legacy whisper settings migrate onto the whisper-cpp app", async ({
transcribeArgs: ["-ojf", "-l", "en", "-m", "{model}", "-of", "{outputBase}", "{audioFile}"],
});
await page.goto("/settings");
- await expect(page.locator('select[name="transcriptionApp"]')).toHaveValue("whisper-cpp");
+ await expect(page.getByLabel("worker 1 engine")).toHaveValue("whisper-cpp");
});
-test("legacy settings whose binary is chough migrate onto the chough app", async ({
+test("legacy settings whose binary is chough migrate onto a chough worker", async ({
page,
}) => {
await resetData(null);
@@ -34,5 +35,5 @@ test("legacy settings whose binary is chough migrate onto the chough app", async
transcribeArgs: ["-f", "json", "-o", "{outputBase}", "{audioFile}"],
});
await page.goto("/settings");
- await expect(page.locator('select[name="transcriptionApp"]')).toHaveValue("chough");
+ await expect(page.getByLabel("worker 1 engine")).toHaveValue("chough");
});
diff --git a/editor/e2e/worker-remote.spec.ts b/editor/e2e/worker-remote.spec.ts
@@ -147,3 +147,37 @@ test("a batch dispatched to a remote worker transcribes via the HTTP round-trip"
)
.toBe(true);
});
+
+test("an unreachable remote is auto-disabled and the video fails over to a local worker", async ({
+ page,
+}) => {
+ test.setTimeout(60_000);
+ // Remote (priority 0) points at a dead port; a local worker backs it up.
+ await writeSettings({
+ workers: [
+ { id: "dead", name: "Dead Remote", kind: "remote", enabled: true, priority: 0, remote: { baseUrl: "http://127.0.0.1:59599", token: TOKEN } },
+ { id: "cpu", name: "CPU", kind: "local", enabled: true, priority: 1, appId: "whisper-cpp", config: {} },
+ ],
+ });
+ await makeTranscribeChannel("failover-chan", ["vidfo1"]);
+
+ await page.goto("/channels/failover-chan");
+ await page.getByRole("button", { name: "Transcribe missing" }).click();
+
+ // The video is still transcribed — on the local worker, after a health check
+ // confirms the remote is down and degrades it.
+ await expect
+ .poll(
+ () =>
+ pathExists(
+ "test-transcripts/channels/failover-chan/data/vidfo1/transcript.json",
+ ),
+ { timeout: 30_000 },
+ )
+ .toBe(true);
+
+ // The Workers page shows the dead remote as degraded.
+ await page.goto("/workers");
+ const dead = page.getByRole("listitem").filter({ hasText: "Dead Remote" });
+ await expect(dead).toContainText("degraded", { timeout: 10_000 });
+});