commit 133b5a9965761e5cd19ccba97cc571f5989d81c1
parent 90028d915e239c83fe25c84ec6a5ae03c7461c78
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 22 Jun 2026 15:59:48 -0400
Auto-queue: respect disabled workers; label runner in Active Jobs
Two fixes from using the feature:
- Worker respect: the auto-transcribe runner used workers the operator had
disabled via Workers -> "Set as default". The runner started at boot and
called getWorkerPool().reconfigure() before anything triggered the pool's
lazy ensureInit(), which set the pool's `initialized` flag WITHOUT applying
the saved .worker-defaults.json arrangement — so disabled workers came back
enabled. Removed that pre-emptive reconfigure(); the pool now self-inits on
first use (eligibleSlots/acquire), applying settings AND the saved default.
- Active Jobs labeling: channel-less jobs (the cross-channel runners) were
dumped under a generic "Other" section. They now get their own labeled
sections ("Auto-transcribe"/"Auto-download") via a shared jobKindLabel map
(also used for friendlier kind labels in the running-jobs list), and each
in-flight item is labeled "channel/videoId" so you can see its channel.
New e2e: a saved default disabling cpu keeps cpu unused (asserts every
transcribe-outcome.json worker.id === gpu); the runner shows in an
"Auto-transcribe" section, not "Other". Existing drain/workers/jobs-order
specs and the production build stay green.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Diffstat:
7 files changed, 149 insertions(+), 18 deletions(-)
diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts
@@ -287,10 +287,15 @@ async function runLoop(
let metaCache: ChannelMeta[] = [];
let metaAt = 0;
- // Seed/refresh the global worker pool from current settings so eligibleSlots()
- // and acquire() see the latest worker config (mirrors runWhisperBatch). After
- // an e2e cache-reset the pool is recreated and re-seeds here.
- getWorkerPool().reconfigure();
+ // NOTE: do NOT call getWorkerPool().reconfigure() here. At server boot the
+ // runner is the first thing to touch the pool, and reconfigure() sets the
+ // pool's `initialized` flag WITHOUT applying the saved "Set as default"
+ // arrangement (.worker-defaults.json) — so a pre-empted ensureInit() would
+ // never disable the workers the operator turned off, and the runner would use
+ // them anyway. Leaving it to the pool's lazy ensureInit() (triggered by the
+ // first eligibleSlots()/acquire() below) seeds settings AND applies defaults
+ // exactly once. Live settings changes are pushed by saveSettingsAction's own
+ // reconfigure(applyEnabled:true), so nothing is lost by not refreshing here.
onLog(`Auto-${kind} runner started.`);
@@ -450,6 +455,9 @@ async function launchUnit(args: LaunchArgs): Promise<
audioFilename: "audio.mp3",
strict: false,
tracker: args.tracker,
+ // Show which channel each in-flight transcription belongs to (the runner
+ // is cross-channel, so the bare videoId wouldn't say).
+ taskLabel: `${args.channelSlug}/${args.pick.videoId}`,
onLog: args.onLog,
signal: args.signal,
});
@@ -475,7 +483,7 @@ async function launchUnit(args: LaunchArgs): Promise<
const settings = getSettings();
const task = args.tracker.start({
id: args.pick.videoId,
- label: args.pick.videoId,
+ label: `${args.channelSlug}/${args.pick.videoId}`,
kind: "download",
});
try {
diff --git a/common/controller/transcribeOneFromQueue.ts b/common/controller/transcribeOneFromQueue.ts
@@ -42,6 +42,10 @@ export type TranscribeOneOptions = {
// list still governs future runs via the append on failure below.
failedSet?: ReadonlySet<string>;
tracker?: TaskTracker;
+ // Human label for the per-task progress row. Defaults to the videoId (fine for
+ // a channel-scoped batch); the cross-channel auto-runner passes
+ // "<channel>/<videoId>" so each task shows which channel it's on.
+ taskLabel?: string;
onLog?: (msg: string) => void;
signal?: AbortSignal;
drainSignal?: AbortSignal;
@@ -57,6 +61,7 @@ export async function transcribeOneFromQueue({
strict = false,
failedSet,
tracker,
+ taskLabel,
onLog,
signal,
drainSignal,
@@ -106,7 +111,7 @@ export async function transcribeOneFromQueue({
strictAudio: strict,
tracker,
taskId: videoId,
- taskLabel: videoId,
+ taskLabel: taskLabel ?? videoId,
onLog: log,
signal,
drainSignal,
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,7 +1,9 @@
# Changelog
## [Unreleased]
+- **"Stop & keep progress" no longer mislabels the paused video as a failed transcription.** Using **Stop & keep progress** on a busy parakeet worker (or any partial-capable engine) sends the engine a graceful SIGTERM so it stops after the current window and the video resumes next run. But if the engine took longer than execa's 5-second force-kill window to exit — which a parakeet window routinely does, since finishing/stitching one ~480s window outlasts 5s — execa force-SIGKILLed it and the resulting "Command was killed with SIGTERM … forcefully terminated after 5000 milliseconds" error escaped the pause handling: it was treated as a genuine transcription failure and the video was written to the channel's `failed-transcriptions` file *permanently* (so even though its completed windows were cached for resume, it was skipped as "failed" on every later run). The transcribe path now recognizes that a force-killed **requested pause** is still a pause, not a failure — it returns the `paused` outcome (a skip, not a failure), so nothing lands in `failed-transcriptions` and the next "Transcribe missing" resumes it from the cached windows. Hard **Cancel** and **Drain** were never affected (their abort signal already classifies the kill as a skip). A video wrongly blacklisted by the old behavior won't auto-prune (it has real audio) — clear it with the channel's **Clear failed transcriptions** action to retry. See `common/controller/transcribeOne.ts`.
- **New Auto-queue: automatically transcribe (and download) across all channels by a configurable priority policy, instead of running one channel batch at a time.** Previously the only way to process pending work was to manually fire a per-channel batch (e.g. *Transcribe missing* on one channel), and since every transcription job serialized on a single queue, a batch ran to completion before any other channel got a turn — there was no way to say "do cornbreadman first, then fall back to hasanabi." The new **Auto-queue** page (under Pool → Auto-queue) adds two always-on runners, **auto-transcribe** and **auto-download**, each driven by a **policy tree**: order rules top-to-bottom for **strict** priority, or wrap rules in a group set to **round-robin** or **weighted-fair** (smooth weighted round-robin) to *alternate* between rulesets. A rule (leaf) matches a **channel**, a whole **platform**, or **all** channels, optionally narrowed to a snapshot **bucket** (e.g. prioritize `failedListed` retries over fresh `downloadedNoTranscript`), and any rule or group can carry a **max-workers** cap (a saturated subtree falls through to the next-priority sibling, like an HTB ceil). The highest-priority channel with available work claims the **next freed worker slot** — non-destructive, so a higher-priority video never kills an in-flight transcription, it just wins the next slot; when a channel's work runs out the runner falls back automatically. Transcription concurrency is bounded by the worker pool's eligible slots (so policy decides *which* video runs, the pool decides *how many*); downloads have no pool, so the runner gates to **one download per platform at a time**, matching the per-platform serial queue's politeness. Each runner is a real, drainable/cancellable job (visible on the Jobs pages), and the Auto-queue page shows live per-rule pending counts and a recent-pick log. Independent of the sync **Schedule** (which only decides *when* to re-fetch a channel) — manual batches keep working alongside it. Policies live in `settings.json` under `autoQueue` (defensively sanitized like `syncScheduler`); fairness cursors persist in `transcripts/.auto-queue/state.json`. See `common/jobs/autoQueuePolicy.ts` (pure selection engine + unit tests), `common/controller/autoRunner.ts`, `common/controller/transcribeOneFromQueue.ts` (shared per-video gating, also used by the existing whisper batch), and `editor/app/auto-queue/*`.
+ - **Auto-queue refinements:** (1) the auto-transcribe runner now honors disabled workers — it no longer used CPU workers you'd turned off via Workers → *Set as default*. The runner started at boot and called the worker pool's `reconfigure()` before anything triggered the pool's lazy init, which set the pool's `initialized` flag *without* applying the saved `.worker-defaults.json` arrangement, so disabled workers came back enabled. The runner no longer pre-empts that init (it relies on the pool's own first-use init, which applies both settings and the saved default). (2) The runners no longer show up under a generic **"Other"** group on the **Active Jobs** screen — channel-less jobs are now grouped into their own labeled sections ("Auto-transcribe" / "Auto-download") via a shared `jobKindLabel` map (also used for friendlier kind labels in the running-jobs list), and each in-flight item is labeled `channel/videoId` so you can see which channel it's on. See `common/controller/autoRunner.ts`, `editor/app/jobs/jobKindLabels.ts`, and `editor/app/jobs/components/ActiveJobsLive.tsx`.
- **Charts can now track content *added to the sites* over time, not just when creators uploaded it.** The chart engine previously only binned the time axis on a video's **upload date**. Two acquisition dates are now recorded per video — when *we* downloaded it and when *we* transcribed it — and the chart editor's X-axis gains a **Date field** selector (Uploaded / Downloaded / Transcribed) alongside the bin. A new **Content added** preset group ships three ready charts (cumulative *Library growth (added)*, cumulative *Transcribed over time*, and *Added per month* stacked by channel), and the default dashboard now includes the cumulative library-growth-by-acquisition chart so the progress view is present out of the box. Everything reuses the existing charts UI, so per-channel filtering, cumulative curves, CSV/PNG export, and shareable URLs all work unchanged. Acquisition dates come from the per-video `download-outcome.json` (`finishedAt`) and a new `transcribe-outcome.json` sidecar written when a transcript is finalized; the stats build falls back to file mtimes for content added before the sidecars existed. Requires a one-time data rebuild (`build:index` + `build:stats`) on the bumped `STATS_SCHEMA_VERSION`. See `common/lib/{stats,chartConfig,chartAggregate,chartShare,transcribeOutcome}.ts`, `common/controller/{buildStats,transcribeOne}.ts`, and `common/components/charts/ChartConfigEditor.tsx`.
- **The Schedule page is now a one-stop editor for per-channel sync cadence.** The `/scheduler` page used to be read-only — you could see each channel's interval, last sync, and next-due time, but to *change* a cadence you had to open that channel's editor (Source → Auto-sync), one channel at a time, and the headline global toggles lived only in Settings. Now each row's **Interval** cell is an inline editor: pick a preset (Default / Off / Every 10–30m / Hourly / 6h / 12h / Daily / Weekly) **or** choose **Custom (minutes)…** and type an exact minute count, then **Save** — writing just `syncIntervalMinutes` to that channel's `config.json` and leaving every other field untouched (it does *not* go through the full channel-form merge). The page also gained a **Global controls** block to toggle the master **enable**, the **default interval**, and the **internal heartbeat** right there (the advanced knobs — concurrency, quiet hours, backoff — still link out to Settings). The underlying due logic is unchanged: a channel auto-syncs on the next heartbeat once `now − lastSyncedAt ≥ its interval`. The live status columns keep polling every 5s, but each row's editor holds its own state seeded once from the stored value, so a refresh can't clobber an in-progress edit. The preset list is shared with the channel editor (`editor/app/scheduler/intervalPresets.ts`), and `GET /api/scheduler/status` now carries each channel's raw `configuredIntervalMinutes` so the editor can tell *inherit-default* from *explicit-off* from *explicit-minutes*. See `editor/app/scheduler/actions.ts` and `editor/app/scheduler/components/{ChannelIntervalEditor,SchedulerSettingsForm}.tsx`.
- **Scheduled sync can now run without an external cron job.** The sync scheduler previously only fired when an OS cron entry POSTed to `/api/scheduler/tick` (via `pnpm sync:tick`) — fine on a server, but a chore to set up just to call a function the editor already hosts in-process. The editor can now drive its own heartbeat through a **Next.js instrumentation hook** (`editor/instrumentation.ts`): on server startup it arms a single in-process timer that calls `runSchedulerTick()` directly — no HTTP, no cron, no token. Turn it on with **Settings → Sync scheduler → Internal heartbeat (seconds)**: `0` = off (keep using an external cron heartbeat), any positive value is clamped to `[15, 3600]`s and is the cadence the editor ticks itself at; the `SYNC_HEARTBEAT_SECONDS` env var overrides the setting at runtime. The timer is a self-rescheduling, `unref`'d `setTimeout` loop (so it never holds the process open and never overlaps a tick), re-reading the cadence each fire so a change takes effect on the next tick — though turning it on *from 0* needs a restart, since the timer is armed once at boot. It's modeled on the existing snapshot-scheduler timer and reuses the already-overlap-guarded `runSchedulerTick()`, so internal and external heartbeats are interchangeable and may even coexist. The **Schedule** page header now reports how ticks are driven ("internal heartbeat every N" vs. "external heartbeat (cron)"), and `GET /api/scheduler/status` carries the effective `heartbeatSeconds`. Defaults to off, so dev/test and existing cron installs are unchanged. One caveat for multi-instance deployments: the timer runs once *per server instance*, so a cluster against one data dir should set `SYNC_HEARTBEAT_SECONDS=0` on all but one — see `SCHEDULED_SYNC.md`.
diff --git a/editor/app/jobs/components/ActiveJobsLive.tsx b/editor/app/jobs/components/ActiveJobsLive.tsx
@@ -3,6 +3,7 @@
import Link from "next/link";
import { useEffect, useState } from "react";
import type { ActiveJobsPayload } from "../active/buildActiveJobs";
+import { jobKindLabel } from "../jobKindLabels";
import { RunningJobsList, type RunningJobsListItem } from "./RunningJobsList";
// Polls /api/jobs/active so per-task progress bars advance live (the server
@@ -38,14 +39,18 @@ export function ActiveJobsLive({ initial }: { initial: ActiveJobsPayload }) {
const { jobs, channels } = payload;
const jobsBySlug = new Map<string, RunningJobsListItem[]>();
- const unassigned: RunningJobsListItem[] = [];
+ // Channel-less jobs (e.g. the cross-channel auto-queue runners) are grouped by
+ // kind so each gets its own labeled section instead of a generic "Other".
+ const jobsByKind = new Map<string, RunningJobsListItem[]>();
for (const job of jobs) {
if (job.channelSlug) {
const list = jobsBySlug.get(job.channelSlug) ?? [];
list.push(job);
jobsBySlug.set(job.channelSlug, list);
} else {
- unassigned.push(job);
+ const list = jobsByKind.get(job.kind) ?? [];
+ list.push(job);
+ jobsByKind.set(job.kind, list);
}
}
@@ -81,15 +86,19 @@ export function ActiveJobsLive({ initial }: { initial: ActiveJobsPayload }) {
</section>
);
})}
- {unassigned.length > 0 && (
- <section
- aria-label="Active jobs without a channel"
- className="flex flex-col gap-3 border border-zinc-200 dark:border-zinc-800 rounded-md p-3 bg-white dark:bg-zinc-900"
- >
- <h2 className="text-base font-medium">Other</h2>
- <RunningJobsList jobs={unassigned} />
- </section>
- )}
+ {[...jobsByKind.entries()].map(([kind, kindJobs]) => {
+ const label = jobKindLabel(kind);
+ return (
+ <section
+ key={kind}
+ aria-label={`System jobs: ${label}`}
+ className="flex flex-col gap-3 border border-zinc-200 dark:border-zinc-800 rounded-md p-3 bg-white dark:bg-zinc-900"
+ >
+ <h2 className="text-base font-medium">{label}</h2>
+ <RunningJobsList jobs={kindJobs} />
+ </section>
+ );
+ })}
</div>
);
}
diff --git a/editor/app/jobs/components/RunningJobsList.tsx b/editor/app/jobs/components/RunningJobsList.tsx
@@ -4,6 +4,7 @@ import Link from "next/link";
import { useEffect, useState } from "react";
import { formatDuration } from "yt-dlp-transcript-common/lib/format";
import { JobLogTail } from "../[id]/components/JobLogTail";
+import { jobKindLabel } from "../jobKindLabels";
import { DrainJobButton } from "./DrainJobButton";
import { CancelJobButton } from "./CancelJobButton";
import { BookmarkJobButton } from "./BookmarkJobButton";
@@ -103,7 +104,9 @@ function JobRow({
>
{job.status}
</span>
- <span className="font-mono text-xs">{job.kind}</span>
+ <span className="text-xs font-medium" title={job.kind}>
+ {jobKindLabel(job.kind)}
+ </span>
<Link
href={`/jobs/${job.id}`}
className="font-mono text-xs underline hover:text-zinc-900 dark:hover:text-zinc-100"
diff --git a/editor/app/jobs/jobKindLabels.ts b/editor/app/jobs/jobKindLabels.ts
@@ -0,0 +1,19 @@
+// Human-readable labels for job `kind` values. The registry stores terse
+// machine kinds (e.g. "whisper-all", "auto-transcribe"); the UI shows these.
+// Unknown kinds fall back to the raw value so a new kind is never invisible.
+const JOB_KIND_LABELS: Record<string, string> = {
+ "auto-transcribe": "Auto-transcribe",
+ "auto-download": "Auto-download",
+ "whisper-all": "Transcribe all",
+ "whisper-bucket-downloaded-no-transcript": "Transcribe downloaded audio",
+ "download-from-playlist": "Download from playlist",
+ "download-missing": "Download missing",
+ "download-missing-subs": "Download missing subs",
+ "import-one": "Import video",
+ "retry-bucket": "Retry",
+ sync: "Sync",
+};
+
+export function jobKindLabel(kind: string): string {
+ return JOB_KIND_LABELS[kind] ?? kind;
+}
diff --git a/editor/e2e/auto-queue.spec.ts b/editor/e2e/auto-queue.spec.ts
@@ -44,6 +44,12 @@ const ONE_WORKER = [
},
];
+// Two enabled local workers; tests can then disable one via a saved default.
+const TWO_WORKERS = [
+ { id: "gpu", name: "GPU", kind: "local", enabled: true, priority: 0, appId: "whisper-cpp", config: {} },
+ { id: "cpu", name: "CPU", kind: "local", enabled: true, priority: 1, appId: "whisper-cpp", config: {} },
+];
+
async function makeChannel(slug: string, ids: string[]) {
const root = resolvePath(`test-transcripts/channels/${slug}`);
await mkdir(`${root}/data`, { recursive: true });
@@ -238,6 +244,53 @@ test("strict priority: drains the high-priority channel first, then falls back",
expect(await allTranscribed("beta", ["b1", "b2"])).toBe(true);
});
+test("respects the saved worker default: a disabled worker is never used", async ({
+ request,
+}) => {
+ await resetData(null);
+ await makeChannel("alpha", ["a1", "a2", "a3", "a4"]);
+
+ // Both workers are enabled in settings, but the saved default
+ // (.worker-defaults.json) enables only gpu — cpu must stay disabled, and the
+ // runner must honor that (regression for the boot-time reconfigure() bug that
+ // bypassed applyDefaults()).
+ await mkdir(resolvePath("test-transcripts/.workers"), { recursive: true });
+ await writeFile(
+ resolvePath("test-transcripts/.workers/defaults.json"),
+ JSON.stringify({ enabledWorkerIds: ["gpu"] }),
+ );
+
+ const root: Group = {
+ id: "root",
+ mode: "strict",
+ children: [{ id: "leaf-alpha", match: { type: "channel", value: "alpha" } }],
+ };
+ await writeSettings({
+ adminTitle: "Test Admin",
+ maxTranscriptPageBytes: 8388608,
+ sleepBetweenDownloadsSeconds: 0,
+ minFreeDiskGB: 0,
+ workers: TWO_WORKERS,
+ // No runner cap, so without the fix the runner would use BOTH workers.
+ autoQueue: transcriptionAutoQueue(root, null),
+ });
+
+ await startRunner(request);
+
+ const ids = ["a1", "a2", "a3", "a4"];
+ await expect
+ .poll(async () => allTranscribed("alpha", ids), { timeout: 60_000 })
+ .toBe(true);
+
+ // Every transcription ran on gpu — cpu (disabled by the saved default) was never used.
+ for (const id of ids) {
+ const outcome = await readJson<{ worker?: { id?: string } }>(
+ `test-transcripts/channels/alpha/data/${id}/transcribe-outcome.json`,
+ );
+ expect(outcome.worker?.id).toBe("gpu");
+ }
+});
+
test("round-robin: alternates between the two channels", async ({ request }) => {
await resetData(null);
await makeChannel("alpha", ["a1", "a2"]);
@@ -364,6 +417,38 @@ test("status: reports the runner running and snapshot-derived pending counts", a
expect(status.transcription.picks.length).toBe(0);
});
+test("Active Jobs: the runner shows in its own labeled section, not Other", async ({
+ page,
+}) => {
+ await resetData(null);
+ await makeChannel("alpha", ["a1", "a2"]);
+ const root: Group = {
+ id: "root",
+ mode: "strict",
+ children: [{ id: "leaf-alpha", match: { type: "channel", value: "alpha" } }],
+ };
+ await writeSettings({
+ adminTitle: "Test Admin",
+ maxTranscriptPageBytes: 8388608,
+ sleepBetweenDownloadsSeconds: 0,
+ minFreeDiskGB: 0,
+ workers: ONE_WORKER,
+ autoQueue: transcriptionAutoQueue(root),
+ });
+
+ // Start the perpetual runner; it stays "running" (active) even once idle.
+ await page.request.post(`${baseUrl}/api/auto-queue/control`, {
+ data: { kind: "transcription", action: "start" },
+ });
+
+ await page.goto("/jobs/active");
+ await expect(
+ page.getByRole("heading", { name: "Auto-transcribe" }),
+ ).toBeVisible();
+ // The runner is no longer dumped under a generic "Other" section.
+ await expect(page.getByRole("heading", { name: "Other" })).toHaveCount(0);
+});
+
test("download: prioritizes channels across the per-platform queue", async ({
request,
}) => {