Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 133b5a9965761e5cd19ccba97cc571f5989d81c1
parent 90028d915e239c83fe25c84ec6a5ae03c7461c78
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 22 Jun 2026 15:59:48 -0400

Auto-queue: respect disabled workers; label runner in Active Jobs

Two fixes from using the feature:

- Worker respect: the auto-transcribe runner used workers the operator had
  disabled via Workers -> "Set as default". The runner started at boot and
  called getWorkerPool().reconfigure() before anything triggered the pool's
  lazy ensureInit(), which set the pool's `initialized` flag WITHOUT applying
  the saved .worker-defaults.json arrangement — so disabled workers came back
  enabled. Removed that pre-emptive reconfigure(); the pool now self-inits on
  first use (eligibleSlots/acquire), applying settings AND the saved default.

- Active Jobs labeling: channel-less jobs (the cross-channel runners) were
  dumped under a generic "Other" section. They now get their own labeled
  sections ("Auto-transcribe"/"Auto-download") via a shared jobKindLabel map
  (also used for friendlier kind labels in the running-jobs list), and each
  in-flight item is labeled "channel/videoId" so you can see its channel.

New e2e: a saved default disabling cpu keeps cpu unused (asserts every
transcribe-outcome.json worker.id === gpu); the runner shows in an
"Auto-transcribe" section, not "Other". Existing drain/workers/jobs-order
specs and the production build stay green.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/controller/autoRunner.ts | 18+++++++++++++-----
Mcommon/controller/transcribeOneFromQueue.ts | 7++++++-
Meditor/CHANGELOG.md | 2++
Meditor/app/jobs/components/ActiveJobsLive.tsx | 31++++++++++++++++++++-----------
Meditor/app/jobs/components/RunningJobsList.tsx | 5++++-
Aeditor/app/jobs/jobKindLabels.ts | 19+++++++++++++++++++
Meditor/e2e/auto-queue.spec.ts | 85+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
7 files changed, 149 insertions(+), 18 deletions(-)

diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts @@ -287,10 +287,15 @@ async function runLoop( let metaCache: ChannelMeta[] = []; let metaAt = 0; - // Seed/refresh the global worker pool from current settings so eligibleSlots() - // and acquire() see the latest worker config (mirrors runWhisperBatch). After - // an e2e cache-reset the pool is recreated and re-seeds here. - getWorkerPool().reconfigure(); + // NOTE: do NOT call getWorkerPool().reconfigure() here. At server boot the + // runner is the first thing to touch the pool, and reconfigure() sets the + // pool's `initialized` flag WITHOUT applying the saved "Set as default" + // arrangement (.worker-defaults.json) — so a pre-empted ensureInit() would + // never disable the workers the operator turned off, and the runner would use + // them anyway. Leaving it to the pool's lazy ensureInit() (triggered by the + // first eligibleSlots()/acquire() below) seeds settings AND applies defaults + // exactly once. Live settings changes are pushed by saveSettingsAction's own + // reconfigure(applyEnabled:true), so nothing is lost by not refreshing here. onLog(`Auto-${kind} runner started.`); @@ -450,6 +455,9 @@ async function launchUnit(args: LaunchArgs): Promise< audioFilename: "audio.mp3", strict: false, tracker: args.tracker, + // Show which channel each in-flight transcription belongs to (the runner + // is cross-channel, so the bare videoId wouldn't say). + taskLabel: `${args.channelSlug}/${args.pick.videoId}`, onLog: args.onLog, signal: args.signal, }); @@ -475,7 +483,7 @@ async function launchUnit(args: LaunchArgs): Promise< const settings = getSettings(); const task = args.tracker.start({ id: args.pick.videoId, - label: args.pick.videoId, + label: `${args.channelSlug}/${args.pick.videoId}`, kind: "download", }); try { diff --git a/common/controller/transcribeOneFromQueue.ts b/common/controller/transcribeOneFromQueue.ts @@ -42,6 +42,10 @@ export type TranscribeOneOptions = { // list still governs future runs via the append on failure below. failedSet?: ReadonlySet<string>; tracker?: TaskTracker; + // Human label for the per-task progress row. Defaults to the videoId (fine for + // a channel-scoped batch); the cross-channel auto-runner passes + // "<channel>/<videoId>" so each task shows which channel it's on. + taskLabel?: string; onLog?: (msg: string) => void; signal?: AbortSignal; drainSignal?: AbortSignal; @@ -57,6 +61,7 @@ export async function transcribeOneFromQueue({ strict = false, failedSet, tracker, + taskLabel, onLog, signal, drainSignal, @@ -106,7 +111,7 @@ export async function transcribeOneFromQueue({ strictAudio: strict, tracker, taskId: videoId, - taskLabel: videoId, + taskLabel: taskLabel ?? videoId, onLog: log, signal, drainSignal, diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,7 +1,9 @@ # Changelog ## [Unreleased] +- **"Stop & keep progress" no longer mislabels the paused video as a failed transcription.** Using **Stop & keep progress** on a busy parakeet worker (or any partial-capable engine) sends the engine a graceful SIGTERM so it stops after the current window and the video resumes next run. But if the engine took longer than execa's 5-second force-kill window to exit — which a parakeet window routinely does, since finishing/stitching one ~480s window outlasts 5s — execa force-SIGKILLed it and the resulting "Command was killed with SIGTERM … forcefully terminated after 5000 milliseconds" error escaped the pause handling: it was treated as a genuine transcription failure and the video was written to the channel's `failed-transcriptions` file *permanently* (so even though its completed windows were cached for resume, it was skipped as "failed" on every later run). The transcribe path now recognizes that a force-killed **requested pause** is still a pause, not a failure — it returns the `paused` outcome (a skip, not a failure), so nothing lands in `failed-transcriptions` and the next "Transcribe missing" resumes it from the cached windows. Hard **Cancel** and **Drain** were never affected (their abort signal already classifies the kill as a skip). A video wrongly blacklisted by the old behavior won't auto-prune (it has real audio) — clear it with the channel's **Clear failed transcriptions** action to retry. See `common/controller/transcribeOne.ts`. - **New Auto-queue: automatically transcribe (and download) across all channels by a configurable priority policy, instead of running one channel batch at a time.** Previously the only way to process pending work was to manually fire a per-channel batch (e.g. *Transcribe missing* on one channel), and since every transcription job serialized on a single queue, a batch ran to completion before any other channel got a turn — there was no way to say "do cornbreadman first, then fall back to hasanabi." The new **Auto-queue** page (under Pool → Auto-queue) adds two always-on runners, **auto-transcribe** and **auto-download**, each driven by a **policy tree**: order rules top-to-bottom for **strict** priority, or wrap rules in a group set to **round-robin** or **weighted-fair** (smooth weighted round-robin) to *alternate* between rulesets. A rule (leaf) matches a **channel**, a whole **platform**, or **all** channels, optionally narrowed to a snapshot **bucket** (e.g. prioritize `failedListed` retries over fresh `downloadedNoTranscript`), and any rule or group can carry a **max-workers** cap (a saturated subtree falls through to the next-priority sibling, like an HTB ceil). The highest-priority channel with available work claims the **next freed worker slot** — non-destructive, so a higher-priority video never kills an in-flight transcription, it just wins the next slot; when a channel's work runs out the runner falls back automatically. Transcription concurrency is bounded by the worker pool's eligible slots (so policy decides *which* video runs, the pool decides *how many*); downloads have no pool, so the runner gates to **one download per platform at a time**, matching the per-platform serial queue's politeness. Each runner is a real, drainable/cancellable job (visible on the Jobs pages), and the Auto-queue page shows live per-rule pending counts and a recent-pick log. Independent of the sync **Schedule** (which only decides *when* to re-fetch a channel) — manual batches keep working alongside it. Policies live in `settings.json` under `autoQueue` (defensively sanitized like `syncScheduler`); fairness cursors persist in `transcripts/.auto-queue/state.json`. See `common/jobs/autoQueuePolicy.ts` (pure selection engine + unit tests), `common/controller/autoRunner.ts`, `common/controller/transcribeOneFromQueue.ts` (shared per-video gating, also used by the existing whisper batch), and `editor/app/auto-queue/*`. + - **Auto-queue refinements:** (1) the auto-transcribe runner now honors disabled workers — it no longer used CPU workers you'd turned off via Workers → *Set as default*. The runner started at boot and called the worker pool's `reconfigure()` before anything triggered the pool's lazy init, which set the pool's `initialized` flag *without* applying the saved `.worker-defaults.json` arrangement, so disabled workers came back enabled. The runner no longer pre-empts that init (it relies on the pool's own first-use init, which applies both settings and the saved default). (2) The runners no longer show up under a generic **"Other"** group on the **Active Jobs** screen — channel-less jobs are now grouped into their own labeled sections ("Auto-transcribe" / "Auto-download") via a shared `jobKindLabel` map (also used for friendlier kind labels in the running-jobs list), and each in-flight item is labeled `channel/videoId` so you can see which channel it's on. See `common/controller/autoRunner.ts`, `editor/app/jobs/jobKindLabels.ts`, and `editor/app/jobs/components/ActiveJobsLive.tsx`. - **Charts can now track content *added to the sites* over time, not just when creators uploaded it.** The chart engine previously only binned the time axis on a video's **upload date**. Two acquisition dates are now recorded per video — when *we* downloaded it and when *we* transcribed it — and the chart editor's X-axis gains a **Date field** selector (Uploaded / Downloaded / Transcribed) alongside the bin. A new **Content added** preset group ships three ready charts (cumulative *Library growth (added)*, cumulative *Transcribed over time*, and *Added per month* stacked by channel), and the default dashboard now includes the cumulative library-growth-by-acquisition chart so the progress view is present out of the box. Everything reuses the existing charts UI, so per-channel filtering, cumulative curves, CSV/PNG export, and shareable URLs all work unchanged. Acquisition dates come from the per-video `download-outcome.json` (`finishedAt`) and a new `transcribe-outcome.json` sidecar written when a transcript is finalized; the stats build falls back to file mtimes for content added before the sidecars existed. Requires a one-time data rebuild (`build:index` + `build:stats`) on the bumped `STATS_SCHEMA_VERSION`. See `common/lib/{stats,chartConfig,chartAggregate,chartShare,transcribeOutcome}.ts`, `common/controller/{buildStats,transcribeOne}.ts`, and `common/components/charts/ChartConfigEditor.tsx`. - **The Schedule page is now a one-stop editor for per-channel sync cadence.** The `/scheduler` page used to be read-only — you could see each channel's interval, last sync, and next-due time, but to *change* a cadence you had to open that channel's editor (Source → Auto-sync), one channel at a time, and the headline global toggles lived only in Settings. Now each row's **Interval** cell is an inline editor: pick a preset (Default / Off / Every 10–30m / Hourly / 6h / 12h / Daily / Weekly) **or** choose **Custom (minutes)…** and type an exact minute count, then **Save** — writing just `syncIntervalMinutes` to that channel's `config.json` and leaving every other field untouched (it does *not* go through the full channel-form merge). The page also gained a **Global controls** block to toggle the master **enable**, the **default interval**, and the **internal heartbeat** right there (the advanced knobs — concurrency, quiet hours, backoff — still link out to Settings). The underlying due logic is unchanged: a channel auto-syncs on the next heartbeat once `now − lastSyncedAt ≥ its interval`. The live status columns keep polling every 5s, but each row's editor holds its own state seeded once from the stored value, so a refresh can't clobber an in-progress edit. The preset list is shared with the channel editor (`editor/app/scheduler/intervalPresets.ts`), and `GET /api/scheduler/status` now carries each channel's raw `configuredIntervalMinutes` so the editor can tell *inherit-default* from *explicit-off* from *explicit-minutes*. See `editor/app/scheduler/actions.ts` and `editor/app/scheduler/components/{ChannelIntervalEditor,SchedulerSettingsForm}.tsx`. - **Scheduled sync can now run without an external cron job.** The sync scheduler previously only fired when an OS cron entry POSTed to `/api/scheduler/tick` (via `pnpm sync:tick`) — fine on a server, but a chore to set up just to call a function the editor already hosts in-process. The editor can now drive its own heartbeat through a **Next.js instrumentation hook** (`editor/instrumentation.ts`): on server startup it arms a single in-process timer that calls `runSchedulerTick()` directly — no HTTP, no cron, no token. Turn it on with **Settings → Sync scheduler → Internal heartbeat (seconds)**: `0` = off (keep using an external cron heartbeat), any positive value is clamped to `[15, 3600]`s and is the cadence the editor ticks itself at; the `SYNC_HEARTBEAT_SECONDS` env var overrides the setting at runtime. The timer is a self-rescheduling, `unref`'d `setTimeout` loop (so it never holds the process open and never overlaps a tick), re-reading the cadence each fire so a change takes effect on the next tick — though turning it on *from 0* needs a restart, since the timer is armed once at boot. It's modeled on the existing snapshot-scheduler timer and reuses the already-overlap-guarded `runSchedulerTick()`, so internal and external heartbeats are interchangeable and may even coexist. The **Schedule** page header now reports how ticks are driven ("internal heartbeat every N" vs. "external heartbeat (cron)"), and `GET /api/scheduler/status` carries the effective `heartbeatSeconds`. Defaults to off, so dev/test and existing cron installs are unchanged. One caveat for multi-instance deployments: the timer runs once *per server instance*, so a cluster against one data dir should set `SYNC_HEARTBEAT_SECONDS=0` on all but one — see `SCHEDULED_SYNC.md`. diff --git a/editor/app/jobs/components/ActiveJobsLive.tsx b/editor/app/jobs/components/ActiveJobsLive.tsx @@ -3,6 +3,7 @@ import Link from "next/link"; import { useEffect, useState } from "react"; import type { ActiveJobsPayload } from "../active/buildActiveJobs"; +import { jobKindLabel } from "../jobKindLabels"; import { RunningJobsList, type RunningJobsListItem } from "./RunningJobsList"; // Polls /api/jobs/active so per-task progress bars advance live (the server @@ -38,14 +39,18 @@ export function ActiveJobsLive({ initial }: { initial: ActiveJobsPayload }) { const { jobs, channels } = payload; const jobsBySlug = new Map<string, RunningJobsListItem[]>(); - const unassigned: RunningJobsListItem[] = []; + // Channel-less jobs (e.g. the cross-channel auto-queue runners) are grouped by + // kind so each gets its own labeled section instead of a generic "Other". + const jobsByKind = new Map<string, RunningJobsListItem[]>(); for (const job of jobs) { if (job.channelSlug) { const list = jobsBySlug.get(job.channelSlug) ?? []; list.push(job); jobsBySlug.set(job.channelSlug, list); } else { - unassigned.push(job); + const list = jobsByKind.get(job.kind) ?? []; + list.push(job); + jobsByKind.set(job.kind, list); } } @@ -81,15 +86,19 @@ export function ActiveJobsLive({ initial }: { initial: ActiveJobsPayload }) { </section> ); })} - {unassigned.length > 0 && ( - <section - aria-label="Active jobs without a channel" - className="flex flex-col gap-3 border border-zinc-200 dark:border-zinc-800 rounded-md p-3 bg-white dark:bg-zinc-900" - > - <h2 className="text-base font-medium">Other</h2> - <RunningJobsList jobs={unassigned} /> - </section> - )} + {[...jobsByKind.entries()].map(([kind, kindJobs]) => { + const label = jobKindLabel(kind); + return ( + <section + key={kind} + aria-label={`System jobs: ${label}`} + className="flex flex-col gap-3 border border-zinc-200 dark:border-zinc-800 rounded-md p-3 bg-white dark:bg-zinc-900" + > + <h2 className="text-base font-medium">{label}</h2> + <RunningJobsList jobs={kindJobs} /> + </section> + ); + })} </div> ); } diff --git a/editor/app/jobs/components/RunningJobsList.tsx b/editor/app/jobs/components/RunningJobsList.tsx @@ -4,6 +4,7 @@ import Link from "next/link"; import { useEffect, useState } from "react"; import { formatDuration } from "yt-dlp-transcript-common/lib/format"; import { JobLogTail } from "../[id]/components/JobLogTail"; +import { jobKindLabel } from "../jobKindLabels"; import { DrainJobButton } from "./DrainJobButton"; import { CancelJobButton } from "./CancelJobButton"; import { BookmarkJobButton } from "./BookmarkJobButton"; @@ -103,7 +104,9 @@ function JobRow({ > {job.status} </span> - <span className="font-mono text-xs">{job.kind}</span> + <span className="text-xs font-medium" title={job.kind}> + {jobKindLabel(job.kind)} + </span> <Link href={`/jobs/${job.id}`} className="font-mono text-xs underline hover:text-zinc-900 dark:hover:text-zinc-100" diff --git a/editor/app/jobs/jobKindLabels.ts b/editor/app/jobs/jobKindLabels.ts @@ -0,0 +1,19 @@ +// Human-readable labels for job `kind` values. The registry stores terse +// machine kinds (e.g. "whisper-all", "auto-transcribe"); the UI shows these. +// Unknown kinds fall back to the raw value so a new kind is never invisible. +const JOB_KIND_LABELS: Record<string, string> = { + "auto-transcribe": "Auto-transcribe", + "auto-download": "Auto-download", + "whisper-all": "Transcribe all", + "whisper-bucket-downloaded-no-transcript": "Transcribe downloaded audio", + "download-from-playlist": "Download from playlist", + "download-missing": "Download missing", + "download-missing-subs": "Download missing subs", + "import-one": "Import video", + "retry-bucket": "Retry", + sync: "Sync", +}; + +export function jobKindLabel(kind: string): string { + return JOB_KIND_LABELS[kind] ?? kind; +} diff --git a/editor/e2e/auto-queue.spec.ts b/editor/e2e/auto-queue.spec.ts @@ -44,6 +44,12 @@ const ONE_WORKER = [ }, ]; +// Two enabled local workers; tests can then disable one via a saved default. +const TWO_WORKERS = [ + { id: "gpu", name: "GPU", kind: "local", enabled: true, priority: 0, appId: "whisper-cpp", config: {} }, + { id: "cpu", name: "CPU", kind: "local", enabled: true, priority: 1, appId: "whisper-cpp", config: {} }, +]; + async function makeChannel(slug: string, ids: string[]) { const root = resolvePath(`test-transcripts/channels/${slug}`); await mkdir(`${root}/data`, { recursive: true }); @@ -238,6 +244,53 @@ test("strict priority: drains the high-priority channel first, then falls back", expect(await allTranscribed("beta", ["b1", "b2"])).toBe(true); }); +test("respects the saved worker default: a disabled worker is never used", async ({ + request, +}) => { + await resetData(null); + await makeChannel("alpha", ["a1", "a2", "a3", "a4"]); + + // Both workers are enabled in settings, but the saved default + // (.worker-defaults.json) enables only gpu — cpu must stay disabled, and the + // runner must honor that (regression for the boot-time reconfigure() bug that + // bypassed applyDefaults()). + await mkdir(resolvePath("test-transcripts/.workers"), { recursive: true }); + await writeFile( + resolvePath("test-transcripts/.workers/defaults.json"), + JSON.stringify({ enabledWorkerIds: ["gpu"] }), + ); + + const root: Group = { + id: "root", + mode: "strict", + children: [{ id: "leaf-alpha", match: { type: "channel", value: "alpha" } }], + }; + await writeSettings({ + adminTitle: "Test Admin", + maxTranscriptPageBytes: 8388608, + sleepBetweenDownloadsSeconds: 0, + minFreeDiskGB: 0, + workers: TWO_WORKERS, + // No runner cap, so without the fix the runner would use BOTH workers. + autoQueue: transcriptionAutoQueue(root, null), + }); + + await startRunner(request); + + const ids = ["a1", "a2", "a3", "a4"]; + await expect + .poll(async () => allTranscribed("alpha", ids), { timeout: 60_000 }) + .toBe(true); + + // Every transcription ran on gpu — cpu (disabled by the saved default) was never used. + for (const id of ids) { + const outcome = await readJson<{ worker?: { id?: string } }>( + `test-transcripts/channels/alpha/data/${id}/transcribe-outcome.json`, + ); + expect(outcome.worker?.id).toBe("gpu"); + } +}); + test("round-robin: alternates between the two channels", async ({ request }) => { await resetData(null); await makeChannel("alpha", ["a1", "a2"]); @@ -364,6 +417,38 @@ test("status: reports the runner running and snapshot-derived pending counts", a expect(status.transcription.picks.length).toBe(0); }); +test("Active Jobs: the runner shows in its own labeled section, not Other", async ({ + page, +}) => { + await resetData(null); + await makeChannel("alpha", ["a1", "a2"]); + const root: Group = { + id: "root", + mode: "strict", + children: [{ id: "leaf-alpha", match: { type: "channel", value: "alpha" } }], + }; + await writeSettings({ + adminTitle: "Test Admin", + maxTranscriptPageBytes: 8388608, + sleepBetweenDownloadsSeconds: 0, + minFreeDiskGB: 0, + workers: ONE_WORKER, + autoQueue: transcriptionAutoQueue(root), + }); + + // Start the perpetual runner; it stays "running" (active) even once idle. + await page.request.post(`${baseUrl}/api/auto-queue/control`, { + data: { kind: "transcription", action: "start" }, + }); + + await page.goto("/jobs/active"); + await expect( + page.getByRole("heading", { name: "Auto-transcribe" }), + ).toBeVisible(); + // The runner is no longer dumped under a generic "Other" section. + await expect(page.getByRole("heading", { name: "Other" })).toHaveCount(0); +}); + test("download: prioritizes channels across the per-platform queue", async ({ request, }) => {