commit f15e4c9c051b697dca3568fc3aeeab65a410cf61
parent 3cf0a6bc10850e7f11b0a462f32033db0fa4593f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 26 Aug 2026 16:56:50 -0400
editor: an operation's settings live on its operation page
The four fieldsets that configured OPERATIONS move off /settings and onto
/operations/<id>, each as its own <form> with its own action: Digest,
Diarization, Speaker attribution — and the speaker work LANE, which is not any
one operation's and so is rendered per LANE, beside the pause and the sweep the
three speaker operations already share. /operations/diarization now draws
Diarization then Speaker work lane; both attribution pages draw Speaker
attribution then the same lane form; /operations/digest draws Digest alone,
because the digest lane's facts are already in the digest block.
Which form a page gets is switched on the descriptor's `settingsBlock`, never on
the id — so both attribution operations render one form because they name one
block, and an operation added to the registry gets its settings with no route
work.
THE MARKERS ARE GONE. `digestFormPresent`, `diarizationFormPresent`,
`backfillFormPresent` and `attributionFormPresent` existed only because one form
saved everything: an unchecked checkbox is absent from a FormData, so a submit
from a form lacking the block would read every switch as off — disarming a
corpus sweep, dropping the diarization capture lane (which lets Clean-audio
delete held audio), flipping `allowRedownload` on, or arming ~194,000 model
calls. One form per block makes that structural, and `saveSettingsAction` now
carries the four blocks the way it already carried autoQueue: read from
`getSettings()`, untouched. 562 lines out of SettingsForm, 171 out of its action.
Buttons are per block (Save digest settings, Save lane settings, …) because two
forms share a page and an unscoped /save settings/i would be ambiguous. Legends
and every field NAME are verbatim, so the persisted keys and the by-name
selectors are unchanged.
Specs: the three form-driving tests move to e2e/operation-settings.spec.ts and
repoint. The unrelated-save one becomes the genuinely cross-form test it always
described — digest on its page, admin title on /settings — and a new one proves
the property the markers used to buy: saving the diarization form beside the
lane form leaves `backfill.enabled` alone. Attribution and the lane had no
form-driving spec at all before; both have one now. 7/7 pass.
tsc in six packages, common 828/828, editor units 85/85, and every operation
page checked against the fixture: three with a form, download/transcode/
transcription with none, /settings down to its four remaining fieldsets.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Diffstat:
13 files changed, 1098 insertions(+), 901 deletions(-)
diff --git a/editor/app/operations/[id]/page.tsx b/editor/app/operations/[id]/page.tsx
@@ -7,12 +7,22 @@ import { selectableBucketsForKind } from "yt-dlp-transcript-common/jobs/autoQueu
import {
operationCatalog,
operationLabel,
+ type OperationSettingsBlock,
} from "yt-dlp-transcript-common/lib/operations";
+import { getSettings, type SiteSettings } from "yt-dlp-transcript-common/lib/settings";
+import {
+ listDigestApps,
+ type DigestAppDescriptor,
+} from "yt-dlp-transcript-common/lib/digestApps";
import type { AutoQueueKind } from "yt-dlp-transcript-common/jobs/autoQueueState";
import { getRegistry } from "yt-dlp-transcript-common/jobs/registry";
import { buildAutoQueueStatusPayload } from "../status";
import { sweepLaneIdFor, type SweepLaneId } from "../lanes";
import { OperationDetail } from "../components/OperationDetail";
+import { DigestSettingsForm } from "../components/settings/DigestSettingsForm";
+import { DiarizationSettingsForm } from "../components/settings/DiarizationSettingsForm";
+import { AttributionSettingsForm } from "../components/settings/AttributionSettingsForm";
+import { LaneSettingsForm } from "../components/settings/LaneSettingsForm";
export const dynamic = "force-dynamic";
@@ -45,6 +55,37 @@ function descriptorFor(id: string) {
return operationCatalog().find((o) => o.id === id) ?? null;
}
+// THE OPERATION'S OWN SETTINGS FORM, chosen by the block the descriptor
+// DECLARES — not by a table keyed by operation id. That is the same rule
+// `runner` and `sweepLaneIdFor` follow, and it is what lets both attribution
+// operations render one form: they name one block, so they get one switch arm.
+//
+// Exhaustive on purpose (the union is closed): a settings block added to the
+// registry is a type error here rather than a page that quietly renders nothing.
+function settingsFormFor(
+ block: OperationSettingsBlock | undefined,
+ settings: SiteSettings,
+ digestApps: DigestAppDescriptor[],
+) {
+ switch (block) {
+ case "digest":
+ return (
+ <DigestSettingsForm initial={settings.digest} digestApps={digestApps} />
+ );
+ case "diarization":
+ return <DiarizationSettingsForm initial={settings.diarization} />;
+ case "attribution":
+ return (
+ <AttributionSettingsForm
+ initial={settings.attribution}
+ digestApps={digestApps}
+ />
+ );
+ case undefined:
+ return null;
+ }
+}
+
export async function generateMetadata({
params,
}: {
@@ -85,6 +126,27 @@ export default async function OperationPage({
name: c.config.name ?? null,
}));
+ // The two settings slots. Both are built HERE because they need server-only
+ // reads (getSettings, and listDigestApps which reaches process.env and
+ // imports execa) that a client component cannot do.
+ const settings = getSettings();
+ const digestApps =
+ op.settingsBlock === "digest" || op.settingsBlock === "attribution"
+ ? listDigestApps()
+ : [];
+ const operationSettings = settingsFormFor(
+ op.settingsBlock,
+ settings,
+ digestApps,
+ );
+ // PER LANE, not per operation: `settings.backfill` governs BACKFILL_QUEUE,
+ // which three operations share, so this same form is drawn on all three of
+ // their pages beside the pause and sweep they also share. The digest lane has
+ // no equivalent — its lane facts (`sweepEnabled`, order and reach) live in the
+ // digest block and the OrderReach control.
+ const laneSettings =
+ laneId === "backfill" ? <LaneSettingsForm initial={settings.backfill} /> : null;
+
const laneJobKinds =
runnerKind || !laneId ? [] : JOB_KINDS_BY_LANE[laneId];
const activeJobs =
@@ -130,6 +192,8 @@ export default async function OperationPage({
id: depId,
label: operationLabel(depId),
}))}
+ operationSettings={operationSettings}
+ laneSettings={laneSettings}
/>
</div>
);
diff --git a/editor/app/operations/components/OperationDetail.tsx b/editor/app/operations/components/OperationDetail.tsx
@@ -1,5 +1,6 @@
"use client";
+import type { ReactNode } from "react";
import type { AutoQueueKind } from "yt-dlp-transcript-common/jobs/autoQueueState";
import type { AutoQueueStatusPayload } from "../status";
import type { SweepLaneId } from "../lanes";
@@ -33,6 +34,8 @@ export function OperationDetail({
runnerKind,
laneId,
dependsOn,
+ operationSettings,
+ laneSettings,
}: {
id: string;
initial: AutoQueueStatusPayload;
@@ -54,6 +57,17 @@ export function OperationDetail({
// This operation's declared inputs, labelled. Only rendered when there is no
// console — where it is most of what the page has to say.
dependsOn: { id: string; label: string }[];
+ // THE FORM FOR THIS OPERATION'S SETTINGS BLOCK, and the form for its LANE's
+ // block — both built on the server (they need getSettings and the digest app
+ // list) and passed in as slots, so this client component never learns which
+ // operation has which block. Which one an operation gets is decided from the
+ // descriptor's `settingsBlock`, in page.tsx.
+ //
+ // Two props rather than one because they are two scopes: `operationSettings`
+ // belongs to THIS operation, while `laneSettings` belongs to the shared queue
+ // and is therefore the same form on every member page of that lane.
+ operationSettings?: ReactNode;
+ laneSettings?: ReactNode;
}) {
const { data, refresh } = useOperationsStatus(initial);
@@ -85,6 +99,8 @@ export function OperationDetail({
data={data}
activeJobs={activeJobs}
onRefresh={refresh}
+ operationSettings={operationSettings}
+ laneSettings={laneSettings}
/>
) : (
<NoConsoleView id={id} data={data} dependsOn={dependsOn} />
@@ -107,12 +123,16 @@ function SweepOperationView({
data,
activeJobs,
onRefresh,
+ operationSettings,
+ laneSettings,
}: {
id: string;
laneId: SweepLaneId;
data: AutoQueueStatusPayload;
activeJobs: RunningJobsListItem[];
onRefresh: () => Promise<void>;
+ operationSettings?: ReactNode;
+ laneSettings?: ReactNode;
}) {
const hydrated = useHydrated();
const lane = data.lanes[laneId];
@@ -165,6 +185,16 @@ function SweepOperationView({
<RunningJobsList jobs={activeJobs} />
<SweepLane lane={lane} band={band} onRefresh={onRefresh} />
+
+ {/* INSIDE the section, deliberately: `data-lane` is what the e2e suite
+ scopes this panel's controls by, and a settings form outside it would
+ be outside every existing scoped selector. Operation block first, lane
+ block second — this operation's own switches, then the shared queue's.
+ Both are plain <form>s with their own actions, so saving one cannot
+ read the other's checkboxes as off; that is what retired the hidden
+ `*FormPresent` markers the single settings form needed. */}
+ {operationSettings}
+ {laneSettings}
</section>
);
}
diff --git a/editor/app/operations/components/SaveBar.tsx b/editor/app/operations/components/SaveBar.tsx
@@ -63,9 +63,16 @@ export function SaveBar({
Discard changes
</Button>
)}
- {/* The ONLY role="status" on this page, twice over (one per lane). A spec
- asserts getByRole("status").first() reads exactly "Saved.", so any
- other live region here would break it. */}
+ {/* The ONLY role="status" on a RUNNER page, twice over (one per lane). A
+ spec asserts getByRole("status").first() reads exactly "Saved." on
+ /operations/transcription, so any other live region there would break
+ it.
+
+ The boundary is RUNNER pages, and slice 3 is why that now has to be
+ said: a sweep-fed operation's page carries settings forms with a
+ role="status" of their own. Those pages have no policy tree and this
+ bar never renders on them, so the two never meet — and the specs for
+ those forms scope by form[data-settings-block] rather than by role. */}
{result?.ok === true && (
<span role="status" className="text-sm text-success">
Saved.
diff --git a/editor/app/operations/components/settings/AttributionSettingsForm.tsx b/editor/app/operations/components/settings/AttributionSettingsForm.tsx
@@ -0,0 +1,151 @@
+"use client";
+
+import { useActionState } from "react";
+import type { SiteSettings } from "yt-dlp-transcript-common/lib/settings";
+import type { DigestAppDescriptor } from "yt-dlp-transcript-common/lib/digestApps";
+import { Field } from "../../../components/forms/Field";
+import { saveAttributionSettingsAction } from "../../settingsActions";
+import type { SaveResult } from "../../../settings/actions";
+
+// The `attribution` block of settings.json — ONE block behind TWO operations,
+// so this same form is drawn on /operations/attribution-text and on
+// /operations/attribution-diarized. Both pages save the same block, which is
+// why the operations declare the same `settingsBlock` rather than one each.
+export function AttributionSettingsForm({
+ initial,
+ digestApps,
+}: {
+ initial: SiteSettings["attribution"];
+ digestApps: DigestAppDescriptor[];
+}) {
+ const [state, formAction, pending] = useActionState<
+ SaveResult | undefined,
+ FormData
+ >(saveAttributionSettingsAction, undefined);
+
+ return (
+ <form
+ action={formAction}
+ data-settings-block="attribution"
+ className="flex flex-col gap-3"
+ >
+ <fieldset className="flex flex-col gap-3 border border-border rounded p-3">
+ <legend className="px-1 text-sm font-medium">Speaker attribution</legend>
+ <p className="text-xs text-muted-foreground">
+ Putting names to the speakers. Nothing here runs on its own — it
+ registers two operations, and the Speaker work lane below decides when
+ they get the machine. Both are on the same shared pause as speaker
+ diarization.
+ </p>
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="attributionEnabled"
+ defaultChecked={initial.enabled}
+ className="mt-1"
+ />
+ <span className="flex flex-col gap-1">
+ <span className="font-medium">Enable speaker attribution</span>
+ <span className="text-xs text-muted-foreground">
+ Off means neither lane below reports work, whatever their own
+ switches say.
+ </span>
+ </span>
+ </label>
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="attributionDiarized"
+ defaultChecked={initial.diarizedEnabled}
+ className="mt-1"
+ />
+ <span className="flex flex-col gap-1">
+ <span className="font-medium">
+ Name the speakers found in the audio
+ </span>
+ <span className="text-xs text-muted-foreground">
+ About <strong>one model call per video</strong>: the diarizer
+ already grouped the voices, so the model only has to put a name to
+ each group. Needs <code>diarization.json</code>, which is why
+ videos without it are counted separately.
+ </span>
+ </span>
+ </label>
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="attributionTextOnly"
+ defaultChecked={initial.textOnlyEnabled}
+ className="mt-1"
+ />
+ <span className="flex flex-col gap-1">
+ <span className="font-medium">
+ Guess the speakers from the transcript
+ </span>
+ <span className="text-xs text-muted-foreground">
+ For videos with no diarization. Roughly{" "}
+ <strong>one call per transcript chunk</strong> — about 194,000
+ across this archive, or 25–55 days of GPU time, which is the same
+ order as the digest sweep and would compete with it. Quality is
+ genuinely uncertain: it misfires on rapid back-and-forth, and
+ auto-caption channels have no speaker turns in the first place. It
+ never overwrites a record made from the audio.
+ </span>
+ </span>
+ </label>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Engine</span>
+ <select
+ name="attributionAppId"
+ defaultValue={initial.appId}
+ className="rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ {digestApps.map((a) => (
+ <option key={a.id} value={a.id}>
+ {a.label}
+ {a.metered ? " (metered)" : ""}
+ </option>
+ ))}
+ </select>
+ <span className="text-xs text-muted-foreground">
+ Uses the digest engines and their configuration, set on the Digest
+ operation's page — same constrained-JSON workload, so there is
+ no second copy of the URL and context size to keep in step.
+ </span>
+ </label>
+ <Field
+ label="Model override"
+ name="attributionModel"
+ defaultValue={initial.model}
+ hint="Empty means the engine's own configured model. Set separately from the digest's so a digest bake-off does not invalidate every attribution record as a side effect — this value is part of what makes a record fresh."
+ />
+ <Field
+ label="Prompt generation"
+ name="attributionPromptVersion"
+ defaultValue={String(initial.promptVersion)}
+ type="number"
+ hint="A record generated at an older number is stale and will be redone. Raise it to force a corpus-wide regeneration after changing the prompt; it cannot be set below the shipped value, because pinning it lower would freeze superseded output into the archive looking current."
+ />
+ <div className="flex items-center gap-3">
+ <button
+ type="submit"
+ disabled={pending}
+ className="px-3 py-1.5 rounded-md bg-primary text-primary-foreground text-sm font-medium hover:opacity-90 disabled:opacity-50"
+ >
+ {pending ? "Saving…" : "Save attribution settings"}
+ </button>
+ {state?.ok === true && (
+ <span role="status" className="text-sm text-success">
+ Saved.
+ </span>
+ )}
+ {state?.ok === false && (
+ <span role="alert" className="text-sm text-destructive">
+ {state.error}
+ </span>
+ )}
+ </div>
+ </fieldset>
+ </form>
+ );
+}
diff --git a/editor/app/operations/components/settings/DiarizationSettingsForm.tsx b/editor/app/operations/components/settings/DiarizationSettingsForm.tsx
@@ -0,0 +1,216 @@
+"use client";
+
+import { useActionState } from "react";
+import type { SiteSettings } from "yt-dlp-transcript-common/lib/settings";
+import { Field } from "../../../components/forms/Field";
+import { saveDiarizationSettingsAction } from "../../settingsActions";
+import type { SaveResult } from "../../../settings/actions";
+
+// The `diarization` block of settings.json, on /operations/diarization.
+// Moved off /settings by slice 3. Legend and field names verbatim.
+export function DiarizationSettingsForm({
+ initial,
+}: {
+ initial: SiteSettings["diarization"];
+}) {
+ const [state, formAction, pending] = useActionState<
+ SaveResult | undefined,
+ FormData
+ >(saveDiarizationSettingsAction, undefined);
+
+ return (
+ <form
+ action={formAction}
+ data-settings-block="diarization"
+ className="flex flex-col gap-3"
+ >
+ <fieldset className="flex flex-col gap-3 border border-border rounded p-3">
+ <legend className="px-1 text-sm font-medium">Diarization</legend>
+ <p className="text-xs text-muted-foreground">
+ Who is speaking when, captured to <code>diarization.json</code> beside
+ each transcript. This runs on <strong>audio</strong>, and audio is the
+ one input here that goes away — the Clean-audio sweep deletes it once a
+ video is transcribed. So this is a one-shot window per video: the
+ speaker turns can be captured now or not at all, while everything built
+ on top of them can be redone from the saved file at any time.
+ </p>
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="diarizationEnabled"
+ defaultChecked={initial.enabled}
+ className="mt-1"
+ />
+ <span className="flex flex-col gap-1">
+ <span className="font-medium">Capture speaker diarization</span>
+ <span className="text-xs text-muted-foreground">
+ Turning this on also <strong>holds disk</strong>: the Clean-audio
+ sweep stops deleting audio for a transcribed video until its
+ sidecar exists, and each channel's "Est. reclaim"
+ drops to match. That hold is the point — it is what keeps the
+ perishable input alive long enough to capture — but it means free
+ space stops being reclaimed until diarization catches up.
+ </span>
+ </span>
+ </label>
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="diarizationInlineAfterTranscribe"
+ defaultChecked={initial.inlineAfterTranscribe}
+ className="mt-1"
+ />
+ <span className="flex flex-col gap-1">
+ <span className="font-medium">
+ Diarize inline, right after each transcription
+ </span>
+ <span className="text-xs text-muted-foreground">
+ Off by default, from measurement rather than caution. On this box
+ transcription runs at ~221 s per audio-hour on the GPU (16× real
+ time, over 3,602 videos); diarization runs on the CPU at roughly
+ 500–680, because no diarization model has been ported to ggml and
+ neither ONNX nor PyTorch has a Vulkan path on Linux. Inline
+ therefore makes the whole pipeline ~3–4× slower and idles the GPU.
+ For a large batch, leave this off, let transcription run at full
+ speed with the box above holding the audio, and catch up
+ afterwards with <strong>Diarize speakers</strong> on each channel.
+ Turn it on for steady state.
+ </span>
+ </span>
+ </label>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Engine</span>
+ <select
+ name="diarizationEngine"
+ defaultValue={initial.engine}
+ className="rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ <option value="sherpa-onnx">
+ sherpa-onnx — clustered, CPU only (default)
+ </option>
+ <option value="sortformer">
+ Sortformer — end-to-end, GPU or CPU
+ </option>
+ </select>
+ <span className="text-xs text-muted-foreground">
+ <strong>sherpa-onnx</strong> groups voices that sound alike, and it
+ splits far too eagerly: on a 13-minute video with one host it finds
+ 13 speakers, and on the worst file in this corpus it finds 35.{" "}
+ <strong>Sortformer</strong> decides turns directly instead of
+ grouping them afterwards and returns 4 in both cases, agreeing with
+ the other engine about how much of the video the main speaker talks
+ for. It has no threshold, and it caps at 4 speakers — a panel of
+ five will merge two rather than invent twenty.
+ <br />
+ <br />
+ sherpa-onnx is the <strong>faster</strong> of the two (492 against
+ 894 seconds per audio-hour here), so this is a quality choice, not a
+ speed one. Switching also{" "}
+ <strong>marks every recording captured by the other engine as work
+ to redo</strong>, because the two disagree about how many speakers
+ exist — on the audio still held here that is weeks of it.
+ </span>
+ </label>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Sortformer device</span>
+ <select
+ name="diarizationBackend"
+ defaultValue={initial.backend}
+ className="rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ <option value="vulkan">Vulkan — the graphics card</option>
+ <option value="cpu">CPU</option>
+ </select>
+ <span className="text-xs text-muted-foreground">
+ Ignored unless the engine above is Sortformer. Both produce{" "}
+ <strong>identical</strong> speaker turns, so this only trades one
+ resource for another: the card is ~1.5× faster and uses one core
+ instead of two, but takes about 4.4 GB of the same 8 GB card
+ transcription uses — so capture stands aside while transcription is
+ working, and resumes when the card is free. On CPU it needs 4.84 GB
+ of system memory instead.
+ </span>
+ </label>
+ <Field
+ label="Sortformer engine binary"
+ name="diarizationSortformerBin"
+ defaultValue={initial.sortformerBin}
+ hint="Absolute path to the diarize-file binary built by scripts/build-sortformer.sh. Empty means the Sortformer engine is not configured, and every run reports a skip rather than a failure."
+ />
+ <Field
+ label="Sortformer model (GGUF)"
+ name="diarizationSortformerModel"
+ defaultValue={initial.sortformerModel}
+ hint="Absolute path to the .gguf downloaded by scripts/build-sortformer.sh. Its filename is recorded in every sidecar, so changing the model is what marks earlier captures worth redoing."
+ />
+ <Field
+ label="Segmentation model (ONNX)"
+ name="diarizationSegModel"
+ defaultValue={initial.segModel}
+ hint="Absolute path to a pyannote segmentation-3.0 ONNX model. Empty means the lane is not configured and every run reports a skip rather than a failure."
+ />
+ <Field
+ label="Speaker-embedding model (ONNX)"
+ name="diarizationEmbModel"
+ defaultValue={initial.embModel}
+ hint="Absolute path to a speaker-embedding ONNX model (e.g. NeMo TitaNet)."
+ />
+ <Field
+ label="Python interpreter"
+ name="diarizationPython"
+ defaultValue={initial.python}
+ hint="Interpreter with sherpa-onnx installed. sherpa-onnx ships wheels only up to CPython 3.13, so on a 3.14 system this must point at a dedicated venv."
+ />
+ <Field
+ label="Clustering threshold"
+ name="diarizationThreshold"
+ defaultValue={String(initial.threshold)}
+ type="number"
+ step="0.01"
+ hint="The knob that decides how many speakers come out; larger merges more. It is recorded in every sidecar, so changing it is what lets a later pass tell which files are worth regenerating."
+ />
+ <Field
+ label="Engine threads"
+ name="diarizationThreads"
+ defaultValue={String(initial.threads)}
+ type="number"
+ hint="Threads per diarize run."
+ />
+ <Field
+ label="Max audio hours"
+ name="diarizationMaxAudioHours"
+ defaultValue={String(initial.maxAudioHours)}
+ type="number"
+ step="0.5"
+ hint="Videos longer than this are reported as deferred instead of being diarized, and are never counted as work still to do. This is a stopgap for an out-of-memory crash on very long recordings: the engine's memory use grows with the SQUARE of the number of speaker turns, so a dense six-hour stream can exhaust 16 GB after 40 minutes of work and produce nothing. 0 turns the limit off."
+ />
+ <Field
+ label="Diarization concurrency"
+ name="diarizationConcurrency"
+ defaultValue={String(initial.concurrency)}
+ type="number"
+ hint="Videos diarized at once during a backfill. Default 1: this is CPU-bound work competing with GPU feeding and the digest sweep for the same 8 threads."
+ />
+ <div className="flex items-center gap-3">
+ <button
+ type="submit"
+ disabled={pending}
+ className="px-3 py-1.5 rounded-md bg-primary text-primary-foreground text-sm font-medium hover:opacity-90 disabled:opacity-50"
+ >
+ {pending ? "Saving…" : "Save diarization settings"}
+ </button>
+ {state?.ok === true && (
+ <span role="status" className="text-sm text-success">
+ Saved.
+ </span>
+ )}
+ {state?.ok === false && (
+ <span role="alert" className="text-sm text-destructive">
+ {state.error}
+ </span>
+ )}
+ </div>
+ </fieldset>
+ </form>
+ );
+}
diff --git a/editor/app/settings/components/DigestAppsField.tsx b/editor/app/operations/components/settings/DigestAppsField.tsx
diff --git a/editor/app/operations/components/settings/DigestSettingsForm.tsx b/editor/app/operations/components/settings/DigestSettingsForm.tsx
@@ -0,0 +1,240 @@
+"use client";
+
+import { useActionState } from "react";
+import type { SiteSettings } from "yt-dlp-transcript-common/lib/settings";
+// From the CLIENT-SAFE digest module, NOT settings.ts's re-exports of them:
+// settings.ts opens with `import fs from "node:fs"`, so pulling the option
+// lists through it drags node:fs into the client bundle and the page dies at
+// runtime. Same rule the digest modules are already split along.
+import {
+ DEFAULT_DIGEST_TIMESTAMP_MODE,
+ DIGEST_SECTION_KINDS as DIGEST_SECTION_OPTIONS,
+ DIGEST_TIMESTAMP_MODES as DIGEST_TIMESTAMP_MODE_OPTIONS,
+} from "yt-dlp-transcript-common/lib/digest";
+import type { DigestAppDescriptor } from "yt-dlp-transcript-common/lib/digestApps";
+import { Field } from "../../../components/forms/Field";
+import { DigestAppsField } from "./DigestAppsField";
+import { saveDigestSettingsAction } from "../../settingsActions";
+import type { SaveResult } from "../../../settings/actions";
+
+// The `digest` block of settings.json, on /operations/digest. Moved off
+// /settings by slice 3: an operation's settings belong on its operation page,
+// beside the state, the sweep and the pause an operator reads there.
+//
+// The legend and every field NAME are verbatim from the old fieldset — the
+// action reads those names and two e2e specs select by them.
+export function DigestSettingsForm({
+ initial,
+ digestApps,
+}: {
+ initial: SiteSettings["digest"];
+ digestApps: DigestAppDescriptor[];
+}) {
+ const [state, formAction, pending] = useActionState<
+ SaveResult | undefined,
+ FormData
+ >(saveDigestSettingsAction, undefined);
+
+ return (
+ <form
+ action={formAction}
+ data-settings-block="digest"
+ className="flex flex-col gap-3"
+ >
+ <fieldset className="flex flex-col gap-3 border border-border rounded p-3">
+ <legend className="px-1 text-sm font-medium">Digest</legend>
+ <p className="text-xs text-muted-foreground">
+ AI chapters and topic tags generated from existing transcripts by a
+ local model, written to <code>ai-digest.json</code> beside each one.
+ Run a sweep from a channel's <strong>Digest</strong> stage. Every
+ section records the exact engine, model, prompt version and prompt
+ shape that produced it, and a re-run regenerates only what no longer
+ matches — so changing anything here is what makes the next run redo
+ work, and leaving it alone is what makes a re-run nearly free.
+ </p>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Local engine</span>
+ <select
+ name="digestLocalAppId"
+ defaultValue={initial.localAppId}
+ className="rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ {digestApps
+ .filter((a) => !a.metered)
+ .map((a) => (
+ <option key={a.id} value={a.id}>
+ {a.label}
+ </option>
+ ))}
+ </select>
+ <span className="text-xs text-muted-foreground">
+ The lane that carries the corpus. Runs on your own hardware; nothing
+ leaves the machine.
+ </span>
+ </label>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Sections to generate</span>
+ <span className="flex flex-wrap gap-3">
+ {DIGEST_SECTION_OPTIONS.map((section) => (
+ <label key={section} className="flex items-center gap-1 text-sm">
+ <input
+ type="checkbox"
+ name="digestSections"
+ value={section}
+ defaultChecked={initial.sections.includes(section)}
+ />
+ {section}
+ </label>
+ ))}
+ </span>
+ <span className="text-xs text-muted-foreground">
+ Tags double the model calls but cost only 5–15% more time: a tag call
+ re-sends the same transcript as the chapter call before it, so it
+ reuses the engine's cached prompt and pays for decoding ~30
+ output tokens instead of ~250. Generating them <em>later</em>, in
+ their own pass, pays for the transcript again — measured at 44% of a
+ full chapters pass. So if you want tags at all, check them now
+ rather than after. Unchecking everything falls back to chapters
+ rather than generating nothing.
+ </span>
+ </label>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Timestamp mode</span>
+ <select
+ name="digestTimestampMode"
+ defaultValue={initial.timestampMode}
+ className="rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ {DIGEST_TIMESTAMP_MODE_OPTIONS.map((mode) => (
+ <option key={mode} value={mode}>
+ {mode}
+ {mode === DEFAULT_DIGEST_TIMESTAMP_MODE ? " (recommended)" : ""}
+ </option>
+ ))}
+ </select>
+ <span className="text-xs text-muted-foreground">
+ How each chunk's transcript markers are numbered.{" "}
+ <strong>chunk-local</strong> re-bases every chunk to 00:00:00 and
+ adds the offset back before any guard runs; on the long tail it cut
+ wasted chunks from 29.4% to 11.8% and the worst coverage gap from
+ 1:05:16 to 24:13, because the model stops having to hold a large
+ offset. <strong>absolute</strong> is kept for comparison. This
+ changes the recorded identity, so switching it re-runs the corpus
+ instead of silently mixing two shapes.
+ </span>
+ </label>
+ <Field
+ label="Prompt variant label"
+ name="digestPromptVariant"
+ defaultValue={initial.promptVariant}
+ hint="Free-text label for a non-default prompt shape, recorded in every section's provenance (trimmed, max 40 chars). Setting or changing it invalidates digests generated under a different label — which is exactly what makes a bake-off round re-run its sample instead of skipping it as fresh. Leave blank unless you are running one."
+ />
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="digestYieldToCpuWorkers"
+ defaultChecked={initial.yieldToCpuWorkers}
+ className="mt-1"
+ />
+ <span className="flex flex-col gap-1">
+ <span className="font-medium">
+ Also yield the GPU to CPU-only transcription workers
+ </span>
+ <span className="text-xs text-muted-foreground">
+ Off by default. The digest lane steps aside while transcription
+ uses the card, but a worker pinned to{" "}
+ <code>device: cpu</code> competes for no GPU shaders at all —
+ treating it as contention held the digest lane at zero throughput
+ for nothing. A worker with <em>no</em> device set still counts,
+ because that is the engine binary's own default and it may be
+ the GPU. Turn this on to make every local worker block the digest
+ lane regardless of device.
+ </span>
+ </span>
+ </label>
+ <details className="text-sm">
+ <summary className="cursor-pointer font-medium">
+ Per-engine configuration
+ </summary>
+ <div className="mt-2">
+ <DigestAppsField
+ apps={digestApps}
+ initial={initial.apps}
+ name="digestAppsJson"
+ />
+ </div>
+ </details>
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="digestRemoteEnabled"
+ defaultChecked={initial.remoteEnabled}
+ className="mt-1"
+ />
+ <span className="flex flex-col gap-1">
+ <span className="font-medium">
+ Enable the metered (paid) digest lane
+ </span>
+ <span className="text-xs text-muted-foreground">
+ Off by default, and deliberately: this lane sends transcripts to a
+ paid API. It exists for the long tail of very long videos, which
+ is a small fraction of the corpus by count and a large one by
+ tokens. With it off, the metered option is disabled in every
+ channel's Digest stage rather than failing after the fact.
+ </span>
+ </span>
+ </label>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Metered engine</span>
+ <select
+ name="digestRemoteAppId"
+ defaultValue={initial.remoteAppId}
+ className="rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ {digestApps
+ .filter((a) => a.metered)
+ .map((a) => (
+ <option key={a.id} value={a.id}>
+ {a.label}
+ </option>
+ ))}
+ </select>
+ </label>
+ <Field
+ label="Long-tail cutoff (seconds)"
+ name="digestLongTailSeconds"
+ defaultValue={String(initial.longTailSeconds)}
+ type="number"
+ hint="Videos at least this long are what the metered lane takes when you run it. Default 14400 (4 h)."
+ />
+ <Field
+ label="Spend cap (USD per job)"
+ name="digestSpendCapUsd"
+ defaultValue={String(initial.spendCapUsd)}
+ type="number"
+ step="0.01"
+ hint="Hard ceiling on cumulative metered spend within one job; the lane parks itself when it is reached. 0 means no cap. Only ever consulted for a metered engine."
+ />
+ <div className="flex items-center gap-3">
+ <button
+ type="submit"
+ disabled={pending}
+ className="px-3 py-1.5 rounded-md bg-primary text-primary-foreground text-sm font-medium hover:opacity-90 disabled:opacity-50"
+ >
+ {pending ? "Saving…" : "Save digest settings"}
+ </button>
+ {state?.ok === true && (
+ <span role="status" className="text-sm text-success">
+ Saved.
+ </span>
+ )}
+ {state?.ok === false && (
+ <span role="alert" className="text-sm text-destructive">
+ {state.error}
+ </span>
+ )}
+ </div>
+ </fieldset>
+ </form>
+ );
+}
diff --git a/editor/app/operations/components/settings/LaneSettingsForm.tsx b/editor/app/operations/components/settings/LaneSettingsForm.tsx
@@ -0,0 +1,133 @@
+"use client";
+
+import { useActionState } from "react";
+import type { SiteSettings } from "yt-dlp-transcript-common/lib/settings";
+import { Field } from "../../../components/forms/Field";
+import { saveBackfillLaneSettingsAction } from "../../settingsActions";
+import type { SaveResult } from "../../../settings/actions";
+
+// The `backfill` block of settings.json — the speaker work LANE, not any one
+// operation's settings. It governs BACKFILL_QUEUE, which diarization,
+// attribution-text and attribution-diarized share, so it is rendered once per
+// LANE and therefore appears on all three of their pages, exactly as the
+// shared pause and the shared sweep above it already do.
+//
+// That is why it takes no operation: there is nothing here that belongs to one.
+export function LaneSettingsForm({
+ initial,
+}: {
+ initial: SiteSettings["backfill"];
+}) {
+ const [state, formAction, pending] = useActionState<
+ SaveResult | undefined,
+ FormData
+ >(saveBackfillLaneSettingsAction, undefined);
+
+ return (
+ <form
+ action={formAction}
+ data-settings-block="backfill"
+ className="flex flex-col gap-3"
+ >
+ <fieldset className="flex flex-col gap-3 border border-border rounded p-3">
+ {/* THE ONE FIELDSET THAT STILL SAYS "LANE", and it is the right one:
+ this is where the shared pause and the shared sweep live, and that
+ these operations share a queue is a load-bearing fact rather than a
+ leaked implementation detail. Everywhere an operator reads a FIGURE
+ they now read an operation name instead. */}
+ <legend className="px-1 text-sm font-medium">
+ Speaker work lane
+ </legend>
+ <p className="text-xs text-muted-foreground">
+ Catch-up for derived data the existing corpus predates — speaker
+ diarization and the two speaker-name operations. These operations
+ share one queue and one pause: the settings here decide how much of
+ the machine all of them together may use, not any one of them
+ individually. Each feature declares what it needs and how to tell
+ whether a video already has it.
+ </p>
+ <p className="text-xs text-muted-foreground">
+ <strong className="font-medium text-foreground">
+ Which operations a corpus sweep runs, and over which channels, is
+ not set here.
+ </strong>{" "}
+ That scope has to be written at the moment the sweep is armed — a
+ scope saved separately would be resurrected as a corpus-wide run by
+ the next restart — so it lives with the plan it produces, in the
+ sweep controls directly above, where you can read the itinerary before
+ committing to it.
+ </p>
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="backfillEnabled"
+ defaultChecked={initial.enabled}
+ className="mt-1"
+ />
+ <span className="flex flex-col gap-1">
+ <span className="font-medium">Run the backfill lane</span>
+ <span className="text-xs text-muted-foreground">
+ Off means the lane still <em>reports</em> what is missing —
+ that is what the indicators are for — but runs nothing.
+ </span>
+ </span>
+ </label>
+ <Field
+ label="Resource share"
+ name="backfillWeight"
+ defaultValue={String(initial.weight)}
+ type="number"
+ step="0.05"
+ hint="0 (the default) means idle-only: the lane runs only while transcription is quiet, and stands aside the moment it isn't. Above 0 it takes that fraction of its slots as a guaranteed share, floored at 1 — so a small number is a slow lane, not a stopped one."
+ />
+ <Field
+ label="Lane concurrency"
+ name="backfillConcurrency"
+ defaultValue={String(initial.concurrency)}
+ type="number"
+ hint="Slots the lane may use when it is not standing aside. Default 1 — this is CPU-bound work competing with GPU feeding and the digest sweep for the same threads."
+ />
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="backfillAllowRedownload"
+ defaultChecked={initial.allowRedownload}
+ className="mt-1"
+ />
+ <span className="flex flex-col gap-1">
+ <span className="font-medium">
+ Re-acquire media that has already been deleted
+ </span>
+ <span className="text-xs text-muted-foreground">
+ Off by default, and that default is measured: 836 videos still
+ have media on disk and ~76,270 would need re-downloading — 91×
+ the reachable work, against 45 GB free. With this on, each file is
+ fetched, used, and <strong>deleted again immediately</strong>{" "}
+ (unless the video is marked "do not clean"), and nothing
+ starts at all when free space is under the disk floor.
+ </span>
+ </span>
+ </label>
+ <div className="flex items-center gap-3">
+ <button
+ type="submit"
+ disabled={pending}
+ className="px-3 py-1.5 rounded-md bg-primary text-primary-foreground text-sm font-medium hover:opacity-90 disabled:opacity-50"
+ >
+ {pending ? "Saving…" : "Save lane settings"}
+ </button>
+ {state?.ok === true && (
+ <span role="status" className="text-sm text-success">
+ Saved.
+ </span>
+ )}
+ {state?.ok === false && (
+ <span role="alert" className="text-sm text-destructive">
+ {state.error}
+ </span>
+ )}
+ </div>
+ </fieldset>
+ </form>
+ );
+}
diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts
@@ -24,10 +24,6 @@ import {
} from "yt-dlp-transcript-common/lib/settings";
import { DEFAULT_TRANSCRIPTION_APP_ID } from "yt-dlp-transcript-common/lib/transcriptionApps";
import {
- isDigestSectionKind,
- isDigestTimestampMode,
-} from "yt-dlp-transcript-common/lib/digest";
-import {
DEFAULT_COOKIE_MODE,
isCookieMode,
} from "yt-dlp-transcript-common/lib/cookiePolicy";
@@ -261,177 +257,6 @@ export async function saveSettingsAction(
// Build pipeline. Values are clamped/coerced by sanitizeBuildPipeline inside
// writeSettings, so we only read the form here (NaN/blank → default). The
// deploy-page toggle also writes `mode`; whichever saves last wins.
- // Digest. The metered lane's switch is a checkbox like any other, but note the
- // asymmetry: everything else here defaults to the CURRENT value on a partial
- // save, while `remoteEnabled` is read straight from the form so it can never be
- // turned on by an unrelated save. writeSettings re-sanitizes the whole block.
- const dD = getSettings().digest;
- const digestAppsRaw = String(formData.get("digestAppsJson") ?? "").trim();
- let digestApps: unknown = dD.apps;
- if (digestAppsRaw) {
- try {
- digestApps = JSON.parse(digestAppsRaw);
- } catch {
- return { ok: false, error: "Digest app config payload is malformed" };
- }
- }
- // Filtered to KNOWN kinds here rather than leaning on writeSettings' sanitizer.
- // sanitizeDigest would drop an unknown value anyway, but typing it honestly is
- // what lets the digest block below be checked against DigestSettings instead of
- // cast — and the cast is what hid the dropped-fields bug.
- const digestSectionsRaw = formData
- .getAll("digestSections")
- .map(String)
- .filter(isDigestSectionKind);
- // A hidden marker, because unchecked checkboxes are simply ABSENT from a
- // FormData: without it, a submit from any form that lacks the digest fields
- // would read remoteEnabled as false and silently reset the block.
- const digestFormPresent = formData.get("digestFormPresent") === "1";
- const digestTimestampModeRaw = String(
- formData.get("digestTimestampMode") ?? "",
- ).trim();
- const digestSettings: SiteSettings["digest"] = digestFormPresent
- ? {
- // Spread the CURRENT block first. Every field this form does not render
- // must survive a save untouched, and before this spread they did not:
- // the block was built field-by-field and cast, so `yieldToTranscription`
- // and — much worse — `sweepEnabled`/`sweepChannels` were absent from the
- // object, and sanitizeDigest re-derived them from nothing. Saving any
- // unrelated setting DISARMED AN ARMED CORPUS SWEEP. The cast is gone
- // too, so the next added field is a type error rather than a silent
- // reset.
- ...dD,
- remoteEnabled: formData.get("digestRemoteEnabled") === "on",
- longTailSeconds: Number.parseInt(
- String(formData.get("digestLongTailSeconds") ?? "").trim(),
- 10,
- ),
- localAppId:
- String(formData.get("digestLocalAppId") ?? "").trim() ||
- dD.localAppId,
- remoteAppId:
- String(formData.get("digestRemoteAppId") ?? "").trim() ||
- dD.remoteAppId,
- // Cast, not sanitized here: this is hand-parsed JSON from the form and
- // writeSettings runs sanitizeDigestApps over it. Narrowed at exactly
- // this field so every OTHER field in the block stays type-checked.
- apps: digestApps as SiteSettings["digest"]["apps"],
- // Not edited by this form — the dashboard/channel controls own the pause.
- digestsPaused: dD.digestsPaused,
- spendCapUsd: Number.parseFloat(
- String(formData.get("digestSpendCapUsd") ?? "").trim(),
- ),
- sections:
- digestSectionsRaw.length > 0 ? digestSectionsRaw : dD.sections,
- // Prompt SHAPE — both freshness-affecting (see digestPromptVariant),
- // so a silent reset here would invalidate every digest generated
- // under a non-default shape. Each still FALLS BACK to the current
- // value rather than to the default when the field is absent: a form
- // that rebuilds this block but omits a field is exactly how the reset
- // bug happens, and the fallback is what makes omission harmless.
- timestampMode: isDigestTimestampMode(digestTimestampModeRaw)
- ? digestTimestampModeRaw
- : dD.timestampMode,
- promptVariant: formData.has("digestPromptVariant")
- ? String(formData.get("digestPromptVariant") ?? "").trim()
- : dD.promptVariant,
- // Rendered by the form, so read it from the form — but only when the
- // digest fields are actually present, which the branch already assures.
- yieldToCpuWorkers: formData.get("digestYieldToCpuWorkers") === "on",
- }
- : dD;
-
- // Diarization. Same hidden-marker discipline as the digest block above, and
- // for the same reason: unchecked checkboxes are absent from a FormData, so
- // without the marker any unrelated save would read `enabled` as false and
- // silently disarm the capture lane — which would let the next Clean-audio
- // sweep delete audio that was being held for diarization. Spread the CURRENT
- // block first so fields this form does not render survive untouched.
- const dDiar = getSettings().diarization;
- const diarizationFormPresent =
- formData.get("diarizationFormPresent") === "1";
- const num = (key: string, fallback: number) => {
- const n = Number.parseFloat(String(formData.get(key) ?? "").trim());
- return Number.isFinite(n) ? n : fallback;
- };
- const diarizationSettings: SiteSettings["diarization"] =
- diarizationFormPresent
- ? {
- ...dDiar,
- enabled: formData.get("diarizationEnabled") === "on",
- inlineAfterTranscribe:
- formData.get("diarizationInlineAfterTranscribe") === "on",
- threshold: num("diarizationThreshold", dDiar.threshold),
- threads: num("diarizationThreads", dDiar.threads),
- concurrency: num("diarizationConcurrency", dDiar.concurrency),
- maxAudioHours: num("diarizationMaxAudioHours", dDiar.maxAudioHours),
- python:
- String(formData.get("diarizationPython") ?? "").trim() ||
- dDiar.python,
- segModel: String(formData.get("diarizationSegModel") ?? "").trim(),
- embModel: String(formData.get("diarizationEmbModel") ?? "").trim(),
- // Read as plain strings and left to sanitizeDiarization to validate:
- // an unrecognized value there falls back to the DEFAULT engine, which
- // is the one every sidecar on disk already matches. Narrowing here
- // instead would mean this form and the sanitizer could disagree about
- // what a valid engine is, and the corpus pays for that disagreement in
- // weeks of regeneration.
- engine: String(
- formData.get("diarizationEngine") ?? dDiar.engine,
- ) as typeof dDiar.engine,
- backend: String(
- formData.get("diarizationBackend") ?? dDiar.backend,
- ) as typeof dDiar.backend,
- sortformerBin: String(
- formData.get("diarizationSortformerBin") ?? "",
- ).trim(),
- sortformerModel: String(
- formData.get("diarizationSortformerModel") ?? "",
- ).trim(),
- }
- : dDiar;
-
- // Backfill lane. Same hidden-marker discipline as the two blocks above, and
- // here it protects two things specifically: `allowRedownload`, which an
- // unrelated save must never flip ON (it writes media to a 97%-full disk), and
- // `sweepEnabled` + its scope, which this form does NOT render at all — those
- // are owned by the sweep controls, and reading them from an absent form field
- // would disarm a running multi-day sweep on any settings save.
- const dBack = getSettings().backfill;
- const backfillFormPresent = formData.get("backfillFormPresent") === "1";
- const backfillSettings: SiteSettings["backfill"] = backfillFormPresent
- ? {
- ...dBack,
- enabled: formData.get("backfillEnabled") === "on",
- weight: num("backfillWeight", dBack.weight),
- concurrency: num("backfillConcurrency", dBack.concurrency),
- allowRedownload: formData.get("backfillAllowRedownload") === "on",
- }
- : dBack;
-
- // Speaker attribution. Same hidden-marker discipline, and here it guards the
- // most expensive switch in the form: `textOnlyEnabled` arms a lane costing
- // roughly one model call per transcript chunk — ~194,000 across this corpus.
- // An unrelated save must never be able to flip that on.
- const dAttr = getSettings().attribution;
- const attributionFormPresent =
- formData.get("attributionFormPresent") === "1";
- const attributionSettings: SiteSettings["attribution"] =
- attributionFormPresent
- ? {
- ...dAttr,
- enabled: formData.get("attributionEnabled") === "on",
- appId: String(formData.get("attributionAppId") ?? dAttr.appId),
- // Trimmed but NOT defaulted: empty is meaningful here ("use the
- // engine's own model"), so an operator clearing the field must be able
- // to clear it.
- model: String(formData.get("attributionModel") ?? dAttr.model).trim(),
- diarizedEnabled: formData.get("attributionDiarized") === "on",
- textOnlyEnabled: formData.get("attributionTextOnly") === "on",
- promptVersion: num("attributionPromptVersion", dAttr.promptVersion),
- }
- : dAttr;
-
const dB = defaultBuildPipeline();
const buildModeRaw = String(formData.get("buildMode") ?? "").trim();
const buildPipeline = {
@@ -484,12 +309,17 @@ export async function saveSettingsAction(
// Saved Videos page edits it). writeSettings re-sanitizes it regardless.
savedVideoBackup: getSettings().savedVideoBackup,
buildPipeline,
- // Same: the Digest section of this form owns these fields, but an unrelated
- // save must not reset them (and must never silently flip remoteEnabled on).
- digest: digestSettings,
- diarization: diarizationSettings,
- backfill: backfillSettings,
- attribution: attributionSettings,
+ // Each edited on its own operation page; preserved here. Slice 3 moved
+ // these four fieldsets to /operations/<id>, and with them the hidden
+ // `*FormPresent` markers that used to gate them — a marker only existed
+ // because one form saved everything, and reading an absent checkbox as
+ // `false` could disarm a sweep, drop the diarization capture lane, flip
+ // `allowRedownload` on, or arm ~194,000 model calls. Reading the CURRENT
+ // block is now the whole protection, exactly as it is for autoQueue.
+ digest: getSettings().digest,
+ diarization: getSettings().diarization,
+ backfill: getSettings().backfill,
+ attribution: getSettings().attribution,
};
try {
await writeSettings(next);
diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx
@@ -1,27 +1,16 @@
"use client";
-import Link from "next/link";
import { useActionState, useState } from "react";
import {
saveSettingsAction,
type SaveResult,
} from "../actions";
import type { SiteSettings } from "yt-dlp-transcript-common/lib/settings";
-// From the CLIENT-SAFE digest module, NOT settings.ts's re-exports of them:
-// settings.ts opens with `import fs from "node:fs"`, so pulling the option
-// lists through it drags node:fs into the client bundle and the page dies at
-// runtime. Same rule the digest modules are already split along.
-import {
- DEFAULT_DIGEST_TIMESTAMP_MODE,
- DIGEST_SECTION_KINDS as DIGEST_SECTION_OPTIONS,
- DIGEST_TIMESTAMP_MODES as DIGEST_TIMESTAMP_MODE_OPTIONS,
-} from "yt-dlp-transcript-common/lib/digest";
import {
DOWNLOAD_FORMAT_LABELS,
DOWNLOAD_FORMAT_PRESETS,
} from "yt-dlp-transcript-common/ytdlp/downloadFormat";
import type { TranscriptionAppDescriptor } from "yt-dlp-transcript-common/lib/transcriptionApps";
-import type { DigestAppDescriptor } from "yt-dlp-transcript-common/lib/digestApps";
import {
SYNC_INTERVAL_MAX_MINUTES,
SYNC_INTERVAL_MIN_MINUTES,
@@ -29,7 +18,6 @@ import {
import { DurationField } from "../../components/DurationField";
import { Field } from "../../components/forms/Field";
import { FULL_SWEEP_PRESETS } from "../../scheduler/intervalPresets";
-import { DigestAppsField } from "./DigestAppsField";
import {
SocialLinksField,
toSocialRow,
@@ -40,15 +28,12 @@ import { WorkersField } from "./WorkersField";
type Props = {
initial: SiteSettings;
apps: TranscriptionAppDescriptor[];
- // Built on the server: digestApps.ts reaches process.env and imports execa, so
- // it must never end up in the client bundle. Same reason as `apps`.
- digestApps: DigestAppDescriptor[];
// Known worker-tag vocabulary (operation ids + resources), built on the
// server for the same bundle reason — the catalog lives in operations.ts.
workerTags?: string[];
};
-export function SettingsForm({ initial, apps, digestApps, workerTags }: Props) {
+export function SettingsForm({ initial, apps, workerTags }: Props) {
const [state, formAction] = useActionState<SaveResult | undefined, FormData>(
saveSettingsAction,
undefined,
@@ -525,568 +510,6 @@ export function SettingsForm({ initial, apps, digestApps, workerTags }: Props) {
/>
</fieldset>
<fieldset className="flex flex-col gap-3 border border-border rounded p-3">
- <legend className="px-1 text-sm font-medium">Digest</legend>
- {/*
- THE MARKER THAT ARMS THE WHOLE BLOCK. Unchecked checkboxes are simply
- absent from a FormData, so without it a submit from any other form
- would read `digestRemoteEnabled` as false and silently reset this
- section. saveSettingsAction gates every field below on it.
- */}
- <input type="hidden" name="digestFormPresent" value="1" readOnly />
- <p className="text-xs text-muted-foreground">
- AI chapters and topic tags generated from existing transcripts by a
- local model, written to <code>ai-digest.json</code> beside each one.
- Run a sweep from a channel's <strong>Digest</strong> stage. Every
- section records the exact engine, model, prompt version and prompt
- shape that produced it, and a re-run regenerates only what no longer
- matches — so changing anything here is what makes the next run redo
- work, and leaving it alone is what makes a re-run nearly free.
- </p>
- <label className="flex flex-col gap-1 text-sm">
- <span className="font-medium">Local engine</span>
- <select
- name="digestLocalAppId"
- defaultValue={initial.digest.localAppId}
- className="rounded border border-border bg-card px-2 py-1 text-sm"
- >
- {digestApps
- .filter((a) => !a.metered)
- .map((a) => (
- <option key={a.id} value={a.id}>
- {a.label}
- </option>
- ))}
- </select>
- <span className="text-xs text-muted-foreground">
- The lane that carries the corpus. Runs on your own hardware; nothing
- leaves the machine.
- </span>
- </label>
- <label className="flex flex-col gap-1 text-sm">
- <span className="font-medium">Sections to generate</span>
- <span className="flex flex-wrap gap-3">
- {DIGEST_SECTION_OPTIONS.map((section) => (
- <label key={section} className="flex items-center gap-1 text-sm">
- <input
- type="checkbox"
- name="digestSections"
- value={section}
- defaultChecked={initial.digest.sections.includes(section)}
- />
- {section}
- </label>
- ))}
- </span>
- <span className="text-xs text-muted-foreground">
- Tags double the model calls but cost only 5–15% more time: a tag call
- re-sends the same transcript as the chapter call before it, so it
- reuses the engine's cached prompt and pays for decoding ~30
- output tokens instead of ~250. Generating them <em>later</em>, in
- their own pass, pays for the transcript again — measured at 44% of a
- full chapters pass. So if you want tags at all, check them now
- rather than after. Unchecking everything falls back to chapters
- rather than generating nothing.
- </span>
- </label>
- <label className="flex flex-col gap-1 text-sm">
- <span className="font-medium">Timestamp mode</span>
- <select
- name="digestTimestampMode"
- defaultValue={initial.digest.timestampMode}
- className="rounded border border-border bg-card px-2 py-1 text-sm"
- >
- {DIGEST_TIMESTAMP_MODE_OPTIONS.map((mode) => (
- <option key={mode} value={mode}>
- {mode}
- {mode === DEFAULT_DIGEST_TIMESTAMP_MODE ? " (recommended)" : ""}
- </option>
- ))}
- </select>
- <span className="text-xs text-muted-foreground">
- How each chunk's transcript markers are numbered.{" "}
- <strong>chunk-local</strong> re-bases every chunk to 00:00:00 and
- adds the offset back before any guard runs; on the long tail it cut
- wasted chunks from 29.4% to 11.8% and the worst coverage gap from
- 1:05:16 to 24:13, because the model stops having to hold a large
- offset. <strong>absolute</strong> is kept for comparison. This
- changes the recorded identity, so switching it re-runs the corpus
- instead of silently mixing two shapes.
- </span>
- </label>
- <Field
- label="Prompt variant label"
- name="digestPromptVariant"
- defaultValue={initial.digest.promptVariant}
- hint="Free-text label for a non-default prompt shape, recorded in every section's provenance (trimmed, max 40 chars). Setting or changing it invalidates digests generated under a different label — which is exactly what makes a bake-off round re-run its sample instead of skipping it as fresh. Leave blank unless you are running one."
- />
- <label className="flex items-start gap-2 text-sm">
- <input
- type="checkbox"
- name="digestYieldToCpuWorkers"
- defaultChecked={initial.digest.yieldToCpuWorkers}
- className="mt-1"
- />
- <span className="flex flex-col gap-1">
- <span className="font-medium">
- Also yield the GPU to CPU-only transcription workers
- </span>
- <span className="text-xs text-muted-foreground">
- Off by default. The digest lane steps aside while transcription
- uses the card, but a worker pinned to{" "}
- <code>device: cpu</code> competes for no GPU shaders at all —
- treating it as contention held the digest lane at zero throughput
- for nothing. A worker with <em>no</em> device set still counts,
- because that is the engine binary's own default and it may be
- the GPU. Turn this on to make every local worker block the digest
- lane regardless of device.
- </span>
- </span>
- </label>
- <details className="text-sm">
- <summary className="cursor-pointer font-medium">
- Per-engine configuration
- </summary>
- <div className="mt-2">
- <DigestAppsField
- apps={digestApps}
- initial={initial.digest.apps}
- name="digestAppsJson"
- />
- </div>
- </details>
- <label className="flex items-start gap-2 text-sm">
- <input
- type="checkbox"
- name="digestRemoteEnabled"
- defaultChecked={initial.digest.remoteEnabled}
- className="mt-1"
- />
- <span className="flex flex-col gap-1">
- <span className="font-medium">
- Enable the metered (paid) digest lane
- </span>
- <span className="text-xs text-muted-foreground">
- Off by default, and deliberately: this lane sends transcripts to a
- paid API. It exists for the long tail of very long videos, which
- is a small fraction of the corpus by count and a large one by
- tokens. With it off, the metered option is disabled in every
- channel's Digest stage rather than failing after the fact.
- </span>
- </span>
- </label>
- <label className="flex flex-col gap-1 text-sm">
- <span className="font-medium">Metered engine</span>
- <select
- name="digestRemoteAppId"
- defaultValue={initial.digest.remoteAppId}
- className="rounded border border-border bg-card px-2 py-1 text-sm"
- >
- {digestApps
- .filter((a) => a.metered)
- .map((a) => (
- <option key={a.id} value={a.id}>
- {a.label}
- </option>
- ))}
- </select>
- </label>
- <Field
- label="Long-tail cutoff (seconds)"
- name="digestLongTailSeconds"
- defaultValue={String(initial.digest.longTailSeconds)}
- type="number"
- hint="Videos at least this long are what the metered lane takes when you run it. Default 14400 (4 h)."
- />
- <Field
- label="Spend cap (USD per job)"
- name="digestSpendCapUsd"
- defaultValue={String(initial.digest.spendCapUsd)}
- type="number"
- step="0.01"
- hint="Hard ceiling on cumulative metered spend within one job; the lane parks itself when it is reached. 0 means no cap. Only ever consulted for a metered engine."
- />
- </fieldset>
- <fieldset className="flex flex-col gap-3 border border-border rounded p-3">
- <legend className="px-1 text-sm font-medium">Diarization</legend>
- {/*
- Same marker discipline as Digest above — and it matters MORE here.
- Without it an unrelated save would read `diarizationEnabled` as false,
- disarming the cleanup guard, and the next Clean-audio sweep would
- delete audio that was being held for diarization. That deletion is
- not recoverable.
- */}
- <input
- type="hidden"
- name="diarizationFormPresent"
- value="1"
- readOnly
- />
- <p className="text-xs text-muted-foreground">
- Who is speaking when, captured to <code>diarization.json</code> beside
- each transcript. This runs on <strong>audio</strong>, and audio is the
- one input here that goes away — the Clean-audio sweep deletes it once a
- video is transcribed. So this is a one-shot window per video: the
- speaker turns can be captured now or not at all, while everything built
- on top of them can be redone from the saved file at any time.
- </p>
- <label className="flex items-start gap-2 text-sm">
- <input
- type="checkbox"
- name="diarizationEnabled"
- defaultChecked={initial.diarization.enabled}
- className="mt-1"
- />
- <span className="flex flex-col gap-1">
- <span className="font-medium">Capture speaker diarization</span>
- <span className="text-xs text-muted-foreground">
- Turning this on also <strong>holds disk</strong>: the Clean-audio
- sweep stops deleting audio for a transcribed video until its
- sidecar exists, and each channel's "Est. reclaim"
- drops to match. That hold is the point — it is what keeps the
- perishable input alive long enough to capture — but it means free
- space stops being reclaimed until diarization catches up.
- </span>
- </span>
- </label>
- <label className="flex items-start gap-2 text-sm">
- <input
- type="checkbox"
- name="diarizationInlineAfterTranscribe"
- defaultChecked={initial.diarization.inlineAfterTranscribe}
- className="mt-1"
- />
- <span className="flex flex-col gap-1">
- <span className="font-medium">
- Diarize inline, right after each transcription
- </span>
- <span className="text-xs text-muted-foreground">
- Off by default, from measurement rather than caution. On this box
- transcription runs at ~221 s per audio-hour on the GPU (16× real
- time, over 3,602 videos); diarization runs on the CPU at roughly
- 500–680, because no diarization model has been ported to ggml and
- neither ONNX nor PyTorch has a Vulkan path on Linux. Inline
- therefore makes the whole pipeline ~3–4× slower and idles the GPU.
- For a large batch, leave this off, let transcription run at full
- speed with the box above holding the audio, and catch up
- afterwards with <strong>Diarize speakers</strong> on each channel.
- Turn it on for steady state.
- </span>
- </span>
- </label>
- <label className="flex flex-col gap-1 text-sm">
- <span className="font-medium">Engine</span>
- <select
- name="diarizationEngine"
- defaultValue={initial.diarization.engine}
- className="rounded border border-border bg-card px-2 py-1 text-sm"
- >
- <option value="sherpa-onnx">
- sherpa-onnx — clustered, CPU only (default)
- </option>
- <option value="sortformer">
- Sortformer — end-to-end, GPU or CPU
- </option>
- </select>
- <span className="text-xs text-muted-foreground">
- <strong>sherpa-onnx</strong> groups voices that sound alike, and it
- splits far too eagerly: on a 13-minute video with one host it finds
- 13 speakers, and on the worst file in this corpus it finds 35.{" "}
- <strong>Sortformer</strong> decides turns directly instead of
- grouping them afterwards and returns 4 in both cases, agreeing with
- the other engine about how much of the video the main speaker talks
- for. It has no threshold, and it caps at 4 speakers — a panel of
- five will merge two rather than invent twenty.
- <br />
- <br />
- sherpa-onnx is the <strong>faster</strong> of the two (492 against
- 894 seconds per audio-hour here), so this is a quality choice, not a
- speed one. Switching also{" "}
- <strong>marks every recording captured by the other engine as work
- to redo</strong>, because the two disagree about how many speakers
- exist — on the audio still held here that is weeks of it.
- </span>
- </label>
- <label className="flex flex-col gap-1 text-sm">
- <span className="font-medium">Sortformer device</span>
- <select
- name="diarizationBackend"
- defaultValue={initial.diarization.backend}
- className="rounded border border-border bg-card px-2 py-1 text-sm"
- >
- <option value="vulkan">Vulkan — the graphics card</option>
- <option value="cpu">CPU</option>
- </select>
- <span className="text-xs text-muted-foreground">
- Ignored unless the engine above is Sortformer. Both produce{" "}
- <strong>identical</strong> speaker turns, so this only trades one
- resource for another: the card is ~1.5× faster and uses one core
- instead of two, but takes about 4.4 GB of the same 8 GB card
- transcription uses — so capture stands aside while transcription is
- working, and resumes when the card is free. On CPU it needs 4.84 GB
- of system memory instead.
- </span>
- </label>
- <Field
- label="Sortformer engine binary"
- name="diarizationSortformerBin"
- defaultValue={initial.diarization.sortformerBin}
- hint="Absolute path to the diarize-file binary built by scripts/build-sortformer.sh. Empty means the Sortformer engine is not configured, and every run reports a skip rather than a failure."
- />
- <Field
- label="Sortformer model (GGUF)"
- name="diarizationSortformerModel"
- defaultValue={initial.diarization.sortformerModel}
- hint="Absolute path to the .gguf downloaded by scripts/build-sortformer.sh. Its filename is recorded in every sidecar, so changing the model is what marks earlier captures worth redoing."
- />
- <Field
- label="Segmentation model (ONNX)"
- name="diarizationSegModel"
- defaultValue={initial.diarization.segModel}
- hint="Absolute path to a pyannote segmentation-3.0 ONNX model. Empty means the lane is not configured and every run reports a skip rather than a failure."
- />
- <Field
- label="Speaker-embedding model (ONNX)"
- name="diarizationEmbModel"
- defaultValue={initial.diarization.embModel}
- hint="Absolute path to a speaker-embedding ONNX model (e.g. NeMo TitaNet)."
- />
- <Field
- label="Python interpreter"
- name="diarizationPython"
- defaultValue={initial.diarization.python}
- hint="Interpreter with sherpa-onnx installed. sherpa-onnx ships wheels only up to CPython 3.13, so on a 3.14 system this must point at a dedicated venv."
- />
- <Field
- label="Clustering threshold"
- name="diarizationThreshold"
- defaultValue={String(initial.diarization.threshold)}
- type="number"
- step="0.01"
- hint="The knob that decides how many speakers come out; larger merges more. It is recorded in every sidecar, so changing it is what lets a later pass tell which files are worth regenerating."
- />
- <Field
- label="Engine threads"
- name="diarizationThreads"
- defaultValue={String(initial.diarization.threads)}
- type="number"
- hint="Threads per diarize run."
- />
- <Field
- label="Max audio hours"
- name="diarizationMaxAudioHours"
- defaultValue={String(initial.diarization.maxAudioHours)}
- type="number"
- step="0.5"
- hint="Videos longer than this are reported as deferred instead of being diarized, and are never counted as work still to do. This is a stopgap for an out-of-memory crash on very long recordings: the engine's memory use grows with the SQUARE of the number of speaker turns, so a dense six-hour stream can exhaust 16 GB after 40 minutes of work and produce nothing. 0 turns the limit off."
- />
- <Field
- label="Diarization concurrency"
- name="diarizationConcurrency"
- defaultValue={String(initial.diarization.concurrency)}
- type="number"
- hint="Videos diarized at once during a backfill. Default 1: this is CPU-bound work competing with GPU feeding and the digest sweep for the same 8 threads."
- />
- </fieldset>
- <fieldset className="flex flex-col gap-3 border border-border rounded p-3">
- {/* THE ONE FIELDSET THAT STILL SAYS "LANE", and it is the right one:
- this is where the shared pause and the shared sweep live, and that
- these operations share a queue is a load-bearing fact rather than a
- leaked implementation detail. Everywhere an operator reads a FIGURE
- they now read an operation name instead. */}
- <legend className="px-1 text-sm font-medium">
- Speaker work lane
- </legend>
- {/*
- The marker again, and here it guards two things an unrelated save must
- never touch: `allowRedownload` (which writes media to a nearly-full
- disk) and the sweep flag + scope, which this form does not render at
- all — reading those from absent fields would disarm a running
- multi-day sweep on any settings save.
- */}
- <input type="hidden" name="backfillFormPresent" value="1" readOnly />
- <p className="text-xs text-muted-foreground">
- Catch-up for derived data the existing corpus predates — speaker
- diarization and the two speaker-name lanes below. These operations
- share one queue and one pause: the settings here decide how much of
- the machine all of them together may use, not any one of them
- individually. Each feature declares what it needs and how to tell
- whether a video already has it.
- </p>
- <p className="text-xs text-muted-foreground">
- <strong className="font-medium text-foreground">
- Which operations a corpus sweep runs, and over which channels, is
- not set here.
- </strong>{" "}
- That scope has to be written at the moment the sweep is armed — a
- scope saved separately would be resurrected as a corpus-wide run by
- the next restart — so it lives with the plan it produces, on{" "}
- <Link
- href="/operations"
- className="underline underline-offset-2 hover:text-foreground"
- >
- Operations
- </Link>
- , where you can read the itinerary before committing to it.
- </p>
- <label className="flex items-start gap-2 text-sm">
- <input
- type="checkbox"
- name="backfillEnabled"
- defaultChecked={initial.backfill.enabled}
- className="mt-1"
- />
- <span className="flex flex-col gap-1">
- <span className="font-medium">Run the backfill lane</span>
- <span className="text-xs text-muted-foreground">
- Off means the lane still <em>reports</em> what is missing —
- that is what the indicators are for — but runs nothing.
- </span>
- </span>
- </label>
- <Field
- label="Resource share"
- name="backfillWeight"
- defaultValue={String(initial.backfill.weight)}
- type="number"
- step="0.05"
- hint="0 (the default) means idle-only: the lane runs only while transcription is quiet, and stands aside the moment it isn't. Above 0 it takes that fraction of its slots as a guaranteed share, floored at 1 — so a small number is a slow lane, not a stopped one."
- />
- <Field
- label="Lane concurrency"
- name="backfillConcurrency"
- defaultValue={String(initial.backfill.concurrency)}
- type="number"
- hint="Slots the lane may use when it is not standing aside. Default 1 — this is CPU-bound work competing with GPU feeding and the digest sweep for the same threads."
- />
- <label className="flex items-start gap-2 text-sm">
- <input
- type="checkbox"
- name="backfillAllowRedownload"
- defaultChecked={initial.backfill.allowRedownload}
- className="mt-1"
- />
- <span className="flex flex-col gap-1">
- <span className="font-medium">
- Re-acquire media that has already been deleted
- </span>
- <span className="text-xs text-muted-foreground">
- Off by default, and that default is measured: 836 videos still
- have media on disk and ~76,270 would need re-downloading — 91×
- the reachable work, against 45 GB free. With this on, each file is
- fetched, used, and <strong>deleted again immediately</strong>{" "}
- (unless the video is marked "do not clean"), and nothing
- starts at all when free space is under the disk floor.
- </span>
- </span>
- </label>
- </fieldset>
- <fieldset className="flex flex-col gap-3 border border-border rounded p-3">
- <legend className="px-1 text-sm font-medium">Speaker attribution</legend>
- {/*
- The marker, for the third time, and here it guards the most expensive
- thing in this form: `textOnlyEnabled` arms a lane that costs roughly
- one model call per transcript CHUNK — ~194,000 across this corpus, the
- same order as the digest sweep. An unrelated settings save must not be
- able to switch that on.
- */}
- <input type="hidden" name="attributionFormPresent" value="1" readOnly />
- <p className="text-xs text-muted-foreground">
- Putting names to the speakers. Nothing here runs on its own — it
- registers two operations, and the Speaker work lane above decides when
- they get the machine. Both are on the same shared pause as speaker
- diarization.
- </p>
- <label className="flex items-start gap-2 text-sm">
- <input
- type="checkbox"
- name="attributionEnabled"
- defaultChecked={initial.attribution.enabled}
- className="mt-1"
- />
- <span className="flex flex-col gap-1">
- <span className="font-medium">Enable speaker attribution</span>
- <span className="text-xs text-muted-foreground">
- Off means neither lane below reports work, whatever their own
- switches say.
- </span>
- </span>
- </label>
- <label className="flex items-start gap-2 text-sm">
- <input
- type="checkbox"
- name="attributionDiarized"
- defaultChecked={initial.attribution.diarizedEnabled}
- className="mt-1"
- />
- <span className="flex flex-col gap-1">
- <span className="font-medium">
- Name the speakers found in the audio
- </span>
- <span className="text-xs text-muted-foreground">
- About <strong>one model call per video</strong>: the diarizer
- already grouped the voices, so the model only has to put a name to
- each group. Needs <code>diarization.json</code>, which is why
- videos without it are counted separately.
- </span>
- </span>
- </label>
- <label className="flex items-start gap-2 text-sm">
- <input
- type="checkbox"
- name="attributionTextOnly"
- defaultChecked={initial.attribution.textOnlyEnabled}
- className="mt-1"
- />
- <span className="flex flex-col gap-1">
- <span className="font-medium">
- Guess the speakers from the transcript
- </span>
- <span className="text-xs text-muted-foreground">
- For videos with no diarization. Roughly{" "}
- <strong>one call per transcript chunk</strong> — about 194,000
- across this archive, or 25–55 days of GPU time, which is the same
- order as the digest sweep and would compete with it. Quality is
- genuinely uncertain: it misfires on rapid back-and-forth, and
- auto-caption channels have no speaker turns in the first place. It
- never overwrites a record made from the audio.
- </span>
- </span>
- </label>
- <label className="flex flex-col gap-1 text-sm">
- <span className="font-medium">Engine</span>
- <select
- name="attributionAppId"
- defaultValue={initial.attribution.appId}
- className="rounded border border-border bg-card px-2 py-1 text-sm"
- >
- {digestApps.map((a) => (
- <option key={a.id} value={a.id}>
- {a.label}
- {a.metered ? " (metered)" : ""}
- </option>
- ))}
- </select>
- <span className="text-xs text-muted-foreground">
- Uses the digest engines and their configuration above — same
- constrained-JSON workload, so there is no second copy of the URL and
- context size to keep in step.
- </span>
- </label>
- <Field
- label="Model override"
- name="attributionModel"
- defaultValue={initial.attribution.model}
- hint="Empty means the engine's own configured model. Set separately from the digest's so a digest bake-off does not invalidate every attribution record as a side effect — this value is part of what makes a record fresh."
- />
- <Field
- label="Prompt generation"
- name="attributionPromptVersion"
- defaultValue={String(initial.attribution.promptVersion)}
- type="number"
- hint="A record generated at an older number is stale and will be redone. Raise it to force a corpus-wide regeneration after changing the prompt; it cannot be set below the shipped value, because pinning it lower would freeze superseded output into the archive looking current."
- />
- </fieldset>
- <fieldset className="flex flex-col gap-3 border border-border rounded p-3">
<legend className="px-1 text-sm font-medium">Social links</legend>
<p className="text-xs text-muted-foreground">
Default social links shown in every site's footer. Each site can
diff --git a/editor/app/settings/page.tsx b/editor/app/settings/page.tsx
@@ -2,7 +2,6 @@ import type { Metadata } from "next";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import { getSettings } from "yt-dlp-transcript-common/lib/settings";
import { listTranscriptionApps } from "yt-dlp-transcript-common/lib/transcriptionApps";
-import { listDigestApps } from "yt-dlp-transcript-common/lib/digestApps";
import { operationCatalog } from "yt-dlp-transcript-common/lib/operations";
import { WORKER_RESOURCE_TAGS } from "yt-dlp-transcript-common/lib/workers";
import { readXSessionStatus } from "yt-dlp-transcript-common/social/xSessionBroker";
@@ -48,7 +47,6 @@ export default async function SettingsPage() {
<SettingsForm
initial={settings}
apps={listTranscriptionApps()}
- digestApps={listDigestApps()}
workerTags={[
...new Set([
...operationCatalog().map((o) => o.id),
diff --git a/editor/e2e/operation-settings.spec.ts b/editor/e2e/operation-settings.spec.ts
@@ -0,0 +1,242 @@
+import { test, expect } from "@playwright/test";
+import { readJson, resetData } from "./helpers";
+
+// PER-OPERATION SETTINGS, on the operation's own page.
+//
+// Slice 3 moved four fieldsets off /settings — Digest, Diarization, the speaker
+// work LANE and Speaker attribution — onto /operations/<id>, each with its own
+// <form> and its own action. That split is what retired the hidden
+// `*FormPresent` markers: while one form saved everything, an unchecked
+// checkbox (absent from a FormData) submitted by a form lacking the block would
+// have read every switch as off, so each block needed a marker to prove its
+// fields were on the page. Separate forms make that structural instead.
+//
+// Every field NAME and every legend is unchanged from the old fieldsets, so
+// these are the same assertions against a different route — except the last
+// two, which cover blocks that had no form-driving spec at all before.
+
+test.beforeEach(async () => {
+ await resetData("empty");
+});
+
+test("digest settings round-trip through the form", async ({ page }) => {
+ await page.goto("/operations/digest");
+ const form = page.locator('form[data-settings-block="digest"]');
+
+ await form.getByLabel("Timestamp mode").selectOption("absolute");
+ await form.getByLabel(/prompt variant label/i).fill("bakeoff-r3");
+ await form.getByLabel(/spend cap/i).fill("12.5");
+ await form.getByLabel(/long-tail cutoff/i).fill("7200");
+ await form.getByRole("checkbox", { name: /enable the metered/i }).check();
+ // Per-app numCtx rides in the hidden digestAppsJson payload.
+ await form.getByText("Per-engine configuration").click();
+ await form
+ .getByLabel("Context window (num_ctx) for ollama-direct")
+ .fill("4096");
+
+ await form.getByRole("button", { name: "Save digest settings" }).click();
+ await expect(form.getByRole("status")).toHaveText("Saved.");
+
+ const saved = await readJson<{
+ digest?: {
+ timestampMode: string;
+ promptVariant: string;
+ spendCapUsd: number;
+ longTailSeconds: number;
+ remoteEnabled: boolean;
+ sections: string[];
+ apps: Record<string, { numCtx?: number }>;
+ };
+ }>("test-settings.json");
+ expect(saved.digest?.timestampMode).toBe("absolute");
+ expect(saved.digest?.promptVariant).toBe("bakeoff-r3");
+ expect(saved.digest?.spendCapUsd).toBe(12.5);
+ expect(saved.digest?.longTailSeconds).toBe(7200);
+ expect(saved.digest?.remoteEnabled).toBe(true);
+ expect(saved.digest?.apps["ollama-direct"]?.numCtx).toBe(4096);
+ // Never left empty — an empty section list would generate nothing.
+ expect(saved.digest?.sections.length).toBeGreaterThan(0);
+});
+
+test("a save on /settings does not reset the digest prompt shape", async ({
+ page,
+}) => {
+ // THE BUG THIS EXISTS FOR, and it is now a genuinely CROSS-FORM test: the
+ // digest action rebuilds the whole block on every save, so any
+ // freshness-affecting field it fails to carry through is silently reset — and
+ // timestampMode/promptVariant are folded into the recorded identity, so a
+ // reset would invalidate every digest generated under the non-default shape
+ // while looking like a no-op. What used to protect the block on an unrelated
+ // save was a hidden marker; what protects it now is that the unrelated save
+ // happens in a different form on a different page and reads
+ // `getSettings().digest` verbatim.
+ await page.goto("/operations/digest");
+ const digest = page.locator('form[data-settings-block="digest"]');
+ await digest.getByLabel("Timestamp mode").selectOption("absolute");
+ await digest.getByLabel(/prompt variant label/i).fill("keepme");
+ await digest.getByRole("button", { name: "Save digest settings" }).click();
+ await expect(digest.getByRole("status")).toHaveText("Saved.");
+
+ // Reload rather than submitting straight again: React 19 resets <form action>
+ // inputs to the defaultValue of the render they were in, so a second submit
+ // from the same render would re-post the digest values from BEFORE the first
+ // save and prove nothing. A reload is also the real scenario — the operator
+ // sets the digest up, comes back later, and changes something unrelated.
+ await page.reload();
+ await expect(digest.getByLabel("Timestamp mode")).toHaveValue("absolute");
+
+ await page.goto("/settings");
+ await page.getByLabel(/admin title/i).fill("Unrelated Change");
+ await page.getByRole("button", { name: /save settings/i }).click();
+ await expect(
+ page.getByRole("status").filter({ hasText: "Saved" }),
+ ).toBeVisible();
+
+ const saved = await readJson<{
+ adminTitle: string;
+ digest?: { timestampMode: string; promptVariant: string };
+ }>("test-settings.json");
+ expect(saved.adminTitle).toBe("Unrelated Change");
+ expect(saved.digest?.timestampMode).toBe("absolute");
+ expect(saved.digest?.promptVariant).toBe("keepme");
+});
+
+test("the lane's switch survives a save of the operation form beside it", async ({
+ page,
+}) => {
+ // THE PROPERTY THE MARKERS USED TO PROTECT, proven structurally. Two forms
+ // sit on this one page — the diarization block and the shared backfill LANE
+ // block — and `backfillEnabled` is a checkbox, so it is absent from the
+ // diarization form's FormData entirely. Before the split, one FormData
+ // carried both and only a hidden marker stopped the second from reading the
+ // first's absent checkbox as off.
+ await page.goto("/operations/diarization");
+ const lane = page.locator('form[data-settings-block="backfill"]');
+ await lane.getByRole("checkbox", { name: "Run the backfill lane" }).check();
+ await lane.getByRole("button", { name: "Save lane settings" }).click();
+ await expect(lane.getByRole("status")).toHaveText("Saved.");
+
+ const diarization = page.locator('form[data-settings-block="diarization"]');
+ await diarization
+ .getByRole("checkbox", { name: "Capture speaker diarization" })
+ .check();
+ await diarization
+ .getByRole("button", { name: "Save diarization settings" })
+ .click();
+ await expect(diarization.getByRole("status")).toHaveText("Saved.");
+
+ const saved = await readJson<{
+ backfill?: { enabled: boolean };
+ diarization?: { enabled: boolean };
+ }>("test-settings.json");
+ expect(saved.backfill?.enabled).toBe(true);
+ expect(saved.diarization?.enabled).toBe(true);
+});
+
+// THE BUG THIS EXISTS FOR: the sortformer engine shipped with its settings
+// reachable only by hand-editing settings.json, because the form simply had no
+// controls for them. A field the form does not render is a field an operator
+// cannot set — and, worse, the diarization action rebuilds the whole block on
+// save, so a control added to the markup but forgotten in the action would look
+// like it worked and silently revert.
+test("the sortformer engine settings round-trip through the form", async ({
+ page,
+}) => {
+ await page.goto("/operations/diarization");
+ const form = page.locator('form[data-settings-block="diarization"]');
+
+ // Targeted by NAME, not by label. getByLabel matches by substring and these
+ // labels wrap their whole explanatory hint, so the accessible name is a
+ // paragraph — "Engine" matches nothing exactly and matches "Engine threads"
+ // loosely. The control's name is the thing the action actually reads.
+ await form.locator('select[name="diarizationEngine"]').selectOption("sortformer");
+ await form.locator('select[name="diarizationBackend"]').selectOption("cpu");
+ await form
+ .getByLabel("Sortformer engine binary")
+ .fill("/opt/sortformer/diarize-file");
+ await form
+ .getByLabel("Sortformer model (GGUF)")
+ .fill("/opt/sortformer/model.gguf");
+
+ await form.getByRole("button", { name: "Save diarization settings" }).click();
+ await expect(form.getByRole("status")).toHaveText("Saved.");
+
+ const saved = await readJson<{
+ diarization?: {
+ engine: string;
+ backend: string;
+ sortformerBin: string;
+ sortformerModel: string;
+ segModel: string;
+ threshold: number;
+ };
+ }>("test-settings.json");
+ expect(saved.diarization?.engine).toBe("sortformer");
+ expect(saved.diarization?.backend).toBe("cpu");
+ expect(saved.diarization?.sortformerBin).toBe("/opt/sortformer/diarize-file");
+ expect(saved.diarization?.sortformerModel).toBe("/opt/sortformer/model.gguf");
+ // The sherpa fields are still carried, not clobbered by selecting the other
+ // engine — switching back must not mean re-entering three paths.
+ expect(saved.diarization?.threshold).toBe(0.9);
+});
+
+// The attribution block had NO form-driving spec at all before slice 3, which
+// is how the most expensive switch in the console — textOnlyEnabled, ~194,000
+// model calls across this corpus — went untested from a browser.
+test("attribution settings round-trip on the text-only operation's page", async ({
+ page,
+}) => {
+ await page.goto("/operations/attribution-text");
+ const form = page.locator('form[data-settings-block="attribution"]');
+
+ await form
+ .getByRole("checkbox", { name: "Enable speaker attribution" })
+ .check();
+ await form
+ .getByRole("checkbox", { name: "Guess the speakers from the transcript" })
+ .check();
+ await form.getByLabel("Model override").fill("qwen2.5:14b");
+ await form.getByRole("button", { name: "Save attribution settings" }).click();
+ await expect(form.getByRole("status")).toHaveText("Saved.");
+
+ const saved = await readJson<{
+ attribution?: {
+ enabled: boolean;
+ textOnlyEnabled: boolean;
+ model: string;
+ };
+ }>("test-settings.json");
+ expect(saved.attribution?.enabled).toBe(true);
+ expect(saved.attribution?.textOnlyEnabled).toBe(true);
+ expect(saved.attribution?.model).toBe("qwen2.5:14b");
+});
+
+test("both attribution operations draw the same block, and the lane's", async ({
+ page,
+}) => {
+ // ONE BLOCK, TWO OPERATIONS: both declare settingsBlock "attribution", so
+ // both pages render this form and either one saves the same fields. The lane
+ // form is beside it on both, because all three speaker operations share
+ // BACKFILL_QUEUE.
+ await page.goto("/operations/attribution-diarized");
+ await expect(
+ page.locator('form[data-settings-block="attribution"]'),
+ ).toBeVisible();
+ await expect(
+ page.getByRole("group", { name: "Speaker attribution" }),
+ ).toBeVisible();
+ await expect(
+ page.getByRole("group", { name: "Speaker work lane" }),
+ ).toBeVisible();
+});
+
+test("an operation with no settings block renders no settings form", async ({
+ page,
+}) => {
+ // The switch is over the DESCRIPTOR's block, so an operation that declares
+ // none gets nothing — not an empty fieldset, and not another operation's.
+ await page.goto("/operations/transcode");
+ await expect(page.locator("form[data-settings-block]")).toHaveCount(0);
+ await page.goto("/operations/download");
+ await expect(page.locator("form[data-settings-block]")).toHaveCount(0);
+});
diff --git a/editor/e2e/settings.spec.ts b/editor/e2e/settings.spec.ts
@@ -166,140 +166,3 @@ test("saves build pipeline settings (mode, concurrency, image)", async ({
.getByRole("button", { name: "Docker" }),
).toHaveAttribute("aria-pressed", "true");
});
-
-// ---------------------------------------------------------------------------
-// Digest
-// ---------------------------------------------------------------------------
-//
-// Every digest knob used to be unreachable from a browser: saveSettingsAction
-// had a full digest branch gated on a `digestFormPresent` marker that NOTHING in
-// the repo emitted, while two places told the operator to "enable it in
-// Settings → Digest" — a section that did not exist.
-
-test("digest settings round-trip through the form", async ({ page }) => {
- await page.goto("/settings");
-
- await page.getByLabel("Timestamp mode").selectOption("absolute");
- await page.getByLabel(/prompt variant label/i).fill("bakeoff-r3");
- await page.getByLabel(/spend cap/i).fill("12.5");
- await page.getByLabel(/long-tail cutoff/i).fill("7200");
- await page
- .getByRole("checkbox", { name: /enable the metered/i })
- .check();
- // Per-app numCtx rides in the hidden digestAppsJson payload.
- await page.getByRole("group", { name: "Digest" }).getByText(
- "Per-engine configuration",
- ).click();
- await page.getByLabel("Context window (num_ctx) for ollama-direct").fill("4096");
-
- await page.getByRole("button", { name: /save settings/i }).click();
- await expect(
- page.getByRole("status").filter({ hasText: "Saved" }),
- ).toBeVisible();
-
- const saved = await readJson<{
- digest?: {
- timestampMode: string;
- promptVariant: string;
- spendCapUsd: number;
- longTailSeconds: number;
- remoteEnabled: boolean;
- sections: string[];
- apps: Record<string, { numCtx?: number }>;
- };
- }>("test-settings.json");
- expect(saved.digest?.timestampMode).toBe("absolute");
- expect(saved.digest?.promptVariant).toBe("bakeoff-r3");
- expect(saved.digest?.spendCapUsd).toBe(12.5);
- expect(saved.digest?.longTailSeconds).toBe(7200);
- expect(saved.digest?.remoteEnabled).toBe(true);
- expect(saved.digest?.apps["ollama-direct"]?.numCtx).toBe(4096);
- // Never left empty — an empty section list would generate nothing.
- expect(saved.digest?.sections.length).toBeGreaterThan(0);
-});
-
-test("an unrelated settings save does not reset the digest prompt shape", async ({
- page,
-}) => {
- // THE BUG THIS EXISTS FOR. The digest branch REBUILDS the whole block on every
- // save, so any freshness-affecting field it fails to carry through is silently
- // reset — and timestampMode/promptVariant are folded into the recorded
- // identity, so a reset would invalidate every digest generated under the
- // non-default shape while looking like a no-op.
- await page.goto("/settings");
- await page.getByLabel("Timestamp mode").selectOption("absolute");
- await page.getByLabel(/prompt variant label/i).fill("keepme");
- await page.getByRole("button", { name: /save settings/i }).click();
- await expect(
- page.getByRole("status").filter({ hasText: "Saved" }),
- ).toBeVisible();
-
- // Reload rather than submitting straight again: React 19 resets <form action>
- // inputs to the defaultValue of the render they were in, so a second submit
- // from the same render would re-post the digest values from BEFORE the first
- // save and prove nothing. A reload is also the real scenario — the operator
- // sets the digest up, comes back later, and changes something unrelated.
- await page.reload();
- await expect(page.getByLabel("Timestamp mode")).toHaveValue("absolute");
- await page.getByLabel(/admin title/i).fill("Unrelated Change");
- await page.getByRole("button", { name: /save settings/i }).click();
- await expect(
- page.getByRole("status").filter({ hasText: "Saved" }),
- ).toBeVisible();
-
- const saved = await readJson<{
- adminTitle: string;
- digest?: { timestampMode: string; promptVariant: string };
- }>("test-settings.json");
- expect(saved.adminTitle).toBe("Unrelated Change");
- expect(saved.digest?.timestampMode).toBe("absolute");
- expect(saved.digest?.promptVariant).toBe("keepme");
-});
-
-// THE BUG THIS EXISTS FOR: the sortformer engine shipped with its settings
-// reachable only by hand-editing settings.json, because the form simply had no
-// controls for them. A field the form does not render is a field an operator
-// cannot set — and, worse, the diarization branch rebuilds the whole block on
-// save, so a control added to the markup but forgotten in the action would look
-// like it worked and silently revert.
-test("the sortformer engine settings round-trip through the form", async ({
- page,
-}) => {
- await page.goto("/settings");
-
- // Targeted by NAME, not by label. getByLabel matches by substring and these
- // labels wrap their whole explanatory hint, so the accessible name is a
- // paragraph — "Engine" matches nothing exactly and matches "Engine threads"
- // loosely. The control's name is the thing the action actually reads.
- await page.locator('select[name="diarizationEngine"]').selectOption("sortformer");
- await page.locator('select[name="diarizationBackend"]').selectOption("cpu");
- await page
- .getByLabel("Sortformer engine binary")
- .fill("/opt/sortformer/diarize-file");
- await page
- .getByLabel("Sortformer model (GGUF)")
- .fill("/opt/sortformer/model.gguf");
-
- await page.getByRole("button", { name: /save settings/i }).click();
- await expect(
- page.getByRole("status").filter({ hasText: "Saved" }),
- ).toBeVisible();
-
- const saved = await readJson<{
- diarization?: {
- engine: string;
- backend: string;
- sortformerBin: string;
- sortformerModel: string;
- segModel: string;
- threshold: number;
- };
- }>("test-settings.json");
- expect(saved.diarization?.engine).toBe("sortformer");
- expect(saved.diarization?.backend).toBe("cpu");
- expect(saved.diarization?.sortformerBin).toBe("/opt/sortformer/diarize-file");
- expect(saved.diarization?.sortformerModel).toBe("/opt/sortformer/model.gguf");
- // The sherpa fields are still carried, not clobbered by selecting the other
- // engine — switching back must not mean re-entering three paths.
- expect(saved.diarization?.threshold).toBe(0.9);
-});