commit a7501cb3f4b84831663a44b93cc68c68252d05d7
parent 8d0e46617961beba4f1303d90eec7cc021844ad0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 23 Sep 2026 20:13:09 -0400
merge: one-core/phase-3-s4a — one settings schema (zod), one editor writer, SETTINGS.md generated
Phase 3 slice 4a. common/lib/settingsSchema.ts is the one zod schema for
settings.json (read lenient, write strict, unknown keys stripped, clamps
reused not rewritten); the autoQueue sanitizer lives in lib/autoQueueSchema.ts
(allow-list 10 → 9); editor writes go through saveSettings(patch) from one
file; settings.json.example and SETTINGS.md are generated from the schema and
its per-block field docs, with a test pinning the committed bytes. Gates: tsc,
common 1663, editor unit 67, scripts 156, mcp 219, both builds green with no
zod in client chunks, e2e 157/157; settings numbers diff empty on read and
write over frozen inputs.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
39 files changed, 4693 insertions(+), 2229 deletions(-)
diff --git a/SETTINGS.md b/SETTINGS.md
@@ -0,0 +1,723 @@
+# settings.json keys
+
+<!-- GENERATED by common/bin/settings-example.ts from common/lib/settingsSchema.ts and the *_FIELD_DOCS records beside each block type — do not edit by hand. -->
+
+Global operational settings shared by every site this editor powers, persisted to `settings.json` at the repo root (or `$SETTINGS_FILE`). Per-site presentation lives in `sites/<id>/site.json`. Every key is optional: a missing key reads as its default, an ill-typed one is coerced to its default or clamped, and an unknown one is dropped on the next save.
+
+Regenerate this file and `settings.json.example` with `pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts`.
+
+`settings.json.example` is the default object with one key left out, `workers`: a file that does not name it gets a worker list synthesized from `transcriptionApp` on read. A file that spells `workers: []` READS as no transcription at all — until the next save, when the writer synthesizes a worker the same way.
+
+A copied example PINS every default it spells — including each lane's `autoQueue.<lane>.held` — so a default changed in a later release will not reach that file. Delete any key you would rather have track the defaults.
+
+| Key | Default |
+|---|---|
+| [`adminTitle`](#admintitle) | `"Transcript Browser Admin"` |
+| [`maxTranscriptPageBytes`](#maxtranscriptpagebytes) | `8388608` |
+| [`transcriptionApp`](#transcriptionapp) | `"whisper-cpp"` |
+| [`transcriptionApps`](#transcriptionapps) | `{}` |
+| [`workers`](#workers) | `[]` |
+| [`cookiesFromBrowser`](#cookiesfrombrowser) | `""` |
+| [`cookieMode`](#cookiemode) | `"when-required"` |
+| [`sleepBetweenDownloadsSeconds`](#sleepbetweendownloadsseconds) | `10` |
+| [`downloadFormat`](#downloadformat) | `"auto"` |
+| [`minFreeDiskGB`](#minfreediskgb) | `5` |
+| [`resumeMarginGB`](#resumemargingb) | `2` |
+| [`parallelTranscriptions`](#paralleltranscriptions) | `2` |
+| [`inlineTranscribeOnFallback`](#inlinetranscribeonfallback) | `false` |
+| [`skipLiveDownloads`](#skiplivedownloads) | `true` |
+| [`verifyAvailabilityBeforeClean`](#verifyavailabilitybeforeclean) | `true` |
+| [`buildArchives`](#buildarchives) | `true` |
+| [`archiveStorage`](#archivestorage) | object — see below |
+| [`reportDebouncePreset`](#reportdebouncepreset) | `"fast"` |
+| [`autoRefreshIntervalSeconds`](#autorefreshintervalseconds) | `5` |
+| [`syncScheduler`](#syncscheduler) | object — see below |
+| [`autoQueue`](#autoqueue) | object — see below |
+| [`channelPriority`](#channelpriority) | object — see below |
+| [`socialLinks`](#sociallinks) | `[]` |
+| [`homepageUrl`](#homepageurl) | `""` |
+| [`savedVideoBackup`](#savedvideobackup) | object — see below |
+| [`storage`](#storage) | object — see below |
+| [`buildPipeline`](#buildpipeline) | object — see below |
+| [`digest`](#digest) | object — see below |
+| [`diarization`](#diarization) | object — see below |
+| [`backfill`](#backfill) | object — see below |
+| [`attribution`](#attribution) | object — see below |
+
+## `adminTitle`
+
+Title for the EDITOR admin shell only (the editor manages all sites and so is not tied to any one site's branding). Public sites get their own titles from site.json.
+
+Default: `"Transcript Browser Admin"`
+
+## `maxTranscriptPageBytes`
+
+Target size (bytes) of one exported transcript page shard — the unit the export site fetches. Clamped into [TRANSCRIPT_PAGE_MIN_BYTES, TRANSCRIPT_PAGE_HARD_CAP_BYTES] (256 KiB – 20 MiB); default 8 MiB.
+
+Default: `8388608`
+
+## `transcriptionApp`
+
+Active transcription app id (key into TRANSCRIPTION_APPS, e.g. "whisper-cpp" or "chough"). Selected globally; see common/lib/transcriptionApps.ts.
+
+Default: `"whisper-cpp"`
+
+## `transcriptionApps`
+
+Per-app configuration, keyed by app id. Each app reads only its own block; a missing block means "use the app's defaults". DEPRECATED in favor of `workers` (each local worker carries its own config); kept one release to drive migration and allow rollback. See common/lib/workers.ts.
+
+#### `transcriptionApps.<appId>`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `bin` | Binary path/name override. Empty/undefined falls back to app.defaultBin(). |
+| `model` | whisper.cpp model path (substituted for {model}); for chough this is the optional CHOUGH_MODEL env (chough auto-downloads a model when unset). |
+| `remoteUrl` | chough remote server URL (CHOUGH_URL). Empty/undefined = local transcription. |
+| `chunkSize` | chough chunk size in seconds (-c). Undefined = chough's own default. |
+| `customArgs` | whisper.cpp custom argv template using the {audioFile}/{outputBase}/{model} placeholders. Undefined = DEFAULT_TRANSCRIBE_ARGS. |
+| `device` | parakeet compute device passed to parakeet-cli (--device / PARAKEET_DEVICE), e.g. "cuda:0", "cpu". Undefined = parakeet-cli's default device. |
+
+Default:
+
+```json
+{}
+```
+
+## `workers`
+
+Configured transcription workers (named processing slots). The scheduler distributes each video to the highest-priority free worker. A settings.json predating this field is migrated to a single enabled worker from the active app (see defaultWorkersFromApps). See common/lib/workers.ts.
+
+#### `workers[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `id` | Stable slug; used in settings, task ids, and logs. |
+| `name` | Human label shown in the UI. |
+| `kind` | "local" runs an app from TRANSCRIPTION_APPS on this machine (`appId` + `config`); "remote" delegates to another instance on the LAN (`remote`); "llm" is a bare ollama endpoint serving digest/attribution calls only (`llm`). |
+| `enabled` | Whether the scheduler may hand this slot work. Each worker is one slot, so parallelism is toggled per slot on the Workers page. Anything but an explicit `false` reads as enabled. |
+| `priority` | Lower = preferred. Ties broken by array order in the scheduler. |
+| `tags` | Capability routing. A tag is an OPERATION id from the backfill catalog ("diarization", "attribution-text", …) or a contended RESOURCE (WORKER_RESOURCE_TAGS). The scheduler consults them through workerMatches below: an untagged worker takes anything, a tagged worker takes only work whose requirement intersects its tags. Unknown tags are tolerated (they match nothing and warn in the settings UI), never fatal. |
+| `appId` | LOCAL: an instance of a TRANSCRIPTION_APPS entry + its per-worker config. |
+| `config` | LOCAL: the per-worker engine config (binary, model, device, …) — an AppInstanceConfig, see `transcriptionApps.<appId>`. |
+| `remote` | REMOTE: how to reach the delegate instance. |
+| `llm` | LLM: how to reach the bare model endpoint. |
+
+#### `workers[].config`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `bin` | Binary path/name override. Empty/undefined falls back to app.defaultBin(). |
+| `model` | whisper.cpp model path (substituted for {model}); for chough this is the optional CHOUGH_MODEL env (chough auto-downloads a model when unset). |
+| `remoteUrl` | chough remote server URL (CHOUGH_URL). Empty/undefined = local transcription. |
+| `chunkSize` | chough chunk size in seconds (-c). Undefined = chough's own default. |
+| `customArgs` | whisper.cpp custom argv template using the {audioFile}/{outputBase}/{model} placeholders. Undefined = DEFAULT_TRANSCRIBE_ARGS. |
+| `device` | parakeet compute device passed to parakeet-cli (--device / PARAKEET_DEVICE), e.g. "cuda:0", "cpu". Undefined = parakeet-cli's default device. |
+
+#### `workers[].remote`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `baseUrl` | e.g. http://gpu-box.lan:3011 |
+| `token` | Outbound bearer token sent with every /api/worker request to this remote. The accepting side validates against its own WORKER_TOKEN env, never this. |
+| `sharedFs` | When true the remote shares the transcripts mount, so we send {channelSlug, videoId} instead of uploading the audio bytes. |
+| `slots` | How many units this remote takes in parallel. The pool expands one remote config into this many independently-schedulable slot entries at reconfigure time (the defaultWorkersFromApps trick, applied live). Absent = probed from the remote's /api/worker/health (its enabled worker count) — see controller/remoteCapacity.ts; 1 until the probe answers. |
+
+#### `workers[].llm`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `baseUrl` | e.g. http://macbook.lan:11434 |
+| `slots` | Concurrent generations to allow this endpoint. Defaults to 1 — one model instance, one generation — unless the operator knows better. |
+
+Default:
+
+```json
+[]
+```
+
+## `cookiesFromBrowser`
+
+Browser spec (e.g. "firefox", "chrome:Default") passed to `yt-dlp --cookies-from-browser`. WHEN it is passed is governed by `cookieMode` below. Empty string = no cookies configured. Per-channel override available (ChannelConfig.cookiesFromBrowser).
+
+Default: `""`
+
+## `cookieMode`
+
+How yt-dlp invocations use the configured cookies (see common/lib/cookiePolicy.ts): "always" passes them on every invocation, "when-required" (default; the historical behavior) only to retry an auth/age failure, "defer" never in normal runs — auth-gated videos are excluded from batches and collected into the per-channel "Needs cookies" bucket for a manual cookie run. Per-channel override available (ChannelConfig.cookieMode).
+
+Default: `"when-required"`
+
+## `sleepBetweenDownloadsSeconds`
+
+Pause (seconds) inserted between per-video yt-dlp invocations in managed batch downloads. yt-dlp's own `-t sleep` only paces requests within one invocation, so without this the managed loop hammers the source IP back-to-back. 0 disables. Per-channel override available.
+
+Default: `10`
+
+## `downloadFormat`
+
+Default yt-dlp `-f` download format for every channel that doesn't set its own (ChannelConfig.downloadFormat). "auto" picks per-source: `original` for Odysee (whose HLS rungs are CDN-truncated), `bestaudio/worst` elsewhere. See common/ytdlp/downloadFormat.ts.
+
+Default: `"auto"`
+
+## `minFreeDiskGB`
+
+Minimum free disk space (GB) required on the transcripts data directory for downloads to run. When free space is below this floor, a download job is prevented from starting and a running batch stops launching new videos (the in-flight one finishes). 0 disables the gate. See common/lib/diskSpace.ts.
+
+Default: `5`
+
+## `resumeMarginGB`
+
+Extra headroom (GB) above minFreeDiskGB that a stopped pipeline must see before it resumes. Resuming at the same number we stopped at flaps — the first restarted download pushes free space back under the floor. This is the hysteresis margin, so "resumed" means the operator actually freed something rather than a scratch file being cleaned up. 0 disables the hysteresis (resume at the floor). See diskGate() in common/lib/diskSpace.ts.
+
+Default: `2`
+
+## `parallelTranscriptions`
+
+Default number of videos transcribed in parallel when a "Transcribe missing" / bucket run doesn't specify its own concurrency. The per-run Concurrency input in the channel UI overrides this for a single run.
+
+Default: `2`
+
+## `inlineTranscribeOnFallback`
+
+When true, the no-subs fallback in the managed downloader runs whisper inline immediately after the audio download succeeds. When false (default), audio is left for the next "Transcribe missing" pass so a batch download finishes faster and whisper can parallelize.
+
+Default: `false`
+
+## `skipLiveDownloads`
+
+When true (default), managed downloads skip videos that are currently live or scheduled/upcoming, decided from a metadata-only prefetch pass. Finished livestream VODs (was_live) are NOT skipped and download normally. A skip is recorded but not archived, so the next sync/download-missing retries the video once the stream ends. Per-channel override available (ChannelConfig.skipLiveDownloads).
+
+Default: `true`
+
+## `verifyAvailabilityBeforeClean`
+
+Whether the transcribed-audio cleanup sweep checks each candidate is still available upstream before deleting its audio, pinning (do-not-clean) any video found permanently gone. The delete is irreversible and a gone video's audio is irreplaceable, so this defaults to true. Turn it off for an offline or URL-less setup, where the check can never resolve and cleanup would otherwise never delete anything. See verifyBeforeClean.ts.
+
+Default: `true`
+
+## `buildArchives`
+
+Whether site builds generate downloadable transcript/live-chat archive zips (into public/archives, linked on the Downloads page). Global default; a site can opt out via site.json `archives: false`, and a single build can skip via the "Skip archive zips" build control. Opt-out: default true.
+
+Default: `true`
+
+## `archiveStorage`
+
+Overflow object storage (Cloudflare R2) for archive zips that exceed the Pages per-file size cap (see Site.archiveMaxBytes). When both fields are set, an oversize archive is uploaded here on deploy — via `wrangler r2 object put`, keyed `<siteId>/archives/<file>` — instead of being dropped, and the Downloads page links to `<publicBaseUrl>/<key>`. Blank/absent → no overflow, so oversize archives stay unavailable ("Too large to host").
+
+#### `archiveStorage`
+
+| Key | Default | Description |
+|---|---|---|
+| `bucket` | `""` | Cloudflare R2 bucket an oversize archive zip is uploaded to on deploy (`wrangler r2 object put`, keyed `<siteId>/archives/<file>`). Blank = no overflow. |
+| `publicBaseUrl` | `""` | Public base URL of that bucket; the Downloads page links `<publicBaseUrl>/<key>`. Both fields must be set for overflow to happen. |
+
+Default:
+
+```json
+{
+ "bucket": "",
+ "publicBaseUrl": ""
+}
+```
+
+## `reportDebouncePreset`
+
+Debounce preset for the global snapshot scheduler: how long it waits after the last report-changing action before regenerating affected channel reports. See REPORT_DEBOUNCE_PRESETS. Default "fast" (~1s, no cap).
+
+Default: `"fast"`
+
+## `autoRefreshIntervalSeconds`
+
+How often (seconds) the editor UI passively re-fetches the current page's server-rendered data via router.refresh(), so sidebar badges and reports stay live without a manual reload. Mounted globally; pauses while the tab is hidden. 0 disables passive refresh entirely. See AUTO_REFRESH_INTERVAL_*.
+
+Default: `5`
+
+## `syncScheduler`
+
+Global configuration for the scheduled (cron-driven) channel sync system. The per-channel cadence lives on ChannelConfig.syncIntervalMinutes; this block holds the defaults and guard rails the scheduler applies across all channels. See common/jobs/syncScheduler.ts.
+
+#### `syncScheduler`
+
+| Key | Default | Description |
+|---|---|---|
+| `enabled` | `false` | Master switch. When false, a tick selects nothing (manual sync still works). |
+| `defaultIntervalMinutes` | `1440` | Fallback cadence (minutes) for channels with no per-channel override. |
+| `maxConcurrentSyncs` | `2` | Cap on sync jobs running/queued at once. A tick queues at most (cap - currently-active) channels; the rest roll to the next tick. This is also the stagger mechanism that keeps a big due-batch from hitting the source all at once. |
+| `quietHoursStart` | `null` | Optional local-clock quiet window during which auto-sync is suppressed. Both null = always allowed. The window may wrap past midnight (e.g. start=22, end=6). Hours are [0,23]; the window is [start, end). |
+| `quietHoursEnd` | `null` | End hour of the quiet window, [0,23], exclusive. See `quietHoursStart`: both must be valid hours or the window is cleared (null = always allowed). |
+| `backoffBaseMinutes` | `30` | Failure backoff bounds. After N consecutive failed scheduled syncs a channel waits min(base * 2^(N-1), max) minutes before it's eligible again. |
+| `backoffMaxMinutes` | `1440` | Ceiling on the failure backoff (see `backoffBaseMinutes`): a channel waits min(base * 2^(N-1), max) minutes after N consecutive failures. Never below the base. |
+| `heartbeatSeconds` | `0` | Cadence (seconds) for the editor's in-process heartbeat — the internal timer armed by the instrumentation hook (editor/instrumentation.ts) that calls the scheduler tick directly, so no external cron is needed. 0 = off: rely on the external `pnpm sync:tick` heartbeat instead. Any positive value is clamped to [SYNC_HEARTBEAT_MIN_SECONDS, SYNC_HEARTBEAT_MAX_SECONDS]. The env var SYNC_HEARTBEAT_SECONDS overrides this at runtime. See SCHEDULED_SYNC.md. |
+| `keepLatestCheckIntervalMinutes` | `1440` | Cadence (minutes) for the scheduled keep-latest deletion check. For each channel with ChannelConfig.keepLatest > 0, the tick re-probes the kept window for source deletion (checkKeptDeletedAction) at most this often and pins any gone videos as do-not-clean. Clamped into the sync-interval window; default daily. The check shares the same concurrency cap and quiet-hours window as scheduled syncs. See editor/app/scheduler/runTick.ts. |
+| `fullSweepIntervalMinutes` | `1440` | Default cadence (minutes) for the sync FULL SWEEP — the deep pass that re-enumerates a channel's whole listing in one yt-dlp spawn, refreshes the stored `playlist` file, and flags videos that have left the listing into maybe-missing.json. Ordinary syncs stay on the cheap newest-first paged walk; a sync only upgrades itself to a sweep when this interval has elapsed since the channel's lastFullSweepAt. Per-channel override: ChannelConfig.fullSweepIntervalMinutes. 0 = never sweep. Default daily. See common/jobs/deepSync.ts. |
+| `fullSweepConfirmMaxSuspects` | `25` | Upper bound on how many maybe-missing suspects a full sweep will resolve in-line with the per-video availability probe (deleted vs private vs unlisted). At or under the cap the sweep runs the targeted check itself, so "Sync all" surfaces upstream deletions with no extra clicks; over it, the suspects are flagged and left for a manual check rather than firing hundreds of probes inside a sync. 0 = never auto-confirm. |
+| `fullSweepShrinkGuardPercent` | `10` | Shrink guard: how far a fresh listing may fall below the stored one before it is treated as suspect rather than acted on. Expressed as a percentage of the previous count, floored at SHRINK_ABS_FLOOR entries so ordinary churn on a small channel doesn't trip it. A suspect listing does not rewrite `playlist` or maybe-missing.json and does not count as a sweep — but a SECOND enumeration reporting a similar count confirms it and is accepted, so a genuine mass deletion costs at most one cadence period. 0 = off (the empty-listing rejection still applies). See controller/acceptListing.ts. |
+
+Default:
+
+```json
+{
+ "enabled": false,
+ "defaultIntervalMinutes": 1440,
+ "maxConcurrentSyncs": 2,
+ "quietHoursStart": null,
+ "quietHoursEnd": null,
+ "backoffBaseMinutes": 30,
+ "backoffMaxMinutes": 1440,
+ "heartbeatSeconds": 0,
+ "keepLatestCheckIntervalMinutes": 1440,
+ "fullSweepIntervalMinutes": 1440,
+ "fullSweepConfirmMaxSuspects": 25,
+ "fullSweepShrinkGuardPercent": 10
+}
+```
+
+## `autoQueue`
+
+Configuration for the automatic priority-queue runners (auto-transcribe / auto-download). Each holds a tree policy that decides which channel's video to process next, cross-channel, by priority/round-robin/weighted-fair rules. Independent of syncScheduler (which decides staleness, not work order). See common/jobs/autoQueuePolicy.ts.
+
+#### `autoQueue.<lane>`
+
+| Key | Default | Description |
+|---|---|---|
+| `enabled` | `false` | Master switch for this runner (transcription / download independently). |
+| `maxWorkers` | `null` | Overall ceiling on concurrent in-flight workers for this runner. null = no runner-level cap (the worker pool / platform queues are the real throttle). |
+| `replaceAutoSubs` | `false` | Opt in to the lowest-priority "replace YouTube auto-captions" lane: append this kind's opt-in buckets (autoSubsOnly / downloadedAutoSubsOnly) to the tail of the default union, so videos whose only transcript is YouTube ASR get re-done with our own engine whenever nothing more important is pending. Default false — the corpus-wide cost is large (an audio download plus a transcription per video). A leaf can also target the bucket by name for per-channel opt-in without flipping this switch. Optional: settings written before this field existed lack it; the sanitizer defaults it to false. |
+| `order` | transcription `"listed"`<br>download `"listed"`<br>digest `"cheapest"`<br>backfill `"listed"` | Ordering within each rule (see AutoQueueOrder). Optional exactly like replaceAutoSubs: settings files written before this field existed lack it, and the sanitizer defaults them to "listed" (today's behaviour). |
+| `snoozeUntil` | `null` | Epoch ms until which this runner idles WITHOUT stopping: next() returns null so the loop stays up, re-reads settings each iteration, and resumes by itself when the moment passes. null/absent/past = not snoozed. Survives a restart because it lives in settings.json, not in runner memory. |
+| `held` | transcription `false`<br>download `false`<br>digest `false`<br>backfill `true` | THE LANE'S PAUSE GATE. Shut means the lane holds: every dispatch path asks lib/pauseGates.ts, whose limit()/guard returns 0 so runPool idle-waits. A hold, never a stop — see that file's header.<br><br>OPTIONAL IN THE TYPE, FILLED BY THE SANITIZER. Until slice 1.4 four separate settings fields carried this — `transcriptionsPaused`, `downloadsPaused`, `digest.digestsPaused` and (inverted) `backfill.enabled` — so `undefined` meant "ask the legacy field" and `sanitizePolicy` deliberately refused to default it: a default would have read a paused corpus as running. S0-pause deleted those four, on the precondition that the live settings.json already carried every `held` key, and the default came in with them (`defaultHeldFor` — free everywhere except backfill, whose field was inverted and shipped held).<br><br>It stays optional because a reader may be handed a PARTIAL settings object (laneGuards.test.ts casts one), and `isGateHeld` answers `false` for a lane that carries no key at all rather than throwing. |
+| `root` | object — see below | The lane's rule tree: a group whose children are groups and leaves (see the node table). A missing root is the lane's default — empty for the runner lanes, one catch-all leaf for digest and backfill. While a channel-priority document exists, the four roots are compiled from it and not hand-edited. |
+
+#### `autoQueue.<lane>.root (tree nodes)`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `id` | Stable node id, unique within the lane's tree. Preserved on save when valid and not taken, so persisted fairness state survives an unrelated edit; a missing or duplicate id is replaced with a generated one. |
+| `match` | LEAF ONLY: which videos this leaf owns (see the match table). |
+| `weight` | Relative share under a weighted-fair parent. Default 1. Ignored otherwise. |
+| `maxWorkers` | Optional ceiling on concurrent in-flight workers drawn from this node (and, for a group, its whole subtree). A capped node reads as "no work" and the parent falls through to the next sibling, like an HTB class ceiling. null = no cap. |
+| `mode` | GROUP ONLY: how the children compete — "strict" (first child with work wins), "round-robin", or "weighted-fair" (by each child's `weight`). Unknown values read as "strict". |
+| `children` | GROUP ONLY: the child nodes, in priority order for a strict group. A node with a `children` array is a group; any other node is a leaf. |
+
+#### `autoQueue.<lane>.root … .match`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `type` | What the leaf matches: "channel" (one channel slug in `value`), "platform" (a platform name in `value`), or "all". |
+| `value` | Channel slug (type=channel) or platform name (type=platform). Ignored for type=all. A type=channel leaf with no value matches nothing. |
+| `bucket` | Optional snapshot bucket this leaf draws from, narrowing the default for the runner kind (transcription → downloadedNoTranscript, download → undownloadedIds). E.g. bucket="failedListed" prioritizes retries. |
+| `operation` | Optional OPERATION this leaf draws from — a registered backfill kind id, or "digest". Same meaning as `bucket` one level up: it narrows what the leaf claims, and it draws from ChannelWork.operations rather than ChannelWork.buckets.<br><br>It lives on the MATCH, beside `bucket`, and not on the node. A field on the node would need group inheritance — "this group is the digest subtree" — and inheritance is resolution logic buildPendingByLeaf does not have. Here it needs exactly one sanitizer and exactly one claiming path.<br><br>It is a SEPARATE id space from `bucket`, and the sanitizer enforces that a leaf names at most one of the two (operation wins): `defaultBuckets` is a priority-ordered union, so a name that meant a bucket to one leaf and an operation to another would silently mix two id spaces, and selectableBucketsForKind feeds the editor's bucket dropdown, where an operation must not appear as a bucket. |
+
+Default:
+
+```json
+{
+ "transcription": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "listed",
+ "snoozeUntil": null,
+ "held": false,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": []
+ }
+ },
+ "download": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "listed",
+ "snoozeUntil": null,
+ "held": false,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": []
+ }
+ },
+ "digest": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "cheapest",
+ "snoozeUntil": null,
+ "held": false,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": [
+ {
+ "id": "all",
+ "match": {
+ "type": "all"
+ },
+ "weight": 1,
+ "maxWorkers": null
+ }
+ ]
+ }
+ },
+ "backfill": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "listed",
+ "snoozeUntil": null,
+ "held": true,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": [
+ {
+ "id": "all",
+ "match": {
+ "type": "all"
+ },
+ "weight": 1,
+ "maxWorkers": null
+ }
+ ]
+ }
+ }
+}
+```
+
+## `channelPriority`
+
+THE OPERATOR-FACING PRIORITY MODEL: one tier per channel plus one corpus-wide focus selector. It is the SOURCE the four `autoQueue[lane].root` trees are compiled from (common/lib/channelPriority.ts), not a second mechanism beside them — and its `paused` tier is the one part that is not a tree shape, filtering the runner's channel list instead. An empty document (the default) is today's behaviour exactly: no focus, every channel normal, the stored trees stand.
+
+#### `channelPriority`
+
+| Key | Default | Description |
+|---|---|---|
+| `focus` | object — see below | The corpus-wide focus selector: none, one site's channels, or a list of channels. A focus is compiled into a leading `prio-focus` group in every lane's tree. |
+| `channels` | `{}` | ONLY channels that differ from the default appear. An absent slug is `normal`, unranked — so the default document is empty and "absent document = today's behaviour" holds byte for byte. |
+
+#### `channelPriority.focus`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `kind` | "none" (no focus), "site" (the channels of one site, resolved at compile time so it tracks membership) or "channels" (an explicit list, from "Focus these"). |
+| `siteId` | kind "site" only: the site whose channels are focused. A blank id reads as no focus; an unknown one survives and focuses nothing. |
+| `slugs` | kind "channels" only: the focused channel slugs, trimmed and de-duplicated. An empty list reads as no focus. |
+
+#### `channelPriority.channels.<slug>`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `tier` | THE BASE TIER: what every operation gets unless an override says otherwise. |
+| `rank` | Order WITHIN the tier, ascending. Absent = unranked, which sorts after every ranked sibling and then by slug. ONE rank per channel, not one per lane — the two hand-made lane orders collapse into this on migration. |
+| `overrides` | PER-OPERATION OVERRIDES of the base tier. Only operations that DIFFER from the base appear: the sanitizer normalises an override equal to `tier` away, so the on-disk document stays a list of exceptions to a list of exceptions.<br><br>`{tier:"normal", overrides:{sync:"paused"}}` is "everything but sync" — the lossless reading of the retired `excludeFromSync`. Its inverse, `{tier:"paused", overrides:{sync:"normal"}}`, is "sync only": keep the playlist and metadata current, dispatch nothing. |
+| `autoPaused` | PAUSED BY THE MACHINE, NOT BY THE OPERATOR, and what to put back.<br><br>Set when the drive a channel's media is on stops being there: the watch pass records the tier the channel HAD and forces `paused`, so nothing in any lane dispatches against a `data/` nobody can read. Cleared — and the tier restored — when the drive comes back.<br><br>WHY IT IS A FIELD AND NOT A DERIVED STATE. The lanes read `tier`; making them all ask a second question would be four more places to forget. And the tier the channel is to be RESTORED to is not derivable from anything once it has been overwritten — that is the whole content of this field.<br><br>OPTIONAL, and an older binary that drops it leaves the channel Paused with nothing lost but the automatic restore. The operator's own word always wins: a MANUAL tier change clears it (see clearAutoPause), so a drive coming back can never un-pause a channel somebody paused on purpose. |
+
+#### `channelPriority.channels.<slug>.autoPaused`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `reason` | One reason today. A union so a second one has somewhere to go, and so a surface can say WHICH machine decided rather than "automatic". |
+| `since` | ISO, for "auto-paused — media unreachable since <date>". |
+| `previousTier` | The base tier the channel had before the machine paused it; what a restore puts back. Never `paused` (that would restore to paused — a no-op dressed as a restore). |
+
+Default:
+
+```json
+{
+ "focus": {
+ "kind": "none"
+ },
+ "channels": {}
+}
+```
+
+## `socialLinks`
+
+Default social links applied to every site that doesn't define its own. A site inherits these unless its site.json carries an explicit `socialLinks` array — see Site.socialLinks / resolveSocialLinks in common/lib/site.ts. The one presentation field that lives globally so a shared footer doesn't have to be repeated per site.
+
+#### `socialLinks[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `label` | Visible name, also the accessible label of the icon. |
+| `url` | Link target: http(s), mailto: or a site-relative path. |
+| `svg` | Inline SVG markup. Normalized on save (width/height stripped, fill="currentColor", aria-hidden) and rejected when unsafe (script, foreignObject, event handlers, javascript: URLs) or when it has no viewBox. |
+
+Default:
+
+```json
+[]
+```
+
+## `homepageUrl`
+
+Absolute public URL of the family hub/homepage (e.g. "https://archilyzer.pages.dev"). Every export site links back to it ("the family" backlink) when set. Empty = no hub link rendered. Normalized to a trailing-slash-free http(s) URL.
+
+Default: `""`
+
+## `savedVideoBackup`
+
+Backup configuration for the saved-video store (Phase 4 of the video-persistence feature). When enabled with a destination, the store is mirrored there (additively, no deletes) with a per-backup manifest, and the sync scheduler runs the backup on the configured cadence. See common/controller/backupSavedVideos.ts.
+
+#### `savedVideoBackup`
+
+| Key | Default | Description |
+|---|---|---|
+| `enabled` | `false` | Master switch for the scheduled backup. A backup can still be run manually when this is false, as long as a destination is set. |
+| `dest` | `""` | Destination root the store is mirrored into (a local path or any rsync target). Empty disables both scheduled and manual backups. |
+| `intervalMinutes` | `1440` | Cadence (minutes) for the scheduled backup when enabled. Clamped into the sync-interval window; default daily. |
+
+Default:
+
+```json
+{
+ "enabled": false,
+ "dest": "",
+ "intervalMinutes": 1440
+}
+```
+
+## `storage`
+
+Where a channel's downloaded media goes when it is relocated off the corpus disk. A DEFAULT ONLY: the relocate controller never reads it and always takes an explicit root, so this is the value the per-channel Storage panel prefills and the /channels bulk move falls back to. Blank = no default. See StorageSettings.
+
+#### `storage`
+
+| Key | Default | Description |
+|---|---|---|
+| `locations` | `[]` | The named storage locations a channel's media may be relocated to — one entry per root, each with an id, label, root, `autoRepoint` and the learned volume identity. Order is display order. Managed on /storage. |
+| `defaultLocationId` | `""` | The location prefilled as the destination of a move. "" = no default. |
+| `savedVideosLocationId` | absent | WHERE THE SAVED-VIDEO STORE IS, by location id. "" = in place, under the corpus at `paths.savedVideosDir`.<br><br>A RECORD OF WHAT IS ON DISK, never an intention — the same contract as a channel's `config.dataDir`. It is written by the move, on success, after the copy has verified and the symlink is in place; nothing else writes it, and a reader that disagrees with the disk trusts the disk. Optional so an older settings.json parses (and an older binary that drops it leaves a store that still works, because the symlink is what every reader follows). |
+
+#### `storage.locations[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `id` | /^[a-z0-9][a-z0-9-]{0,63}$/, unique within the list. Stable: it is what `defaultLocationId` and every form and action refer to. |
+| `label` | Human name. Blank sanitizes to the id. |
+| `root` | Absolute directory, trailing "/" stripped. NEVER existence-checked on read — the whole point of a cold location is a drive that may not be mounted when settings are parsed. |
+| `autoRepoint` | Opt-in: when the volume is found mounted somewhere else, re-point without asking (if the preflight passes). Off by default — re-point rewrites every channel symlink on the location, and that is not something to do silently unless the operator asked for it. |
+| `volume` | Identity learned at the last successful probe. Optional because a location may never have been probed, and because in a container block devices are invisible and identity is permanently unknown. |
+
+#### `storage.locations[].volume`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `uuid` | Filesystem UUID, the one stable name a disk has across mountpoints. This is what makes "the platter came up somewhere else" a recoverable situation. |
+| `fstype` | Filesystem type reported by the probe (e.g. "ext4"). Informational; omitted when unknown. |
+| `label` | Filesystem label reported by the probe. Informational; omitted when unknown. |
+| `mountpoint` | Where the volume was mounted at the last successful probe, and the path of the location's root RELATIVE to that mountpoint. Invariant: `root === join(mountpoint, relPath)`. Keeping the two halves is what lets a probe compute a candidate root when the volume reappears elsewhere. |
+| `relPath` | The location root's path RELATIVE to `mountpoint` (see there). Invariant: `root === join(mountpoint, relPath)`. |
+
+Default:
+
+```json
+{
+ "locations": [],
+ "defaultLocationId": ""
+}
+```
+
+## `buildPipeline`
+
+How the static export is built: "basic" reuses the single export/ tree and serializes builds on one queue (the long-standing behavior); "docker" runs each site's build in an isolated container for safe parallelism. The Docker pipeline itself is a follow-up; this block persists the chosen mode plus the container/concurrency knobs the deploy page and the future orchestrator read.
+
+#### `buildPipeline`
+
+| Key | Default | Description |
+|---|---|---|
+| `mode` | `"basic"` | "basic" — `pnpm run build` in export/, serialized on the build queue (shared output tree, no safe parallelism). "docker" — isolated per-site container builds, parallel up to `maxParallelBuilds`. |
+| `maxParallelBuilds` | `2` | Cap on concurrent per-site container builds in docker mode. Ignored in basic mode (which is always serial). Clamped to [1, BUILD_MAX_PARALLEL_MAX]. |
+| `dockerImage` | `"yt-dlp-transcript-browser-build"` | Tag of the reusable build image (built once, reused for every site). |
+| `dockerfile` | `"Dockerfile.build"` | Dockerfile path relative to the monorepo root, used to (re)build the image. |
+
+Default:
+
+```json
+{
+ "mode": "basic",
+ "maxParallelBuilds": 2,
+ "dockerImage": "yt-dlp-transcript-browser-build",
+ "dockerfile": "Dockerfile.build"
+}
+```
+
+## `digest`
+
+AI digest generation (chapters + topic tags over the existing transcripts). Local-first: the metered lane is off by default. See DigestSettings.
+
+#### `digest`
+
+| Key | Default | Description |
+|---|---|---|
+| `remoteEnabled` | `false` | Master switch for the metered (remote-api) lane. OFF by default — an opt-in overflow for the long tail or a channel where local quality is poor, never the default path. |
+| `longTailSeconds` | `14400` | Videos longer than this are "long tail": 8.2% of the corpus by count, 46% of all transcript tokens. The batch's duration-aware ordering and the optional remote overflow both key off it. |
+| `localAppId` | `"ollama-direct"` | The engine each lane uses (ids from common/lib/digestApps.ts). |
+| `remoteAppId` | `"claude-code"` | The engine the metered (remote-api) lane uses — an id from common/lib/digestApps.ts. Unknown ids degrade to the default app rather than failing. |
+| `apps` | `{}` | Per-app config, keyed by app id — the same id-keyed sub-record shape as transcriptionApps. |
+| `yieldToTranscription` | `true` | Yield the GPU to the transcription lane: while transcription is working, the digest batch's limit() returns 0 and the pool idle-waits. ON by default, because `digest:local` is deliberately on a different queue from TRANSCRIPTION_QUEUE and so would otherwise run ollama and the transcription engine on the same 8 GB card. See controller/digestYield.ts. |
+| `yieldToCpuWorkers` | `false` | Whether a busy worker pinned to `device: "cpu"` counts as GPU contention.<br><br>OFF by default, which is the FIX for a real bug: the yield originally tested only `kind === "local"`, so on a box with one GPU worker and two CPU-pinned ones (this box, at parallelTranscriptions 2) the digest lane stopped dead for transcription that competes for zero GPU shaders.<br><br>Only an EXPLICIT "cpu" is treated as non-contending. A worker with no device set is using the engine binary's own default, which may be the GPU, so it still triggers the yield — the unknown case fails safe.<br><br>Composes with `yieldToTranscription`: that is the master switch, this only narrows which workers it reacts to. |
+| `spendCapUsd` | `0` | Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. Only ever consulted for a metered app. |
+| `sections` | list — see below | Which sections a sweep generates.<br><br>Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, and that is not a contradiction: a tag call sends the same transcript as the chapter call before it, so it hits the engine's cached prefix and pays essentially no prefill (+0.4 s across 4 extra calls, against 22.4 s for the first 4). All it pays is decode, and a tag list is ~30 output tokens where a chapter list is ~200–290.<br><br>The corollary matters more than the number: run them in the SAME pass. Tags generated later, on their own, pay full prefill again — measured at 44% of a whole chapters pass, i.e. 3–9× the marginal cost of just including them now. |
+| `timestampMode` | `"chunk-local"` | How each chunk's transcript markers are numbered — see DigestTimestampMode. Was a scored variable in the bake-off rather than a pre-applied fix; the measurement is in and "chunk-local" is now the shipped default. |
+| `promptVariant` | `""` | A free-text label for a non-default prompt shape, folded into the recorded provenance by digestPromptVariant(). Setting it invalidates every digest generated under a different label, which is exactly what makes a bake-off round re-run its sample instead of skipping it as fresh. Empty = default. |
+
+#### `digest.apps.<appId>`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `bin` | Binary path/name override (process-based apps only). |
+| `baseUrl` | Base URL override (HTTP apps only). |
+| `model` | Model id, e.g. "qwen2.5:7b" or "haiku". |
+| `numCtx` | Context window in tokens. MUST reach the engine explicitly for ollama: its 4096 default silently truncates the input and the model then summarizes whatever fragment survived — measured, and the single easiest way to get quietly-wrong output at scale. |
+| `temperature` | Sampling temperature. 0 for a structured extraction task. |
+| `think` | Reasoning-model toggle (ollama's top-level `think`). Only sent when set, so a model that does not support thinking is never handed a field it rejects.<br><br>It matters for throughput, not correctness: measured on this box, qwen3:8b with thinking on spends most of its output budget on a `thinking` block before the JSON body the schema constrains. For an extraction task with a pinned schema that reasoning buys little and costs a multiple of the tokens, and tokens are what a multi-week sweep is priced in. |
+| `timeoutMs` | Per-request wall-clock ceiling (ms). A wedged engine must not stall a sweep. |
+
+Default:
+
+```json
+{
+ "remoteEnabled": false,
+ "longTailSeconds": 14400,
+ "localAppId": "ollama-direct",
+ "remoteAppId": "claude-code",
+ "apps": {},
+ "yieldToTranscription": true,
+ "yieldToCpuWorkers": false,
+ "spendCapUsd": 0,
+ "sections": [
+ "chapters"
+ ],
+ "timestampMode": "chunk-local",
+ "promptVariant": ""
+}
+```
+
+## `diarization`
+
+Speaker diarization captured right after transcription, while the audio is still on disk. OFF by default. See DiarizationSettings.
+
+#### `diarization`
+
+| Key | Default | Description |
+|---|---|---|
+| `enabled` | `false` | Master switch. OFF by default so a transcription batch can start before this lands, with diarization backfilled over the retained audio afterwards.<br><br>Turning it ON also arms the cleanup guard: the Clean-audio sweep stops deleting audio for a transcribed video that has no diarization.json yet. That is the point — it is what keeps the perishable input alive long enough to be captured — but it means enabling this holds disk. |
+| `inlineAfterTranscribe` | `false` | Run diarization inline in the post-transcribe hook.<br><br>OFF by default, and that default is a MEASURED decision, not caution. Measured on this box: GPU transcription runs at 221 s/audio-hour (16.3x realtime, over 3,602 real videos), CPU diarization at ~500-680 s/audio-hour. Diarization is therefore ~2-3x SLOWER than the transcription it follows, so running it inline drops whole-pipeline throughput by roughly 3-4x and leaves the GPU idle while the CPU catches up.<br><br>The intended sequence for a large batch is the opposite: leave this off, let the batch transcribe at full GPU speed with `enabled` holding the audio, and diarize afterwards with the backfill pass. Turn it on for steady state, once the arrival rate is a few videos a day rather than a corpus. |
+| `threshold` | `0.9` | Clustering threshold — the single most consequential knob, since it decides how many speakers come out. Larger merges more aggressively.<br><br>The default is 0.9, NOT sherpa-onnx's own 0.5, and that is measured on this corpus. On a 6-minute excerpt of a two-person interview (known ground truth: 2 speakers), sherpa's default produced 22 clusters; 0.9 produced 6, with the top two at 40%/40% of talk time — recognizably the two hosts. Sweep on the same clip: 0.4→23, 0.5→22, 0.6→17, 0.7→12, 0.8→10, 0.9→6.<br><br>It still over-splits, which is why this is a capture lane and not an answer: the turns are recorded with the threshold that produced them, so a later attribution pass can re-cluster or re-run without needing the audio back. |
+| `threads` | `4` | Engine threads per diarize run. |
+| `engine` | `"sherpa-onnx"` | Which engine runs. "sherpa-onnx" is the shipped default and what every sidecar on disk was produced by; "sortformer" is the ggml engine built by scripts/build-sortformer.sh.<br><br>CHANGING THIS RESTATES THE FRESHNESS IDENTITY (see diarizationTarget), so every sidecar written by the other engine becomes stale and the backfill lane offers to redo it. That is intended — the two disagree about how many speakers exist, and a corpus half-diarized by each is not one corpus — but on the retained audio it is weeks of work, not a toggle.<br><br>Why anyone would: on the same file, sherpa at its tuned threshold returns 13 speakers and sortformer returns 4, agreeing on the dominant speaker's share to within half a point (73.1% vs 73.5%). On the corpus's worst case sherpa returns 35 and sortformer 4. Over-splitting is the failure mode this lane has always had, and sortformer is end-to-end rather than clustered, so it does not have it. The cost is a hard ceiling of 4 speakers and ~1.8x the wall clock. |
+| `backend` | `"vulkan"` | Compute device for the sortformer engine; ignored by sherpa-onnx, which has no Vulkan compute path on Linux.<br><br>"vulkan" is 1.5x faster than a thread-tuned CPU run (894 vs 1305 s/audio-hour, measured on this box) and holds 558 MB resident instead of 4.84 GB by keeping weights and activations in VRAM. It also takes ~4.4 GB of an 8 GB card, which is why the lane YIELDS to transcription rather than sharing — see controller/digestYield.ts. |
+| `python` | `"python3"` | Python interpreter for the default sherpa-onnx engine. sherpa-onnx ships wheels only up to cp313, and this box's system python is 3.14 — so this usually points at a dedicated venv rather than `python3`. |
+| `segModel` | `""` | ONNX model paths for the default engine. Empty = the lane cannot run, which is reported as a skip rather than a failure. |
+| `embModel` | `""` | ONNX speaker-embedding model path for the sherpa-onnx engine. Empty = the lane cannot run, reported as a `not-configured` skip rather than a failure (same as `segModel`). |
+| `sortformerBin` | `""` | Binary and model for the sortformer engine, both produced by scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the same "not-configured" skip as an unset segModel/embModel. |
+| `sortformerModel` | `""` | Model for the sortformer engine, produced by scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the same `not-configured` skip as an unset `sortformerBin`. |
+| `concurrency` | `1` | How many diarize runs may execute at once in the backfill pass. Kept low by default: diarization is CPU-bound and competes with GPU feeding and the digest sweep for the same 8 threads. |
+| `maxAudioHours` | `0` | Videos longer than this are DEFERRED rather than diarized: reported as a third number that is never summed into reachable work, so a capped corpus can never read as finished.<br><br>THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering holds a pairwise distance matrix over speech-segment embeddings — O(n^2) in SEGMENT count — and speaker-turn density varies 40x across this corpus (33-1364 turns/hour), so duration does not actually predict the blowup: a sparse 7h42m video completed while a dense 6h12m one was OOM-killed. Duration is merely the only predictor available for free, from metadata already on disk, BEFORE spending 45 minutes to find out. n^2 at 30k segments is 6.7 GiB and at 40k is 11.9 GiB, which brackets the 10.6 GB and 9.6 GB peaks measured on this 16 GB box.<br><br>0 disables the cap. That is where this goes once windowed diarization lands: windowing divides per-window n by the window count, so the matrix falls by its square, and the cap stops being needed rather than being tuned. |
+
+Default:
+
+```json
+{
+ "enabled": false,
+ "inlineAfterTranscribe": false,
+ "threshold": 0.9,
+ "threads": 4,
+ "engine": "sherpa-onnx",
+ "backend": "vulkan",
+ "python": "python3",
+ "segModel": "",
+ "embModel": "",
+ "sortformerBin": "",
+ "sortformerModel": "",
+ "concurrency": 1,
+ "maxAudioHours": 0
+}
+```
+
+## `backfill`
+
+The generic catch-up lane for derived data the existing corpus predates. OFF by default, and idle-only when on. See BackfillSettings.
+
+#### `backfill`
+
+| Key | Default | Description |
+|---|---|---|
+| `concurrency` | `1` | Slots the lane may use when it is not standing aside. Kept at 1 by default for the same reason diarization.concurrency is: this is CPU-bound work competing with GPU feeding and the digest sweep for the same 8 threads. |
+| `allowRedownload` | `false` | Re-acquire media for videos whose input is GONE (audio deleted after transcription). OFF by default and deliberately so: measured on this corpus, 836 videos still have media and ~76,270 would need a re-download — 91x the reachable work, against 45 GB free at 97% full. When on, each re-fetched file is removed in a `finally` as soon as the backfill has used it, unless the video is marked do-not-clean, or unless the auto-transcribe policy would replace its auto-captions (`replaceAutoSubs`, or a leaf on `downloadedAutoSubsOnly`), in which case the audio is kept for that runner.<br><br>WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: "youtube"` channel — which normally only fetches subtitles — the re-acquire applies a PER-VIDEO transcribe override so yt-dlp lands audio a diarizer can read; the channel's stored config is not changed. Without that override the fetch re-downloads the captions the video already has and lands nothing, which is what happened to ~16,000 videos on eight channels in 2026-08. |
+
+Default:
+
+```json
+{
+ "concurrency": 1,
+ "allowRedownload": false
+}
+```
+
+## `attribution`
+
+Naming the speakers diarization found (or reconstructing them from the transcript when it found none). OFF by default. See AttributionSettings.
+
+#### `attribution`
+
+| Key | Default | Description |
+|---|---|---|
+| `enabled` | `false` | Master switch. Off means the backfill registry reports no attribution work at all — the feature gate every Operation has. |
+| `appId` | `"ollama-direct"` | Which digest app runs the naming. Attribution IS a digest-app workload — constrained JSON decoding over transcript text — so it reuses that registry and that per-app config (settings.digest.apps[appId]) rather than growing a second copy of the ollama URL, context size and timeout. |
+| `model` | `""` | Model override. Empty = the app's configured model, then its default. It is separate from the digest's because the two workloads may want different sizes, and because it is part of the freshness identity: sharing the digest's model field would make a digest bake-off invalidate every attribution record on disk as a side effect. |
+| `diarizedEnabled` | `false` | The lanes, separately. Both default OFF even when `enabled` is on, so turning the feature on to look at it cannot start a corpus sweep.<br><br>They are not a fallback pair. `diarized` is one call per video and grounded in acoustic clustering; `textOnly` is ~30 calls and guesses at identity across chunk seams. An operator may reasonably want the first forever and the second never. |
+| `textOnlyEnabled` | `false` | The text-only attribution lane: names speakers from the transcript alone (~30 model calls per video, guessing identity across chunk seams). Default OFF even when `enabled` is on. See `diarizedEnabled` — the two are separate lanes, not a fallback pair. |
+| `promptVersion` | `1` | The prompt generation a record must match to count as fresh.<br><br>Defaults to (and is floored at) ATTRIBUTION_PROMPT_VERSION, the shipped constant. Raising it forces a corpus-wide regeneration without a code change, which is the honest way to redo everything after a prompt tweak. It cannot be set BELOW the shipped constant, and that floor is the lesson from digestPrompt.ts's version 1 -> 2 note: pinning freshness to an older generation freezes output from a superseded prompt into the corpus, looking identical to output from the current one. |
+
+Default:
+
+```json
+{
+ "enabled": false,
+ "appId": "ollama-direct",
+ "model": "",
+ "diarizedEnabled": false,
+ "textOnlyEnabled": false,
+ "promptVersion": 1
+}
+```
diff --git a/SETUP.md b/SETUP.md
@@ -252,8 +252,10 @@ channels.
## Configuration & environment variables
Most configuration now lives in the editor's **/settings** page, persisted to
-`settings.json` at the repo root (gitignored). `settings.json.example` is a minimal
-starting template. Settings are optional — a missing/partial `settings.json` falls
+`settings.json` at the repo root (gitignored). Every key, its default and what it
+does is in [SETTINGS.md](SETTINGS.md); `settings.json.example` is the defaults as a
+starting template. Both are generated from the settings schema
+(`common/lib/settingsSchema.ts`). Settings are optional — a missing/partial `settings.json` falls
back to built-in defaults, so the app runs out of the box.
Paths and binaries resolve through `getPaths()` in `common/lib/paths.ts`. Override
diff --git a/common/architecture.test.ts b/common/architecture.test.ts
@@ -62,13 +62,6 @@ const ROOTS = [
// Today's back-edges, `<file> -> <imported module>`, each with why it is still
// here. THIS LIST IS A DEBT LEDGER, not a policy.
const ALLOWED: Record<string, string> = {
- // The four auto-queue SANITIZERS are the auto-queue's half of the settings
- // schema. They belong in lib/ with the rest of it; that move is phase 3 slice
- // 4 (one schema library, one writer), not a rename. The tree/settings TYPES
- // already moved (lib/autoQueueTypes.ts).
- "lib/settings.ts -> jobs/autoQueuePolicy":
- "defaultAutoQueue + three sanitizers; moves with the settings schema in phase 3",
-
// Three progress-parser factories for the transcription apps. Parsing a
// subprocess's stdout is dispatch's job, not the model's; the fix is for the
// app descriptor to name a parser the runner resolves, which is phase 1 work.
diff --git a/common/bin/settings-example.ts b/common/bin/settings-example.ts
@@ -0,0 +1,58 @@
+#!/usr/bin/env tsx
+// WRITE settings.json.example AND SETTINGS.md FROM THE SETTINGS SCHEMA.
+//
+// Usage (from the repo root):
+// pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts
+// pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts --check
+//
+// `--check` writes nothing and exits 1 if either committed file differs from
+// what the schema generates (the same claim common/lib/settingsDocs.test.ts
+// makes). Becomes `archilyzer settings example` in one-core phase 4.
+//
+// Reads no settings.json and writes no settings.json: both outputs are
+// functions of the schema alone.
+
+import { readFile, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import {
+ renderSettingsExample,
+ renderSettingsMarkdown,
+} from "../lib/settingsDocs";
+import { parseFlags } from "./_parseFlags";
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+const OUTPUTS: ReadonlyArray<[string, () => string]> = [
+ ["settings.json.example", renderSettingsExample],
+ ["SETTINGS.md", renderSettingsMarkdown],
+];
+
+async function main(): Promise<number> {
+ const flags = parseFlags(process.argv.slice(2));
+ const check = flags.check === "true";
+ let stale = 0;
+ for (const [name, render] of OUTPUTS) {
+ const file = path.join(REPO, name);
+ const want = render();
+ if (check) {
+ const have = await readFile(file, "utf8").catch(() => "");
+ if (have !== want) {
+ console.error(`${name} is stale — regenerate it`);
+ stale++;
+ }
+ continue;
+ }
+ await writeFile(file, want);
+ console.log(`wrote ${name}`);
+ }
+ return stale > 0 ? 1 : 0;
+}
+
+main().then(
+ (code) => process.exit(code),
+ (err) => {
+ console.error(err);
+ process.exit(1);
+ },
+);
diff --git a/common/jobs/autoQueuePolicy.test.ts b/common/jobs/autoQueuePolicy.test.ts
@@ -9,7 +9,6 @@ import {
bucketIdsFrom,
bucketLaneWorkIds,
bucketsForKind,
- defaultAutoQueue,
defaultBucketsForPolicy,
defaultDrawsForPolicy,
isGroup,
@@ -18,10 +17,17 @@ import {
emptyAutoQueueRuntime,
flattenLeaves,
policyDrawsBucket,
- sanitizeAutoQueue,
- sanitizeAutoQueueOrder,
selectNextWork,
} from "./autoQueuePolicy";
+// REPOINTED, NOT REWRITTEN (one-core phase 3 slice 4a). The defaults and the
+// sanitizer moved to lib/autoQueueSchema.ts; every assertion below is the one it
+// was, which is the point — this file is the proof that the move changed no
+// answer, from the `held` defaults to the tree normalisation.
+import {
+ defaultAutoQueue,
+ sanitizeAutoQueue,
+ sanitizeAutoQueueOrder,
+} from "../lib/autoQueueSchema";
import { bucketLaneOperationId } from "../lib/operations";
// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/jobs/autoQueuePolicy.test.ts
diff --git a/common/jobs/autoQueuePolicy.ts b/common/jobs/autoQueuePolicy.ts
@@ -30,6 +30,22 @@ export type {
};
export { LANES, isGroup };
+// THE SANITIZER MOVED DOWN A LAYER (one-core phase 3 slice 4a). Everything that
+// turns a raw settings.json value into a legal `AutoQueueSettings` now lives in
+// `lib/autoQueueSchema.ts`, with the rest of the settings schema — which is what
+// let `lib/settings.ts` stop importing this file. Re-exported here so every
+// existing `from "../jobs/autoQueuePolicy"` import still resolves, exactly as
+// the TYPES above are.
+export {
+ AUTO_QUEUE_MODES,
+ AUTO_QUEUE_ORDERS,
+ AUTO_QUEUE_MAX_WORKERS_MAX,
+ defaultAutoQueue,
+ defaultAutoQueuePolicy,
+ sanitizeAutoQueue,
+ sanitizeAutoQueueOrder,
+} from "../lib/autoQueueSchema";
+
// Pure, side-effect-free policy engine for the automatic priority queue. It
// decides WHICH pending video to process next, across all channels, from a
@@ -46,34 +62,6 @@ export { LANES, isGroup };
// falls through to the next-priority sibling — exactly like HTB's class ceil.
-export const AUTO_QUEUE_MODES: ReadonlyArray<AutoQueueMode> = [
- "strict",
- "round-robin",
- "weighted-fair",
-];
-
-export const AUTO_QUEUE_ORDERS: ReadonlyArray<AutoQueueOrder> = [
- "listed",
- "newest",
- "oldest",
- "cheapest",
-];
-
-
-// Coerce a stored/raw value to a legal order. Anything unrecognised — including
-// a missing field on a settings file written before the field existed — means
-// "listed", i.e. today's behaviour. One sanitizer, because the same enum is
-// stored on all four lane policies — it was stored in three MORE places before
-// slice 1.3 folded digest.recencyOrder and backfill.order into them — and copies
-// of this line would eventually disagree about what an absent field means.
-export function sanitizeAutoQueueOrder(value: unknown): AutoQueueOrder {
- return value === "newest" || value === "oldest" || value === "cheapest"
- ? value
- : "listed";
-}
-
-
-
// Buckets each runner kind draws from, in priority order. A leaf with no
// explicit bucket draws from the whole list (union, deduped); the list order is
// its internal priority. Single source of truth for the runner, the pending-
@@ -263,7 +251,6 @@ export function policyDrawsBucket(
);
}
-export const AUTO_QUEUE_MAX_WORKERS_MAX = 64;
// --- Runtime fairness state (persisted best-effort by autoQueueState.ts) ----
@@ -545,219 +532,3 @@ export function selectNextWork(
return pick(root, pending, runtime, active, []);
}
-// --- Defaults + sanitization (defensive, like sanitizeSyncScheduler) --------
-
-function clampMaxWorkers(value: unknown): number | null {
- if (value == null) return null;
- if (typeof value !== "number" || !Number.isFinite(value)) return null;
- const n = Math.floor(value);
- if (n < 1) return null;
- return Math.min(n, AUTO_QUEUE_MAX_WORKERS_MAX);
-}
-
-function clampWeight(value: unknown): number {
- if (typeof value !== "number" || !Number.isFinite(value)) return 1;
- const n = Math.floor(value);
- return n < 1 ? 1 : Math.min(n, AUTO_QUEUE_MAX_WORKERS_MAX);
-}
-
-function sanitizeMatch(value: unknown): AutoQueueMatch {
- const r = (value ?? {}) as Record<string, unknown>;
- const type: AutoQueueMatchType =
- r.type === "channel" || r.type === "platform" || r.type === "all"
- ? r.type
- : "all";
- const out: AutoQueueMatch = { type };
- if (typeof r.value === "string" && r.value.trim()) out.value = r.value.trim();
- const operation =
- typeof r.operation === "string" && r.operation.trim()
- ? r.operation.trim()
- : "";
- if (operation) {
- // Coerce-to-legal, this file's existing style: a leaf naming BOTH an
- // operation and a bucket is ambiguous, so the stored tree is not allowed to
- // express it. Operation wins and the bucket is dropped, rather than the
- // pair being kept and resolved differently by whichever reader looks first.
- out.operation = operation;
- return out;
- }
- if (typeof r.bucket === "string" && r.bucket.trim()) {
- out.bucket = r.bucket.trim();
- }
- return out;
-}
-
-// Coerce a raw node, assigning a unique id (provided id preserved when valid and
-// not already taken, so persisted fairness state survives an unrelated edit).
-function sanitizeNode(value: unknown, seen: Set<string>): AutoQueueNode {
- const r = (value ?? {}) as Record<string, unknown>;
- const id = takeId(r.id, seen);
- const weight = clampWeight(r.weight);
- const maxWorkers = clampMaxWorkers(r.maxWorkers);
- if (Array.isArray(r.children)) {
- const mode: AutoQueueMode = AUTO_QUEUE_MODES.includes(r.mode as AutoQueueMode)
- ? (r.mode as AutoQueueMode)
- : "strict";
- return {
- id,
- mode,
- weight,
- maxWorkers,
- children: r.children.map((c) => sanitizeNode(c, seen)),
- };
- }
- return { id, match: sanitizeMatch(r.match), weight, maxWorkers };
-}
-
-let idCounter = 0;
-function takeId(raw: unknown, seen: Set<string>): string {
- let id = typeof raw === "string" && raw.trim() ? raw.trim() : "";
- if (!id || seen.has(id)) {
- do {
- id = `node-${++idCounter}`;
- } while (seen.has(id));
- }
- seen.add(id);
- return id;
-}
-
-function sanitizeRoot(value: unknown, seen: Set<string>): AutoQueueGroup {
- const node = sanitizeNode(
- value && typeof value === "object" ? value : { mode: "strict", children: [] },
- seen,
- );
- if (isGroup(node)) return node;
- // A root that deserialized as a leaf is meaningless — wrap into an empty group.
- return { id: node.id, mode: "strict", weight: 1, maxWorkers: null, children: [] };
-}
-
-function emptyRoot(): AutoQueueGroup {
- return { id: "root", mode: "strict", weight: 1, maxWorkers: null, children: [] };
-}
-
-// THE DEFAULT TREE FOR A LANE, and the two answers are different on purpose.
-//
-// The runner lanes default to an EMPTY root: they have shipped that way since
-// the auto-queue existed, an empty tree dispatches nothing, and a settings file
-// that omits a root must keep meaning exactly that.
-//
-// The digest and backfill lanes default to one catch-all leaf, because their
-// work list is an operation's `ids` and a lane with no leaf at all could never
-// draw it. The leaf is inert while `enabled` is false — which is how they
-// default, and what keeps gate B (never enable the backfill lane against
-// ~66,540 missingInput videos by accident) a decision an operator still has to
-// take.
-function defaultRootFor(lane: AutoQueueKind): AutoQueueGroup {
- if (lane === "transcription" || lane === "download") return emptyRoot();
- return {
- id: "root",
- mode: "strict",
- weight: 1,
- maxWorkers: null,
- children: [{ id: "all", match: { type: "all" }, weight: 1, maxWorkers: null }],
- };
-}
-
-// The digest lane's historical ordering is SHORTEST-FIRST, and it is not
-// cosmetic: a 12-minute video is one chunk and a four-hour stream is thirty, so
-// draining the cheap end first is what makes a multi-week sweep show progress.
-// Defaulting the lane to "cheapest" is how that survives the move from the
-// sweep to the tree. The comparator arrives with the runner (slice 1.2); until
-// then next() has none for this order and falls back to today's.
-function defaultOrderFor(lane: AutoQueueKind): AutoQueueOrder {
- return lane === "digest" ? "cheapest" : "listed";
-}
-
-// THE DEFAULT GATE FOR A LANE, and only one lane ships held.
-//
-// It is not a new policy — it is the reading the four retired pause fields gave
-// a file that named no gate, preserved. `transcriptionsPaused`,
-// `downloadsPaused` and `digest.digestsPaused` all defaulted false (free);
-// `backfill.enabled` defaulted FALSE and was INVERTED, so the backfill lane has
-// shipped HELD since it existed. S0-pause deleted the fields, which is what
-// makes defaulting this key correct — and required, because from slice 1.4 until
-// S0-pause an absent `held` had somewhere else to ask, and now it has not.
-//
-// The backfill lane is therefore off twice over on a fresh install: unarmed
-// (`enabled: false`) and held. That is gate B — never enable the backfill lane
-// against ~66,540 missingInput videos by accident — kept as two deliberate acts.
-function defaultHeldFor(lane: AutoQueueKind): boolean {
- return lane === "backfill";
-}
-
-export function defaultAutoQueuePolicy(
- lane: AutoQueueKind = "transcription",
-): AutoQueuePolicy {
- return {
- enabled: false,
- maxWorkers: null,
- replaceAutoSubs: false,
- order: defaultOrderFor(lane),
- snoozeUntil: null,
- held: defaultHeldFor(lane),
- root: defaultRootFor(lane),
- };
-}
-
-export function defaultAutoQueue(): AutoQueueSettings {
- return Object.fromEntries(
- LANES.map((lane) => [lane, defaultAutoQueuePolicy(lane)]),
- ) as AutoQueueSettings;
-}
-
-// A snooze that has already lapsed is not a snooze: normalizing it to null here
-// means every reader (runner, status payload, UI) can treat "non-null" as "still
-// snoozed" without repeating the clock comparison. Re-sanitized on every read of
-// settings.json, so a stale value self-clears without anyone writing.
-function sanitizeSnooze(value: unknown): number | null {
- if (typeof value !== "number" || !Number.isFinite(value)) return null;
- const at = Math.floor(value);
- return at > Date.now() ? at : null;
-}
-
-// A LANE THAT IS NOT IN THE FILE IS THE LANE'S DEFAULT, not an empty object.
-//
-// Every settings.json in existence carries exactly two lanes, so the digest and
-// backfill blocks arrive `undefined` on every read until something writes them.
-// Coercing that to `sanitizePolicy({})` would give them an empty root — a lane
-// that can never draw anything even once an operator enables it — so the whole
-// default policy is the fallback, and only the fields the file actually names
-// override it.
-function sanitizePolicy(value: unknown, lane: AutoQueueKind): AutoQueuePolicy {
- if (value == null) return defaultAutoQueuePolicy(lane);
- const r = (value ?? {}) as Record<string, unknown>;
- const seen = new Set<string>();
- return {
- enabled: r.enabled === true,
- maxWorkers: clampMaxWorkers(r.maxWorkers),
- // Opt-in only: anything but an explicit `true` (including a missing field on
- // a pre-existing settings.json) leaves the lane off.
- replaceAutoSubs: r.replaceAutoSubs === true,
- // Anything unrecognised (including a missing field) means the lane's own
- // default — "listed" for the runner lanes, "cheapest" for digest.
- order: r.order === undefined
- ? defaultOrderFor(lane)
- : sanitizeAutoQueueOrder(r.order),
- snoozeUntil: sanitizeSnooze(r.snoozeUntil),
- // THE LANE'S PAUSE GATE, and the only spelling of one since S0-pause deleted
- // the four legacy fields it migrated from. DEFAULTED, which it deliberately
- // was not while those fields existed: an absent key used to mean "ask the
- // retired field", so filling it in here would have read a paused corpus as
- // running. There is nothing left to ask, and `defaultHeldFor` is the
- // reading those fields gave a file that named no gate — free everywhere
- // except backfill, whose field was inverted and defaulted to held.
- held: typeof r.held === "boolean" ? r.held : defaultHeldFor(lane),
- root:
- r.root === undefined
- ? defaultRootFor(lane)
- : sanitizeRoot(r.root, seen),
- };
-}
-
-export function sanitizeAutoQueue(value: unknown): AutoQueueSettings {
- if (!value || typeof value !== "object") return defaultAutoQueue();
- const r = value as Record<string, unknown>;
- return Object.fromEntries(
- LANES.map((lane) => [lane, sanitizePolicy(r[lane], lane)]),
- ) as AutoQueueSettings;
-}
diff --git a/common/lib/autoQueueSchema.ts b/common/lib/autoQueueSchema.ts
@@ -0,0 +1,284 @@
+// THE AUTO-QUEUE'S HALF OF THE SETTINGS SCHEMA.
+//
+// Four lane policies live under `settings.autoQueue`, and until one-core phase 3
+// slice 4a their defaults and their sanitizer lived in `jobs/autoQueuePolicy.ts`
+// beside the PICKER that reads them. That was the one back-edge `lib/settings.ts`
+// still carried (`common/architecture.test.ts`'s allow-list entry
+// `lib/settings.ts -> jobs/autoQueuePolicy`, now deleted): the model layer
+// importing dispatch to learn the shape of its own file.
+//
+// The split is by ROLE, not by size. What is here is everything that turns a
+// raw JSON value into a legal `AutoQueueSettings` — the defaults, the clamps,
+// the tree normalisation, the lane gate. What stays in `jobs/autoQueuePolicy.ts`
+// is everything that CHOOSES with it: bucket lists, pending-set construction and
+// the SWRR resolver. `autoQueuePolicy.ts` re-exports every name moved here, so
+// no existing import site changed.
+//
+// It never throws. Every function is total over `unknown`, because its input is
+// a file an operator may have hand-edited and a settings read may not fail.
+//
+// NO ZOD HERE, deliberately. Four `"use client"` forms import constants from
+// this module through `jobs/autoQueuePolicy` (LadderRung reads
+// AUTO_QUEUE_MODES), so anything this file imports can land in a browser
+// bundle. The zod field seam that wraps `sanitizeAutoQueue` lives in
+// `lib/settingsFieldSchemas.ts`, which only server code imports.
+
+import {
+ type AutoQueueGroup,
+ type AutoQueueKind,
+ type AutoQueueMatch,
+ type AutoQueueMatchType,
+ type AutoQueueMode,
+ type AutoQueueNode,
+ type AutoQueueOrder,
+ type AutoQueuePolicy,
+ type AutoQueueSettings,
+ LANES,
+ isGroup,
+} from "./autoQueueTypes";
+
+export const AUTO_QUEUE_MODES: ReadonlyArray<AutoQueueMode> = [
+ "strict",
+ "round-robin",
+ "weighted-fair",
+];
+
+export const AUTO_QUEUE_ORDERS: ReadonlyArray<AutoQueueOrder> = [
+ "listed",
+ "newest",
+ "oldest",
+ "cheapest",
+];
+
+
+// Coerce a stored/raw value to a legal order. Anything unrecognised — including
+// a missing field on a settings file written before the field existed — means
+// "listed", i.e. today's behaviour. One sanitizer, because the same enum is
+// stored on all four lane policies — it was stored in three MORE places before
+// slice 1.3 folded digest.recencyOrder and backfill.order into them — and copies
+// of this line would eventually disagree about what an absent field means.
+export function sanitizeAutoQueueOrder(value: unknown): AutoQueueOrder {
+ return value === "newest" || value === "oldest" || value === "cheapest"
+ ? value
+ : "listed";
+}
+
+export const AUTO_QUEUE_MAX_WORKERS_MAX = 64;
+
+// --- Defaults + sanitization (defensive, like sanitizeSyncScheduler) --------
+
+function clampMaxWorkers(value: unknown): number | null {
+ if (value == null) return null;
+ if (typeof value !== "number" || !Number.isFinite(value)) return null;
+ const n = Math.floor(value);
+ if (n < 1) return null;
+ return Math.min(n, AUTO_QUEUE_MAX_WORKERS_MAX);
+}
+
+function clampWeight(value: unknown): number {
+ if (typeof value !== "number" || !Number.isFinite(value)) return 1;
+ const n = Math.floor(value);
+ return n < 1 ? 1 : Math.min(n, AUTO_QUEUE_MAX_WORKERS_MAX);
+}
+
+function sanitizeMatch(value: unknown): AutoQueueMatch {
+ const r = (value ?? {}) as Record<string, unknown>;
+ const type: AutoQueueMatchType =
+ r.type === "channel" || r.type === "platform" || r.type === "all"
+ ? r.type
+ : "all";
+ const out: AutoQueueMatch = { type };
+ if (typeof r.value === "string" && r.value.trim()) out.value = r.value.trim();
+ const operation =
+ typeof r.operation === "string" && r.operation.trim()
+ ? r.operation.trim()
+ : "";
+ if (operation) {
+ // Coerce-to-legal, this file's existing style: a leaf naming BOTH an
+ // operation and a bucket is ambiguous, so the stored tree is not allowed to
+ // express it. Operation wins and the bucket is dropped, rather than the
+ // pair being kept and resolved differently by whichever reader looks first.
+ out.operation = operation;
+ return out;
+ }
+ if (typeof r.bucket === "string" && r.bucket.trim()) {
+ out.bucket = r.bucket.trim();
+ }
+ return out;
+}
+
+// Coerce a raw node, assigning a unique id (provided id preserved when valid and
+// not already taken, so persisted fairness state survives an unrelated edit).
+function sanitizeNode(value: unknown, seen: Set<string>): AutoQueueNode {
+ const r = (value ?? {}) as Record<string, unknown>;
+ const id = takeId(r.id, seen);
+ const weight = clampWeight(r.weight);
+ const maxWorkers = clampMaxWorkers(r.maxWorkers);
+ if (Array.isArray(r.children)) {
+ const mode: AutoQueueMode = AUTO_QUEUE_MODES.includes(r.mode as AutoQueueMode)
+ ? (r.mode as AutoQueueMode)
+ : "strict";
+ return {
+ id,
+ mode,
+ weight,
+ maxWorkers,
+ children: r.children.map((c) => sanitizeNode(c, seen)),
+ };
+ }
+ return { id, match: sanitizeMatch(r.match), weight, maxWorkers };
+}
+
+let idCounter = 0;
+function takeId(raw: unknown, seen: Set<string>): string {
+ let id = typeof raw === "string" && raw.trim() ? raw.trim() : "";
+ if (!id || seen.has(id)) {
+ do {
+ id = `node-${++idCounter}`;
+ } while (seen.has(id));
+ }
+ seen.add(id);
+ return id;
+}
+
+function sanitizeRoot(value: unknown, seen: Set<string>): AutoQueueGroup {
+ const node = sanitizeNode(
+ value && typeof value === "object" ? value : { mode: "strict", children: [] },
+ seen,
+ );
+ if (isGroup(node)) return node;
+ // A root that deserialized as a leaf is meaningless — wrap into an empty group.
+ return { id: node.id, mode: "strict", weight: 1, maxWorkers: null, children: [] };
+}
+
+function emptyRoot(): AutoQueueGroup {
+ return { id: "root", mode: "strict", weight: 1, maxWorkers: null, children: [] };
+}
+
+// THE DEFAULT TREE FOR A LANE, and the two answers are different on purpose.
+//
+// The runner lanes default to an EMPTY root: they have shipped that way since
+// the auto-queue existed, an empty tree dispatches nothing, and a settings file
+// that omits a root must keep meaning exactly that.
+//
+// The digest and backfill lanes default to one catch-all leaf, because their
+// work list is an operation's `ids` and a lane with no leaf at all could never
+// draw it. The leaf is inert while `enabled` is false — which is how they
+// default, and what keeps gate B (never enable the backfill lane against
+// ~66,540 missingInput videos by accident) a decision an operator still has to
+// take.
+function defaultRootFor(lane: AutoQueueKind): AutoQueueGroup {
+ if (lane === "transcription" || lane === "download") return emptyRoot();
+ return {
+ id: "root",
+ mode: "strict",
+ weight: 1,
+ maxWorkers: null,
+ children: [{ id: "all", match: { type: "all" }, weight: 1, maxWorkers: null }],
+ };
+}
+
+// The digest lane's historical ordering is SHORTEST-FIRST, and it is not
+// cosmetic: a 12-minute video is one chunk and a four-hour stream is thirty, so
+// draining the cheap end first is what makes a multi-week sweep show progress.
+// Defaulting the lane to "cheapest" is how that survives the move from the
+// sweep to the tree. The comparator arrives with the runner (slice 1.2); until
+// then next() has none for this order and falls back to today's.
+function defaultOrderFor(lane: AutoQueueKind): AutoQueueOrder {
+ return lane === "digest" ? "cheapest" : "listed";
+}
+
+// THE DEFAULT GATE FOR A LANE, and only one lane ships held.
+//
+// It is not a new policy — it is the reading the four retired pause fields gave
+// a file that named no gate, preserved. `transcriptionsPaused`,
+// `downloadsPaused` and `digest.digestsPaused` all defaulted false (free);
+// `backfill.enabled` defaulted FALSE and was INVERTED, so the backfill lane has
+// shipped HELD since it existed. S0-pause deleted the fields, which is what
+// makes defaulting this key correct — and required, because from slice 1.4 until
+// S0-pause an absent `held` had somewhere else to ask, and now it has not.
+//
+// The backfill lane is therefore off twice over on a fresh install: unarmed
+// (`enabled: false`) and held. That is gate B — never enable the backfill lane
+// against ~66,540 missingInput videos by accident — kept as two deliberate acts.
+function defaultHeldFor(lane: AutoQueueKind): boolean {
+ return lane === "backfill";
+}
+
+export function defaultAutoQueuePolicy(
+ lane: AutoQueueKind = "transcription",
+): AutoQueuePolicy {
+ return {
+ enabled: false,
+ maxWorkers: null,
+ replaceAutoSubs: false,
+ order: defaultOrderFor(lane),
+ snoozeUntil: null,
+ held: defaultHeldFor(lane),
+ root: defaultRootFor(lane),
+ };
+}
+
+export function defaultAutoQueue(): AutoQueueSettings {
+ return Object.fromEntries(
+ LANES.map((lane) => [lane, defaultAutoQueuePolicy(lane)]),
+ ) as AutoQueueSettings;
+}
+
+// A snooze that has already lapsed is not a snooze: normalizing it to null here
+// means every reader (runner, status payload, UI) can treat "non-null" as "still
+// snoozed" without repeating the clock comparison. Re-sanitized on every read of
+// settings.json, so a stale value self-clears without anyone writing.
+function sanitizeSnooze(value: unknown): number | null {
+ if (typeof value !== "number" || !Number.isFinite(value)) return null;
+ const at = Math.floor(value);
+ return at > Date.now() ? at : null;
+}
+
+// A LANE THAT IS NOT IN THE FILE IS THE LANE'S DEFAULT, not an empty object.
+//
+// Every settings.json in existence carries exactly two lanes, so the digest and
+// backfill blocks arrive `undefined` on every read until something writes them.
+// Coercing that to `sanitizePolicy({})` would give them an empty root — a lane
+// that can never draw anything even once an operator enables it — so the whole
+// default policy is the fallback, and only the fields the file actually names
+// override it.
+function sanitizePolicy(value: unknown, lane: AutoQueueKind): AutoQueuePolicy {
+ if (value == null) return defaultAutoQueuePolicy(lane);
+ const r = (value ?? {}) as Record<string, unknown>;
+ const seen = new Set<string>();
+ return {
+ enabled: r.enabled === true,
+ maxWorkers: clampMaxWorkers(r.maxWorkers),
+ // Opt-in only: anything but an explicit `true` (including a missing field on
+ // a pre-existing settings.json) leaves the lane off.
+ replaceAutoSubs: r.replaceAutoSubs === true,
+ // Anything unrecognised (including a missing field) means the lane's own
+ // default — "listed" for the runner lanes, "cheapest" for digest.
+ order: r.order === undefined
+ ? defaultOrderFor(lane)
+ : sanitizeAutoQueueOrder(r.order),
+ snoozeUntil: sanitizeSnooze(r.snoozeUntil),
+ // THE LANE'S PAUSE GATE, and the only spelling of one since S0-pause deleted
+ // the four legacy fields it migrated from. DEFAULTED, which it deliberately
+ // was not while those fields existed: an absent key used to mean "ask the
+ // retired field", so filling it in here would have read a paused corpus as
+ // running. There is nothing left to ask, and `defaultHeldFor` is the
+ // reading those fields gave a file that named no gate — free everywhere
+ // except backfill, whose field was inverted and defaulted to held.
+ held: typeof r.held === "boolean" ? r.held : defaultHeldFor(lane),
+ root:
+ r.root === undefined
+ ? defaultRootFor(lane)
+ : sanitizeRoot(r.root, seen),
+ };
+}
+
+export function sanitizeAutoQueue(value: unknown): AutoQueueSettings {
+ if (!value || typeof value !== "object") return defaultAutoQueue();
+ const r = value as Record<string, unknown>;
+ return Object.fromEntries(
+ LANES.map((lane) => [lane, sanitizePolicy(r[lane], lane)]),
+ ) as AutoQueueSettings;
+}
+
diff --git a/common/lib/autoQueueTypes.ts b/common/lib/autoQueueTypes.ts
@@ -9,6 +9,8 @@
//
// Enforced by ../architecture.test.ts.
+import type { FieldDocs } from "./fieldDocs";
+
// --- Tree types -------------------------------------------------------------
export type AutoQueueMode = "strict" | "round-robin" | "weighted-fair";
@@ -33,40 +35,48 @@ export type AutoQueueMatchType = "channel" | "platform" | "all";
// which is what makes it safe to name here before anything computes it.
export type AutoQueueOrder = "listed" | "newest" | "oldest" | "cheapest";
+// Each field is documented in AUTO_QUEUE_MATCH_FIELD_DOCS below (rendered into SETTINGS.md).
export type AutoQueueMatch = {
type: AutoQueueMatchType;
- // Channel slug (type=channel) or platform name (type=platform). Ignored for
- // type=all. A type=channel leaf with no value matches nothing.
value?: string;
- // Optional snapshot bucket this leaf draws from, narrowing the default for the
- // runner kind (transcription → downloadedNoTranscript, download →
- // undownloadedIds). E.g. bucket="failedListed" prioritizes retries.
bucket?: string;
- // Optional OPERATION this leaf draws from — a registered backfill kind id, or
- // "digest". Same meaning as `bucket` one level up: it narrows what the leaf
- // claims, and it draws from ChannelWork.operations rather than
- // ChannelWork.buckets.
- //
- // It lives on the MATCH, beside `bucket`, and not on the node. A field on the
- // node would need group inheritance — "this group is the digest subtree" —
- // and inheritance is resolution logic buildPendingByLeaf does not have. Here
- // it needs exactly one sanitizer and exactly one claiming path.
- //
- // It is a SEPARATE id space from `bucket`, and the sanitizer enforces that a
- // leaf names at most one of the two (operation wins): `defaultBuckets` is a
- // priority-ordered union, so a name that meant a bucket to one leaf and an
- // operation to another would silently mix two id spaces, and
- // selectableBucketsForKind feeds the editor's bucket dropdown, where an
- // operation must not appear as a bucket.
operation?: string;
};
+export const AUTO_QUEUE_MATCH_FIELD_DOCS: FieldDocs<AutoQueueMatch> = {
+ type:
+ "What the leaf matches: \"channel\" (one channel slug in `value`), \"platform\" (a platform name in `value`), or \"all\".",
+ value:
+ "Channel slug (type=channel) or platform name (type=platform). Ignored " +
+ "for type=all. A type=channel leaf with no value matches nothing.",
+ bucket:
+ "Optional snapshot bucket this leaf draws from, narrowing the default " +
+ "for the runner kind (transcription → downloadedNoTranscript, download " +
+ "→ undownloadedIds). E.g. bucket=\"failedListed\" prioritizes retries.",
+ operation:
+ "Optional OPERATION this leaf draws from — a registered backfill kind " +
+ "id, or \"digest\". Same meaning as `bucket` one level up: it narrows " +
+ "what the leaf claims, and it draws from ChannelWork.operations rather " +
+ "than ChannelWork.buckets.\n\n" +
+ "It lives on the MATCH, beside `bucket`, and not on the node. A field " +
+ "on the node would need group inheritance — \"this group is the digest " +
+ "subtree\" — and inheritance is resolution logic buildPendingByLeaf does" +
+ " not have. Here it needs exactly one sanitizer and exactly one " +
+ "claiming path.\n\n" +
+ "It is a SEPARATE id space from `bucket`, and the sanitizer enforces " +
+ "that a leaf names at most one of the two (operation wins): " +
+ "`defaultBuckets` is a priority-ordered union, so a name that meant a " +
+ "bucket to one leaf and an operation to another would silently mix two " +
+ "id spaces, and selectableBucketsForKind feeds the editor's bucket " +
+ "dropdown, where an operation must not appear as a bucket.",
+};
+
+// A node of a lane's rule tree is a leaf or a group; both are documented in
+// AUTO_QUEUE_NODE_FIELD_DOCS below (rendered into SETTINGS.md).
export type AutoQueueLeaf = {
id: string;
match: AutoQueueMatch;
- // Relative share under a weighted-fair parent. Default 1. Ignored otherwise.
weight?: number;
- // Optional ceiling on concurrent in-flight workers drawn from this leaf.
maxWorkers?: number | null;
};
@@ -80,57 +90,96 @@ export type AutoQueueGroup = {
export type AutoQueueNode = AutoQueueLeaf | AutoQueueGroup;
+export const AUTO_QUEUE_NODE_FIELD_DOCS: FieldDocs<AutoQueueNode> = {
+ id:
+ "Stable node id, unique within the lane's tree. Preserved on save when " +
+ "valid and not taken, so persisted fairness state survives an unrelated " +
+ "edit; a missing or duplicate id is replaced with a generated one.",
+ match:
+ "LEAF ONLY: which videos this leaf owns (see the match table).",
+ weight:
+ "Relative share under a weighted-fair parent. Default 1. Ignored " +
+ "otherwise.",
+ maxWorkers:
+ "Optional ceiling on concurrent in-flight workers drawn from this node " +
+ "(and, for a group, its whole subtree). A capped node reads as \"no " +
+ "work\" and the parent falls through to the next sibling, like an HTB " +
+ "class ceiling. null = no cap.",
+ mode:
+ "GROUP ONLY: how the children compete — \"strict\" (first child with " +
+ "work wins), \"round-robin\", or \"weighted-fair\" (by each child's " +
+ "`weight`). Unknown values read as \"strict\".",
+ children:
+ "GROUP ONLY: the child nodes, in priority order for a strict group. A " +
+ "node with a `children` array is a group; any other node is a leaf.",
+};
+
export function isGroup(node: AutoQueueNode): node is AutoQueueGroup {
return Array.isArray((node as AutoQueueGroup).children);
}
// --- Settings (persisted in settings.json under `autoQueue`) ----------------
+// Each field is documented in AUTO_QUEUE_POLICY_FIELD_DOCS below (rendered into SETTINGS.md).
export type AutoQueuePolicy = {
- // Master switch for this runner (transcription / download independently).
enabled: boolean;
- // Overall ceiling on concurrent in-flight workers for this runner. null = no
- // runner-level cap (the worker pool / platform queues are the real throttle).
maxWorkers: number | null;
- // Opt in to the lowest-priority "replace YouTube auto-captions" lane: append
- // this kind's opt-in buckets (autoSubsOnly / downloadedAutoSubsOnly) to the
- // tail of the default union, so videos whose only transcript is YouTube ASR
- // get re-done with our own engine whenever nothing more important is pending.
- // Default false — the corpus-wide cost is large (an audio download plus a
- // transcription per video). A leaf can also target the bucket by name for
- // per-channel opt-in without flipping this switch. Optional: settings written
- // before this field existed lack it; the sanitizer defaults it to false.
replaceAutoSubs?: boolean;
- // Ordering within each rule (see AutoQueueOrder). Optional exactly like
- // replaceAutoSubs: settings files written before this field existed lack it,
- // and the sanitizer defaults them to "listed" (today's behaviour).
order?: AutoQueueOrder;
- // Epoch ms until which this runner idles WITHOUT stopping: next() returns null
- // so the loop stays up, re-reads settings each iteration, and resumes by
- // itself when the moment passes. null/absent/past = not snoozed. Survives a
- // restart because it lives in settings.json, not in runner memory.
snoozeUntil?: number | null;
- // THE LANE'S PAUSE GATE. Shut means the lane holds: every dispatch path asks
- // lib/pauseGates.ts, whose limit()/guard returns 0 so runPool idle-waits. A
- // hold, never a stop — see that file's header.
- //
- // OPTIONAL IN THE TYPE, FILLED BY THE SANITIZER. Until slice 1.4 four separate
- // settings fields carried this — `transcriptionsPaused`, `downloadsPaused`,
- // `digest.digestsPaused` and (inverted) `backfill.enabled` — so `undefined`
- // meant "ask the legacy field" and `sanitizePolicy` deliberately refused to
- // default it: a default would have read a paused corpus as running. S0-pause
- // deleted those four, on the precondition that the live settings.json already
- // carried every `held` key, and the default came in with them
- // (`defaultHeldFor` — free everywhere except backfill, whose field was
- // inverted and shipped held).
- //
- // It stays optional because a reader may be handed a PARTIAL settings object
- // (laneGuards.test.ts casts one), and `isGateHeld` answers `false` for a lane
- // that carries no key at all rather than throwing.
held?: boolean;
root: AutoQueueGroup;
};
+export const AUTO_QUEUE_POLICY_FIELD_DOCS: FieldDocs<AutoQueuePolicy> = {
+ enabled:
+ "Master switch for this runner (transcription / download " +
+ "independently).",
+ maxWorkers:
+ "Overall ceiling on concurrent in-flight workers for this runner. null " +
+ "= no runner-level cap (the worker pool / platform queues are the real " +
+ "throttle).",
+ replaceAutoSubs:
+ "Opt in to the lowest-priority \"replace YouTube auto-captions\" lane: " +
+ "append this kind's opt-in buckets (autoSubsOnly / " +
+ "downloadedAutoSubsOnly) to the tail of the default union, so videos " +
+ "whose only transcript is YouTube ASR get re-done with our own engine " +
+ "whenever nothing more important is pending. Default false — the " +
+ "corpus-wide cost is large (an audio download plus a transcription per " +
+ "video). A leaf can also target the bucket by name for per-channel opt-" +
+ "in without flipping this switch. Optional: settings written before " +
+ "this field existed lack it; the sanitizer defaults it to false.",
+ order:
+ "Ordering within each rule (see AutoQueueOrder). Optional exactly like " +
+ "replaceAutoSubs: settings files written before this field existed lack" +
+ " it, and the sanitizer defaults them to \"listed\" (today's behaviour).",
+ snoozeUntil:
+ "Epoch ms until which this runner idles WITHOUT stopping: next() " +
+ "returns null so the loop stays up, re-reads settings each iteration, " +
+ "and resumes by itself when the moment passes. null/absent/past = not " +
+ "snoozed. Survives a restart because it lives in settings.json, not in " +
+ "runner memory.",
+ held:
+ "THE LANE'S PAUSE GATE. Shut means the lane holds: every dispatch path " +
+ "asks lib/pauseGates.ts, whose limit()/guard returns 0 so runPool idle-" +
+ "waits. A hold, never a stop — see that file's header.\n\n" +
+ "OPTIONAL IN THE TYPE, FILLED BY THE SANITIZER. Until slice 1.4 four " +
+ "separate settings fields carried this — `transcriptionsPaused`, " +
+ "`downloadsPaused`, `digest.digestsPaused` and (inverted) " +
+ "`backfill.enabled` — so `undefined` meant \"ask the legacy field\" and " +
+ "`sanitizePolicy` deliberately refused to default it: a default would " +
+ "have read a paused corpus as running. S0-pause deleted those four, on " +
+ "the precondition that the live settings.json already carried every " +
+ "`held` key, and the default came in with them (`defaultHeldFor` — free" +
+ " everywhere except backfill, whose field was inverted and shipped " +
+ "held).\n\n" +
+ "It stays optional because a reader may be handed a PARTIAL settings " +
+ "object (laneGuards.test.ts casts one), and `isGateHeld` answers " +
+ "`false` for a lane that carries no key at all rather than throwing.",
+ root:
+ "The lane's rule tree: a group whose children are groups and leaves (see the node table). A missing root is the lane's default — empty for the runner lanes, one catch-all leaf for digest and backfill. While a channel-priority document exists, the four roots are compiled from it and not hand-edited.",
+};
+
export type AutoQueueSettings = Record<AutoQueueKind, AutoQueuePolicy>;
// --- Lanes ------------------------------------------------------------------
diff --git a/common/lib/channelPriority.ts b/common/lib/channelPriority.ts
@@ -58,6 +58,7 @@ import {
type AutoQueueNode,
isGroup,
} from "./autoQueueTypes";
+import type { FieldDocs } from "./fieldDocs";
// --- The vocabulary ---------------------------------------------------------
@@ -119,56 +120,98 @@ export type ChannelFocus =
| { kind: "site"; siteId: string }
| { kind: "channels"; slugs: string[] };
+export const CHANNEL_FOCUS_FIELD_DOCS: FieldDocs<ChannelFocus> = {
+ kind:
+ "\"none\" (no focus), \"site\" (the channels of one site, resolved at " +
+ "compile time so it tracks membership) or \"channels\" (an explicit " +
+ "list, from \"Focus these\").",
+ siteId:
+ "kind \"site\" only: the site whose channels are focused. A blank id " +
+ "reads as no focus; an unknown one survives and focuses nothing.",
+ slugs:
+ "kind \"channels\" only: the focused channel slugs, trimmed and " +
+ "de-duplicated. An empty list reads as no focus.",
+};
+
+// Each field is documented in CHANNEL_PRIORITY_ENTRY_FIELD_DOCS below (rendered into SETTINGS.md).
export type ChannelPriorityEntry = {
- // THE BASE TIER: what every operation gets unless an override says otherwise.
tier: StoredChannelTier;
- // Order WITHIN the tier, ascending. Absent = unranked, which sorts after
- // every ranked sibling and then by slug. ONE rank per channel, not one per
- // lane — the two hand-made lane orders collapse into this on migration.
rank?: number;
- // PER-OPERATION OVERRIDES of the base tier. Only operations that DIFFER from
- // the base appear: the sanitizer normalises an override equal to `tier` away,
- // so the on-disk document stays a list of exceptions to a list of exceptions.
- //
- // `{tier:"normal", overrides:{sync:"paused"}}` is "everything but sync" — the
- // lossless reading of the retired `excludeFromSync`. Its inverse,
- // `{tier:"paused", overrides:{sync:"normal"}}`, is "sync only": keep the
- // playlist and metadata current, dispatch nothing.
overrides?: Partial<Record<PriorityOperation, StoredChannelTier>>;
- // PAUSED BY THE MACHINE, NOT BY THE OPERATOR, and what to put back.
- //
- // Set when the drive a channel's media is on stops being there: the watch
- // pass records the tier the channel HAD and forces `paused`, so nothing in
- // any lane dispatches against a `data/` nobody can read. Cleared — and the
- // tier restored — when the drive comes back.
- //
- // WHY IT IS A FIELD AND NOT A DERIVED STATE. The lanes read `tier`; making
- // them all ask a second question would be four more places to forget. And
- // the tier the channel is to be RESTORED to is not derivable from anything
- // once it has been overwritten — that is the whole content of this field.
- //
- // OPTIONAL, and an older binary that drops it leaves the channel Paused with
- // nothing lost but the automatic restore. The operator's own word always
- // wins: a MANUAL tier change clears it (see clearAutoPause), so a drive
- // coming back can never un-pause a channel somebody paused on purpose.
- autoPaused?: {
- // One reason today. A union so a second one has somewhere to go, and so a
- // surface can say WHICH machine decided rather than "automatic".
- reason: "storage";
- // ISO, for "auto-paused — media unreachable since <date>".
- since: string;
- previousTier: StoredChannelTier;
- };
+ autoPaused?: ChannelAutoPause;
};
+export const CHANNEL_PRIORITY_ENTRY_FIELD_DOCS: FieldDocs<ChannelPriorityEntry> = {
+ tier:
+ "THE BASE TIER: what every operation gets unless an override says " +
+ "otherwise.",
+ rank:
+ "Order WITHIN the tier, ascending. Absent = unranked, which sorts after" +
+ " every ranked sibling and then by slug. ONE rank per channel, not one " +
+ "per lane — the two hand-made lane orders collapse into this on " +
+ "migration.",
+ overrides:
+ "PER-OPERATION OVERRIDES of the base tier. Only operations that DIFFER " +
+ "from the base appear: the sanitizer normalises an override equal to " +
+ "`tier` away, so the on-disk document stays a list of exceptions to a " +
+ "list of exceptions.\n\n" +
+ "`{tier:\"normal\", overrides:{sync:\"paused\"}}` is \"everything but sync\" " +
+ "— the lossless reading of the retired `excludeFromSync`. Its inverse, " +
+ "`{tier:\"paused\", overrides:{sync:\"normal\"}}`, is \"sync only\": keep the" +
+ " playlist and metadata current, dispatch nothing.",
+ autoPaused:
+ "PAUSED BY THE MACHINE, NOT BY THE OPERATOR, and what to put back.\n\n" +
+ "Set when the drive a channel's media is on stops being there: the " +
+ "watch pass records the tier the channel HAD and forces `paused`, so " +
+ "nothing in any lane dispatches against a `data/` nobody can read. " +
+ "Cleared — and the tier restored — when the drive comes back.\n\n" +
+ "WHY IT IS A FIELD AND NOT A DERIVED STATE. The lanes read `tier`; " +
+ "making them all ask a second question would be four more places to " +
+ "forget. And the tier the channel is to be RESTORED to is not derivable" +
+ " from anything once it has been overwritten — that is the whole " +
+ "content of this field.\n\n" +
+ "OPTIONAL, and an older binary that drops it leaves the channel Paused " +
+ "with nothing lost but the automatic restore. The operator's own word " +
+ "always wins: a MANUAL tier change clears it (see clearAutoPause), so a" +
+ " drive coming back can never un-pause a channel somebody paused on " +
+ "purpose.",
+};
+
+// The machine's pause record on a channel entry (see `autoPaused` above).
+// Each field is documented in CHANNEL_AUTO_PAUSE_FIELD_DOCS below (rendered into SETTINGS.md).
+export type ChannelAutoPause = {
+ reason: "storage";
+ since: string;
+ previousTier: StoredChannelTier;
+};
+
+export const CHANNEL_AUTO_PAUSE_FIELD_DOCS: FieldDocs<ChannelAutoPause> = {
+ reason:
+ "One reason today. A union so a second one has somewhere to go, and so " +
+ "a surface can say WHICH machine decided rather than \"automatic\".",
+ since:
+ "ISO, for \"auto-paused — media unreachable since <date>\".",
+ previousTier:
+ "The base tier the channel had before the machine paused it; what a " +
+ "restore puts back. Never `paused` (that would restore to paused — a " +
+ "no-op dressed as a restore).",
+};
+
+// Each field is documented in CHANNEL_PRIORITY_FIELD_DOCS below (rendered into SETTINGS.md).
export type ChannelPriority = {
focus: ChannelFocus;
- // ONLY channels that differ from the default appear. An absent slug is
- // `normal`, unranked — so the default document is empty and "absent document
- // = today's behaviour" holds byte for byte.
channels: Record<string, ChannelPriorityEntry>;
};
+export const CHANNEL_PRIORITY_FIELD_DOCS: FieldDocs<ChannelPriority> = {
+ focus:
+ "The corpus-wide focus selector: none, one site's channels, or a list of channels. A focus is compiled into a leading `prio-focus` group in every lane's tree.",
+ channels:
+ "ONLY channels that differ from the default appear. An absent slug is " +
+ "`normal`, unranked — so the default document is empty and \"absent " +
+ "document = today's behaviour\" holds byte for byte.",
+};
+
export function defaultChannelPriority(): ChannelPriority {
return { focus: { kind: "none" }, channels: {} };
}
diff --git a/common/lib/digest.ts b/common/lib/digest.ts
@@ -17,6 +17,8 @@
// NEVER rename these to `transcript.<x>.<y>` — SUB_FILE_RE in videoStatus.ts
// would claim such a file as a subtitle track.
+import type { FieldDocs } from "./fieldDocs";
+
export const DIGEST_FILENAME = "ai-digest.json";
export const DIGEST_OVERRIDES_FILENAME = "ai-digest.overrides.json";
@@ -144,33 +146,46 @@ export const DEFAULT_DIGEST_APP_ID = OLLAMA_DIGEST_APP_ID;
// Per-app configuration persisted under settings.digest.apps[id]. Every field is
// optional; an app falls back to its own defaults.
+// Each field is documented in DIGEST_APP_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md).
export type DigestAppConfig = {
- // Binary path/name override (process-based apps only).
bin?: string;
- // Base URL override (HTTP apps only).
baseUrl?: string;
- // Model id, e.g. "qwen2.5:7b" or "haiku".
model?: string;
- // Context window in tokens. MUST reach the engine explicitly for ollama: its
- // 4096 default silently truncates the input and the model then summarizes
- // whatever fragment survived — measured, and the single easiest way to get
- // quietly-wrong output at scale.
numCtx?: number;
- // Sampling temperature. 0 for a structured extraction task.
temperature?: number;
- // Reasoning-model toggle (ollama's top-level `think`). Only sent when set, so
- // a model that does not support thinking is never handed a field it rejects.
- //
- // It matters for throughput, not correctness: measured on this box, qwen3:8b
- // with thinking on spends most of its output budget on a `thinking` block
- // before the JSON body the schema constrains. For an extraction task with a
- // pinned schema that reasoning buys little and costs a multiple of the tokens,
- // and tokens are what a multi-week sweep is priced in.
think?: boolean;
- // Per-request wall-clock ceiling (ms). A wedged engine must not stall a sweep.
timeoutMs?: number;
};
+export const DIGEST_APP_CONFIG_FIELD_DOCS: FieldDocs<DigestAppConfig> = {
+ bin:
+ "Binary path/name override (process-based apps only).",
+ baseUrl:
+ "Base URL override (HTTP apps only).",
+ model:
+ "Model id, e.g. \"qwen2.5:7b\" or \"haiku\".",
+ numCtx:
+ "Context window in tokens. MUST reach the engine explicitly for ollama:" +
+ " its 4096 default silently truncates the input and the model then " +
+ "summarizes whatever fragment survived — measured, and the single " +
+ "easiest way to get quietly-wrong output at scale.",
+ temperature:
+ "Sampling temperature. 0 for a structured extraction task.",
+ think:
+ "Reasoning-model toggle (ollama's top-level `think`). Only sent when " +
+ "set, so a model that does not support thinking is never handed a field" +
+ " it rejects.\n\n" +
+ "It matters for throughput, not correctness: measured on this box, " +
+ "qwen3:8b with thinking on spends most of its output budget on a " +
+ "`thinking` block before the JSON body the schema constrains. For an " +
+ "extraction task with a pinned schema that reasoning buys little and " +
+ "costs a multiple of the tokens, and tokens are what a multi-week sweep" +
+ " is priced in.",
+ timeoutMs:
+ "Per-request wall-clock ceiling (ms). A wedged engine must not stall a " +
+ "sweep.",
+};
+
// Whether an engine-reported model resolution looks like a DIFFERENT model
// rather than a benign tag completion. "qwen2.5" resolving to "qwen2.5:7b" or
// "qwen2.5:latest" is ollama filling in a tag; "qwen2.5:7b" coming back as
diff --git a/common/lib/fieldDocs.ts b/common/lib/fieldDocs.ts
@@ -0,0 +1,14 @@
+// A DESCRIPTION FOR EVERY KEY OF A SETTINGS BLOCK, checked by the compiler.
+//
+// Each settings.json block type carries a `<TYPE>_FIELD_DOCS: FieldDocs<Type>`
+// record beside it. The mapped type requires one entry per key — optional keys
+// included, and every member's keys when the type is a union — so adding a
+// field without documenting it is a tsc error, and a stale entry for a removed
+// field is an excess-property error. The records are rendered into SETTINGS.md
+// by lib/settingsDocs.ts; they are the one home of each field's documentation.
+//
+// Pure, no imports: a `"use client"` module may carry a record.
+
+type AllKeys<T> = T extends unknown ? keyof T : never;
+
+export type FieldDocs<T> = { readonly [K in AllKeys<T> & string]: string };
diff --git a/common/lib/settings.ts b/common/lib/settings.ts
@@ -1,11 +1,30 @@
+// THE ONE READER AND THE ONE WRITER OF settings.json.
+//
+// The shape — every field, its default, its clamp, its documentation — is
+// `siteSettingsSchema` in ./settingsSchema.ts (one-core phase 3 slice 4a), and
+// everything that module exports is re-exported here, so the ~200 importers of
+// `lib/settings` (types, constants, clamps, sanitizers) did not move.
+//
+// What is left in this file is exactly what a schema cannot do:
+//
+// - READ STAYS LENIENT. A settings.json is whatever an operator, an older
+// build, or a half-finished write left behind. getSettings never throws:
+// an unreadable or non-object file reads as the empty one, and every field
+// is total. Three migrations are keyed on a field's ABSENCE in the raw file
+// — which a parsed object cannot see, because parsing folds the default in
+// — so they run around the parse, each handed the RAW object.
+// - WRITE STAYS STRICT. writeSettings derives the worker shadow, runs the two
+// validators that THROW (a worker list that cannot transcribe, a social
+// link whose SVG is unsafe), parses through the same schema — which is what
+// drops every key it does not name, retired fields included — and writes
+// atomically (tmp + rename).
+//
+// One schema, both directions: the only differences between what a read and a
+// write produce are those migrations and those two validators.
+
import fs from "node:fs";
import path from "node:path";
import { getPaths } from "./paths";
-import { SYNC_INTERVAL_MAX_MINUTES } from "./channelConfig";
-import {
- isDownloadFormatPreset,
- type DownloadFormatPreset,
-} from "../ytdlp/downloadFormat";
import {
type AppInstanceConfig,
DEFAULT_TRANSCRIBE_ARGS,
@@ -13,1587 +32,98 @@ import {
TRANSCRIPTION_APPS,
} from "./transcriptionApps";
import {
- type Worker,
defaultWorkersFromApps,
- sanitizeWorkerConfig,
sanitizeWorkers,
validateWorkers,
} from "./workers";
-import type { AutoQueueSettings } from "./autoQueueTypes";
-import {
- defaultChannelPriority,
- sanitizeChannelPriority,
- type ChannelPriority,
-} from "./channelPriority";
import { migrateSweepsToLanes } from "./laneMigration";
+import { migrateMediaRootToLocations } from "./storageLocations";
import {
- INTERNAL_LOCATION_ID,
- migrateMediaRootToLocations,
- type StorageLocation,
- type StorageSettings,
- type StorageVolume,
-} from "./storageLocations";
-// The four SANITIZERS still come from the engine. They are the auto-queue's
-// half of the settings schema and belong in lib/ with the rest of it, but that
-// move is phase 3 slice 4 (one schema, one writer) — not a rename. Recorded in
-// ../architecture.test.ts's allow-list until then.
-import {
- defaultAutoQueue,
- sanitizeAutoQueue,
-} from "../jobs/autoQueuePolicy";
-import {
- DEFAULT_DIARIZATION_ENGINE,
- DEFAULT_DIARIZATION_THRESHOLD,
- DIARIZATION_BACKENDS,
- DIARIZATION_ENGINE_IDS,
- type DiarizationBackend,
- type DiarizationEngineId,
-} from "./diarization";
-import { ATTRIBUTION_PROMPT_VERSION } from "./attribution";
-import {
- DEFAULT_COOKIE_MODE,
- isCookieMode,
- type CookieMode,
-} from "./cookiePolicy";
-// From the CLIENT-SAFE digest module, deliberately — digestApps.ts imports execa,
-// and settings.ts must stay reachable from anywhere.
-import {
- CLAUDE_DIGEST_APP_ID,
- DEFAULT_DIGEST_APP_ID,
- DEFAULT_DIGEST_TIMESTAMP_MODE,
- DIGEST_SECTION_KINDS,
- DIGEST_TIMESTAMP_MODES,
- isDigestSectionKind,
- isDigestTimestampMode,
- type DigestAppConfig,
- type DigestSectionKind,
- type DigestTimestampMode,
-} from "./digest";
-
-export type { Worker } from "./workers";
-export type { AutoQueueSettings } from "./autoQueueTypes";
-export type { ChannelPriority } from "./channelPriority";
-
-// Transcribe placeholder/arg helpers now live with the whisper-cpp app in
-// transcriptionApps.ts. Re-exported here so existing import sites keep working.
-export {
- type AppInstanceConfig,
- TRANSCRIBE_PLACEHOLDER_AUDIO,
- TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE,
- TRANSCRIBE_PLACEHOLDER_MODEL,
- TRANSCRIBE_KNOWN_PLACEHOLDERS,
- DEFAULT_TRANSCRIBE_ARGS,
- validateTranscribeArgs,
-} from "./transcriptionApps";
-
-// Global, OPERATIONAL settings shared across every site this editor powers.
-// Per-site presentation (branding, social links, channel groups, membership)
-// lives in sites/<siteId>/site.json — see common/lib/site.ts.
-export type SiteSettings = {
- // Title for the EDITOR admin shell only (the editor manages all sites and so
- // is not tied to any one site's branding). Public sites get their own titles
- // from site.json.
- adminTitle: string;
- maxTranscriptPageBytes: number;
- // Active transcription app id (key into TRANSCRIPTION_APPS, e.g. "whisper-cpp"
- // or "chough"). Selected globally; see common/lib/transcriptionApps.ts.
- transcriptionApp: string;
- // Per-app configuration, keyed by app id. Each app reads only its own block;
- // a missing block means "use the app's defaults". DEPRECATED in favor of
- // `workers` (each local worker carries its own config); kept one release to
- // drive migration and allow rollback. See common/lib/workers.ts.
- transcriptionApps: Record<string, AppInstanceConfig>;
- // Configured transcription workers (named processing slots). The scheduler
- // distributes each video to the highest-priority free worker. A settings.json
- // predating this field is migrated to a single enabled worker from the active
- // app (see defaultWorkersFromApps). See common/lib/workers.ts.
- workers: Worker[];
- // Browser spec (e.g. "firefox", "chrome:Default") passed to
- // `yt-dlp --cookies-from-browser`. WHEN it is passed is governed by
- // `cookieMode` below. Empty string = no cookies configured. Per-channel
- // override available (ChannelConfig.cookiesFromBrowser).
- cookiesFromBrowser: string;
- // How yt-dlp invocations use the configured cookies (see
- // common/lib/cookiePolicy.ts): "always" passes them on every invocation,
- // "when-required" (default; the historical behavior) only to retry an
- // auth/age failure, "defer" never in normal runs — auth-gated videos are
- // excluded from batches and collected into the per-channel "Needs cookies"
- // bucket for a manual cookie run. Per-channel override available
- // (ChannelConfig.cookieMode).
- cookieMode: CookieMode;
- // Pause (seconds) inserted between per-video yt-dlp invocations in
- // managed batch downloads. yt-dlp's own `-t sleep` only paces requests
- // within one invocation, so without this the managed loop hammers the
- // source IP back-to-back. 0 disables. Per-channel override available.
- sleepBetweenDownloadsSeconds: number;
- // Default yt-dlp `-f` download format for every channel that doesn't set its
- // own (ChannelConfig.downloadFormat). "auto" picks per-source: `original` for
- // Odysee (whose HLS rungs are CDN-truncated), `bestaudio/worst` elsewhere. See
- // common/ytdlp/downloadFormat.ts.
- downloadFormat: DownloadFormatPreset;
- // Minimum free disk space (GB) required on the transcripts data directory for
- // downloads to run. When free space is below this floor, a download job is
- // prevented from starting and a running batch stops launching new videos
- // (the in-flight one finishes). 0 disables the gate. See common/lib/diskSpace.ts.
- minFreeDiskGB: number;
- // Extra headroom (GB) above minFreeDiskGB that a stopped pipeline must see
- // before it resumes. Resuming at the same number we stopped at flaps — the
- // first restarted download pushes free space back under the floor. This is
- // the hysteresis margin, so "resumed" means the operator actually freed
- // something rather than a scratch file being cleaned up. 0 disables the
- // hysteresis (resume at the floor). See diskGate() in common/lib/diskSpace.ts.
- resumeMarginGB: number;
- // Default number of videos transcribed in parallel when a "Transcribe
- // missing" / bucket run doesn't specify its own concurrency. The per-run
- // Concurrency input in the channel UI overrides this for a single run.
- parallelTranscriptions: number;
- // When true, the no-subs fallback in the managed downloader runs whisper
- // inline immediately after the audio download succeeds. When false
- // (default), audio is left for the next "Transcribe missing" pass so a
- // batch download finishes faster and whisper can parallelize.
- inlineTranscribeOnFallback: boolean;
- // When true (default), managed downloads skip videos that are currently live
- // or scheduled/upcoming, decided from a metadata-only prefetch pass. Finished
- // livestream VODs (was_live) are NOT skipped and download normally. A skip is
- // recorded but not archived, so the next sync/download-missing retries the
- // video once the stream ends. Per-channel override available
- // (ChannelConfig.skipLiveDownloads).
- skipLiveDownloads: boolean;
- // Whether the transcribed-audio cleanup sweep checks each candidate is still
- // available upstream before deleting its audio, pinning (do-not-clean) any
- // video found permanently gone. The delete is irreversible and a gone video's
- // audio is irreplaceable, so this defaults to true. Turn it off for an offline
- // or URL-less setup, where the check can never resolve and cleanup would
- // otherwise never delete anything. See verifyBeforeClean.ts.
- verifyAvailabilityBeforeClean: boolean;
- // Whether site builds generate downloadable transcript/live-chat archive zips
- // (into public/archives, linked on the Downloads page). Global default; a site
- // can opt out via site.json `archives: false`, and a single build can skip via
- // the "Skip archive zips" build control. Opt-out: default true.
- buildArchives: boolean;
- // Overflow object storage (Cloudflare R2) for archive zips that exceed the
- // Pages per-file size cap (see Site.archiveMaxBytes). When both fields are set,
- // an oversize archive is uploaded here on deploy — via `wrangler r2 object put`,
- // keyed `<siteId>/archives/<file>` — instead of being dropped, and the Downloads
- // page links to `<publicBaseUrl>/<key>`. Blank/absent → no overflow, so oversize
- // archives stay unavailable ("Too large to host").
- archiveStorage?: { bucket: string; publicBaseUrl: string };
- // Debounce preset for the global snapshot scheduler: how long it waits after
- // the last report-changing action before regenerating affected channel
- // reports. See REPORT_DEBOUNCE_PRESETS. Default "fast" (~1s, no cap).
- reportDebouncePreset: ReportDebouncePreset;
- // How often (seconds) the editor UI passively re-fetches the current page's
- // server-rendered data via router.refresh(), so sidebar badges and reports
- // stay live without a manual reload. Mounted globally; pauses while the tab is
- // hidden. 0 disables passive refresh entirely. See AUTO_REFRESH_INTERVAL_*.
- autoRefreshIntervalSeconds: number;
- // Global configuration for the scheduled (cron-driven) channel sync system.
- // The per-channel cadence lives on ChannelConfig.syncIntervalMinutes; this
- // block holds the defaults and guard rails the scheduler applies across all
- // channels. See common/jobs/syncScheduler.ts.
- syncScheduler: SyncSchedulerSettings;
- // Configuration for the automatic priority-queue runners (auto-transcribe /
- // auto-download). Each holds a tree policy that decides which channel's video
- // to process next, cross-channel, by priority/round-robin/weighted-fair rules.
- // Independent of syncScheduler (which decides staleness, not work order). See
- // common/jobs/autoQueuePolicy.ts.
- autoQueue: AutoQueueSettings;
- // THE OPERATOR-FACING PRIORITY MODEL: one tier per channel plus one
- // corpus-wide focus selector. It is the SOURCE the four `autoQueue[lane].root`
- // trees are compiled from (common/lib/channelPriority.ts), not a second
- // mechanism beside them — and its `paused` tier is the one part that is not a
- // tree shape, filtering the runner's channel list instead. An empty document
- // (the default) is today's behaviour exactly: no focus, every channel normal,
- // the stored trees stand.
- channelPriority: ChannelPriority;
- // Default social links applied to every site that doesn't define its own.
- // A site inherits these unless its site.json carries an explicit
- // `socialLinks` array — see Site.socialLinks / resolveSocialLinks in
- // common/lib/site.ts. The one presentation field that lives globally so a
- // shared footer doesn't have to be repeated per site.
- socialLinks: SocialLink[];
- // Absolute public URL of the family hub/homepage (e.g. "https://archilyzer.pages.dev").
- // Every export site links back to it ("the family" backlink) when set. Empty =
- // no hub link rendered. Normalized to a trailing-slash-free http(s) URL.
- homepageUrl: string;
- // Backup configuration for the saved-video store (Phase 4 of the
- // video-persistence feature). When enabled with a destination, the store is
- // mirrored there (additively, no deletes) with a per-backup manifest, and the
- // sync scheduler runs the backup on the configured cadence. See
- // common/controller/backupSavedVideos.ts.
- savedVideoBackup: SavedVideoBackupSettings;
- // Where a channel's downloaded media goes when it is relocated off the corpus
- // disk. A DEFAULT ONLY: the relocate controller never reads it and always
- // takes an explicit root, so this is the value the per-channel Storage panel
- // prefills and the /channels bulk move falls back to. Blank = no default.
- // See StorageSettings.
- storage: StorageSettings;
- // How the static export is built: "basic" reuses the single export/ tree and
- // serializes builds on one queue (the long-standing behavior); "docker" runs
- // each site's build in an isolated container for safe parallelism. The Docker
- // pipeline itself is a follow-up; this block persists the chosen mode plus the
- // container/concurrency knobs the deploy page and the future orchestrator read.
- buildPipeline: BuildPipelineSettings;
- // AI digest generation (chapters + topic tags over the existing transcripts).
- // Local-first: the metered lane is off by default. See DigestSettings.
- digest: DigestSettings;
- // Speaker diarization captured right after transcription, while the audio is
- // still on disk. OFF by default. See DiarizationSettings.
- diarization: DiarizationSettings;
- // The generic catch-up lane for derived data the existing corpus predates.
- // OFF by default, and idle-only when on. See BackfillSettings.
- backfill: BackfillSettings;
- // Naming the speakers diarization found (or reconstructing them from the
- // transcript when it found none). OFF by default. See AttributionSettings.
- attribution: AttributionSettings;
-};
-
-// Configuration for speaker attribution — putting names to the speaker turns.
-//
-// OFF by default, and that default is doing real work rather than being
-// cautious. The text-only lane costs roughly one model call per transcript
-// CHUNK, which on this corpus is ~194,000 calls, the same order as the digest
-// sweep — and the digest sweep has completed 0.17% of its own. Arming both at
-// once puts two multi-week lanes on the same 8 GB card with nothing to arbitrate
-// between them (the backfill lane's yield deliberately watches only the
-// transcription lane). Nothing here arms anything; a pilot decides whether the
-// corpus-wide text-only pass is worth 25-55 GPU-days at all.
-export type AttributionSettings = {
- // Master switch. Off means the backfill registry reports no attribution work
- // at all — the feature gate every Operation has.
- enabled: boolean;
- // Which digest app runs the naming. Attribution IS a digest-app workload —
- // constrained JSON decoding over transcript text — so it reuses that registry
- // and that per-app config (settings.digest.apps[appId]) rather than growing a
- // second copy of the ollama URL, context size and timeout.
- appId: string;
- // Model override. Empty = the app's configured model, then its default. It is
- // separate from the digest's because the two workloads may want different
- // sizes, and because it is part of the freshness identity: sharing the digest's
- // model field would make a digest bake-off invalidate every attribution record
- // on disk as a side effect.
- model: string;
- // The lanes, separately. Both default OFF even when `enabled` is on, so
- // turning the feature on to look at it cannot start a corpus sweep.
- //
- // They are not a fallback pair. `diarized` is one call per video and grounded
- // in acoustic clustering; `textOnly` is ~30 calls and guesses at identity
- // across chunk seams. An operator may reasonably want the first forever and
- // the second never.
- diarizedEnabled: boolean;
- textOnlyEnabled: boolean;
- // The prompt generation a record must match to count as fresh.
- //
- // Defaults to (and is floored at) ATTRIBUTION_PROMPT_VERSION, the shipped
- // constant. Raising it forces a corpus-wide regeneration without a code
- // change, which is the honest way to redo everything after a prompt tweak.
- // It cannot be set BELOW the shipped constant, and that floor is the lesson
- // from digestPrompt.ts's version 1 -> 2 note: pinning freshness to an older
- // generation freezes output from a superseded prompt into the corpus, looking
- // identical to output from the current one.
- promptVersion: number;
-};
-
-// Configuration for the backfill lane — the generic answer to "a derived-data
-// feature landed and 77,000 existing videos do not have it".
-//
-// WHAT THE LANE GETS is its own queueKey (concurrency with transcription) plus a
-// limit() that returns 0 to stand aside — the same mechanism the digest yield
-// uses, which fails OPEN so a bad read costs contention rather than a deadlock.
-// There is no priority system to join: the registry submits every named queue at
-// concurrency 1 and SchedulerTier only orders work within a single key.
-//
-// THE SHARE IS `concurrency` AND THE LANE'S `autoQueue.backfill.maxWorkers`; the
-// yield is the operation's declared `contendsFor`. Slice 1.3 retired the
-// `weight` scalar that used to mean both — see backfillLimit().
-export type BackfillSettings = {
- // Slots the lane may use when it is not standing aside. Kept at 1 by default
- // for the same reason diarization.concurrency is: this is CPU-bound work
- // competing with GPU feeding and the digest sweep for the same 8 threads.
- concurrency: number;
- // Re-acquire media for videos whose input is GONE (audio deleted after
- // transcription). OFF by default and deliberately so: measured on this corpus,
- // 836 videos still have media and ~76,270 would need a re-download — 91x the
- // reachable work, against 45 GB free at 97% full. When on, each re-fetched
- // file is removed in a `finally` as soon as the backfill has used it, unless
- // the video is marked do-not-clean, or unless the auto-transcribe policy would
- // replace its auto-captions (`replaceAutoSubs`, or a leaf on
- // `downloadedAutoSubsOnly`), in which case the audio is kept for that runner.
- //
- // WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: "youtube"`
- // channel — which normally only fetches subtitles — the re-acquire applies a
- // PER-VIDEO transcribe override so yt-dlp lands audio a diarizer can read;
- // the channel's stored config is not changed. Without that override the fetch
- // re-downloads the captions the video already has and lands nothing, which is
- // what happened to ~16,000 videos on eight channels in 2026-08.
- allowRedownload: boolean;
-};
-
-// Configuration for the speaker-diarization capture lane.
-//
-// This exists because AUDIO IS THE ONE PERISHABLE INPUT in the pipeline:
-// cleanAudioFromTranscribed deletes it once a video is transcribed, so
-// diarization has to happen while the audio is still there or not at all. The
-// capture half is deliberately all that ships here — attribution, LLM speaker
-// naming, viewer badges and quote filtering can all be redone later from the
-// saved JSON, whereas the audio cannot.
-export type DiarizationSettings = {
- // Master switch. OFF by default so a transcription batch can start before this
- // lands, with diarization backfilled over the retained audio afterwards.
- //
- // Turning it ON also arms the cleanup guard: the Clean-audio sweep stops
- // deleting audio for a transcribed video that has no diarization.json yet.
- // That is the point — it is what keeps the perishable input alive long enough
- // to be captured — but it means enabling this holds disk.
- enabled: boolean;
- // Run diarization inline in the post-transcribe hook.
- //
- // OFF by default, and that default is a MEASURED decision, not caution.
- // Measured on this box: GPU transcription runs at 221 s/audio-hour (16.3x
- // realtime, over 3,602 real videos), CPU diarization at ~500-680 s/audio-hour.
- // Diarization is therefore ~2-3x SLOWER than the transcription it follows, so
- // running it inline drops whole-pipeline throughput by roughly 3-4x and leaves
- // the GPU idle while the CPU catches up.
- //
- // The intended sequence for a large batch is the opposite: leave this off, let
- // the batch transcribe at full GPU speed with `enabled` holding the audio, and
- // diarize afterwards with the backfill pass. Turn it on for steady state, once
- // the arrival rate is a few videos a day rather than a corpus.
- inlineAfterTranscribe: boolean;
- // Clustering threshold — the single most consequential knob, since it decides
- // how many speakers come out. Larger merges more aggressively.
- //
- // The default is 0.9, NOT sherpa-onnx's own 0.5, and that is measured on this
- // corpus. On a 6-minute excerpt of a two-person interview (known ground truth:
- // 2 speakers), sherpa's default produced 22 clusters; 0.9 produced 6, with the
- // top two at 40%/40% of talk time — recognizably the two hosts. Sweep on the
- // same clip: 0.4→23, 0.5→22, 0.6→17, 0.7→12, 0.8→10, 0.9→6.
- //
- // It still over-splits, which is why this is a capture lane and not an answer:
- // the turns are recorded with the threshold that produced them, so a later
- // attribution pass can re-cluster or re-run without needing the audio back.
- threshold: number;
- // Engine threads per diarize run.
- threads: number;
- // Which engine runs. "sherpa-onnx" is the shipped default and what every
- // sidecar on disk was produced by; "sortformer" is the ggml engine built by
- // scripts/build-sortformer.sh.
- //
- // CHANGING THIS RESTATES THE FRESHNESS IDENTITY (see diarizationTarget), so
- // every sidecar written by the other engine becomes stale and the backfill lane
- // offers to redo it. That is intended — the two disagree about how many
- // speakers exist, and a corpus half-diarized by each is not one corpus — but on
- // the retained audio it is weeks of work, not a toggle.
- //
- // Why anyone would: on the same file, sherpa at its tuned threshold returns 13
- // speakers and sortformer returns 4, agreeing on the dominant speaker's share
- // to within half a point (73.1% vs 73.5%). On the corpus's worst case sherpa
- // returns 35 and sortformer 4. Over-splitting is the failure mode this lane has
- // always had, and sortformer is end-to-end rather than clustered, so it does
- // not have it. The cost is a hard ceiling of 4 speakers and ~1.8x the wall
- // clock.
- engine: DiarizationEngineId;
- // Compute device for the sortformer engine; ignored by sherpa-onnx, which has
- // no Vulkan compute path on Linux.
- //
- // "vulkan" is 1.5x faster than a thread-tuned CPU run (894 vs 1305
- // s/audio-hour, measured on this box) and holds 558 MB resident instead of
- // 4.84 GB by keeping weights and activations in VRAM. It also takes ~4.4 GB of
- // an 8 GB card, which is why the lane YIELDS to transcription rather than
- // sharing — see controller/digestYield.ts.
- backend: DiarizationBackend;
- // Python interpreter for the default sherpa-onnx engine. sherpa-onnx ships
- // wheels only up to cp313, and this box's system python is 3.14 — so this
- // usually points at a dedicated venv rather than `python3`.
- python: string;
- // ONNX model paths for the default engine. Empty = the lane cannot run, which
- // is reported as a skip rather than a failure.
- segModel: string;
- embModel: string;
- // Binary and model for the sortformer engine, both produced by
- // scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the
- // same "not-configured" skip as an unset segModel/embModel.
- sortformerBin: string;
- sortformerModel: string;
- // How many diarize runs may execute at once in the backfill pass. Kept low by
- // default: diarization is CPU-bound and competes with GPU feeding and the
- // digest sweep for the same 8 threads.
- concurrency: number;
- // Videos longer than this are DEFERRED rather than diarized: reported as a
- // third number that is never summed into reachable work, so a capped corpus
- // can never read as finished.
- //
- // THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering holds a
- // pairwise distance matrix over speech-segment embeddings — O(n^2) in SEGMENT
- // count — and speaker-turn density varies 40x across this corpus (33-1364
- // turns/hour), so duration does not actually predict the blowup: a sparse
- // 7h42m video completed while a dense 6h12m one was OOM-killed. Duration is
- // merely the only predictor available for free, from metadata already on disk,
- // BEFORE spending 45 minutes to find out. n^2 at 30k segments is 6.7 GiB and
- // at 40k is 11.9 GiB, which brackets the 10.6 GB and 9.6 GB peaks measured on
- // this 16 GB box.
- //
- // 0 disables the cap. That is where this goes once windowed diarization lands:
- // windowing divides per-window n by the window count, so the matrix falls by
- // its square, and the cap stops being needed rather than being tuned.
- maxAudioHours: number;
-};
-
-// Configuration for the derived-corpus digest layer. Local-first by decision:
-// `remoteEnabled` gates the metered lane and defaults to false, so nothing here
-// can spend money until it is explicitly turned on.
-export type DigestSettings = {
- // Master switch for the metered (remote-api) lane. OFF by default — an opt-in
- // overflow for the long tail or a channel where local quality is poor, never
- // the default path.
- remoteEnabled: boolean;
- // Videos longer than this are "long tail": 8.2% of the corpus by count, 46% of
- // all transcript tokens. The batch's duration-aware ordering and the optional
- // remote overflow both key off it.
- longTailSeconds: number;
- // The engine each lane uses (ids from common/lib/digestApps.ts).
- localAppId: string;
- remoteAppId: string;
- // Per-app config, keyed by app id — the same id-keyed sub-record shape as
- // transcriptionApps.
- apps: Record<string, DigestAppConfig>;
- // Yield the GPU to the transcription lane: while transcription is working, the
- // digest batch's limit() returns 0 and the pool idle-waits. ON by default,
- // because `digest:local` is deliberately on a different queue from
- // TRANSCRIPTION_QUEUE and so would otherwise run ollama and the transcription
- // engine on the same 8 GB card. See controller/digestYield.ts.
- yieldToTranscription: boolean;
- // Whether a busy worker pinned to `device: "cpu"` counts as GPU contention.
- //
- // OFF by default, which is the FIX for a real bug: the yield originally tested
- // only `kind === "local"`, so on a box with one GPU worker and two CPU-pinned
- // ones (this box, at parallelTranscriptions 2) the digest lane stopped dead for
- // transcription that competes for zero GPU shaders.
- //
- // Only an EXPLICIT "cpu" is treated as non-contending. A worker with no device
- // set is using the engine binary's own default, which may be the GPU, so it
- // still triggers the yield — the unknown case fails safe.
- //
- // Composes with `yieldToTranscription`: that is the master switch, this only
- // narrows which workers it reacts to.
- yieldToCpuWorkers: boolean;
- // Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. Only ever
- // consulted for a metered app.
- spendCapUsd: number;
- // Which sections a sweep generates.
- //
- // Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, and that
- // is not a contradiction: a tag call sends the same transcript as the chapter
- // call before it, so it hits the engine's cached prefix and pays essentially no
- // prefill (+0.4 s across 4 extra calls, against 22.4 s for the first 4). All it
- // pays is decode, and a tag list is ~30 output tokens where a chapter list is
- // ~200–290.
- //
- // The corollary matters more than the number: run them in the SAME pass. Tags
- // generated later, on their own, pay full prefill again — measured at 44% of a
- // whole chapters pass, i.e. 3–9× the marginal cost of just including them now.
- sections: DigestSectionKind[];
- // How each chunk's transcript markers are numbered — see DigestTimestampMode.
- // Was a scored variable in the bake-off rather than a pre-applied fix; the
- // measurement is in and "chunk-local" is now the shipped default.
- timestampMode: DigestTimestampMode;
- // A free-text label for a non-default prompt shape, folded into the recorded
- // provenance by digestPromptVariant(). Setting it invalidates every digest
- // generated under a different label, which is exactly what makes a bake-off
- // round re-run its sample instead of skipping it as fresh. Empty = default.
- promptVariant: string;
-};
-
-// "basic" — `pnpm run build` in export/, serialized on the build queue (shared
-// output tree → no safe parallelism).
-// "docker" — isolated per-site container builds (follow-up); enables real
-// parallel multi-site builds capped by maxParallelBuilds.
-export type BuildMode = "basic" | "docker";
-
-export type BuildPipelineSettings = {
- mode: BuildMode;
- // Cap on concurrent per-site container builds in docker mode. Ignored in basic
- // mode (which is always serial). Clamped to [1, BUILD_MAX_PARALLEL_MAX].
- maxParallelBuilds: number;
- // Tag of the reusable build image (built once, reused for every site).
- dockerImage: string;
- // Dockerfile path relative to the monorepo root, used to (re)build the image.
- dockerfile: string;
-};
-
-export type SavedVideoBackupSettings = {
- // Master switch for the scheduled backup. A backup can still be run manually
- // when this is false, as long as a destination is set.
- enabled: boolean;
- // Destination root the store is mirrored into (a local path or any rsync
- // target). Empty disables both scheduled and manual backups.
- dest: string;
- // Cadence (minutes) for the scheduled backup when enabled. Clamped into the
- // sync-interval window; default daily.
- intervalMinutes: number;
-};
-
-export type SyncSchedulerSettings = {
- // Master switch. When false, a tick selects nothing (manual sync still works).
- enabled: boolean;
- // Fallback cadence (minutes) for channels with no per-channel override.
- defaultIntervalMinutes: number;
- // Cap on sync jobs running/queued at once. A tick queues at most
- // (cap - currently-active) channels; the rest roll to the next tick. This is
- // also the stagger mechanism that keeps a big due-batch from hitting the
- // source all at once.
- maxConcurrentSyncs: number;
- // Optional local-clock quiet window during which auto-sync is suppressed.
- // Both null = always allowed. The window may wrap past midnight
- // (e.g. start=22, end=6). Hours are [0,23]; the window is [start, end).
- quietHoursStart: number | null;
- quietHoursEnd: number | null;
- // Failure backoff bounds. After N consecutive failed scheduled syncs a
- // channel waits min(base * 2^(N-1), max) minutes before it's eligible again.
- backoffBaseMinutes: number;
- backoffMaxMinutes: number;
- // Cadence (seconds) for the editor's in-process heartbeat — the internal timer
- // armed by the instrumentation hook (editor/instrumentation.ts) that calls the
- // scheduler tick directly, so no external cron is needed. 0 = off: rely on the
- // external `pnpm sync:tick` heartbeat instead. Any positive value is clamped to
- // [SYNC_HEARTBEAT_MIN_SECONDS, SYNC_HEARTBEAT_MAX_SECONDS]. The env var
- // SYNC_HEARTBEAT_SECONDS overrides this at runtime. See SCHEDULED_SYNC.md.
- heartbeatSeconds: number;
- // Cadence (minutes) for the scheduled keep-latest deletion check. For each
- // channel with ChannelConfig.keepLatest > 0, the tick re-probes the kept
- // window for source deletion (checkKeptDeletedAction) at most this often and
- // pins any gone videos as do-not-clean. Clamped into the sync-interval window;
- // default daily. The check shares the same concurrency cap and quiet-hours
- // window as scheduled syncs. See editor/app/scheduler/runTick.ts.
- keepLatestCheckIntervalMinutes: number;
- // Default cadence (minutes) for the sync FULL SWEEP — the deep pass that
- // re-enumerates a channel's whole listing in one yt-dlp spawn, refreshes the
- // stored `playlist` file, and flags videos that have left the listing into
- // maybe-missing.json. Ordinary syncs stay on the cheap newest-first paged
- // walk; a sync only upgrades itself to a sweep when this interval has elapsed
- // since the channel's lastFullSweepAt. Per-channel override:
- // ChannelConfig.fullSweepIntervalMinutes. 0 = never sweep. Default daily.
- // See common/jobs/deepSync.ts.
- fullSweepIntervalMinutes: number;
- // Upper bound on how many maybe-missing suspects a full sweep will resolve
- // in-line with the per-video availability probe (deleted vs private vs
- // unlisted). At or under the cap the sweep runs the targeted check itself, so
- // "Sync all" surfaces upstream deletions with no extra clicks; over it, the
- // suspects are flagged and left for a manual check rather than firing hundreds
- // of probes inside a sync. 0 = never auto-confirm.
- fullSweepConfirmMaxSuspects: number;
- // Shrink guard: how far a fresh listing may fall below the stored one before
- // it is treated as suspect rather than acted on. Expressed as a percentage of
- // the previous count, floored at SHRINK_ABS_FLOOR entries so ordinary churn on
- // a small channel doesn't trip it. A suspect listing does not rewrite
- // `playlist` or maybe-missing.json and does not count as a sweep — but a
- // SECOND enumeration reporting a similar count confirms it and is accepted, so
- // a genuine mass deletion costs at most one cadence period. 0 = off (the
- // empty-listing rejection still applies). See controller/acceptListing.ts.
- fullSweepShrinkGuardPercent: number;
-};
-
-export type SocialLink = {
- label: string;
- url: string;
- svg: string;
-};
-
-export const SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS = 600;
-export const SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS = 10;
-
-export const MIN_FREE_DISK_GB_DEFAULT = 5;
-export const MIN_FREE_DISK_GB_MAX = 100000;
-
-// Hysteresis margin for the low-disk gate. 2 GB is deliberately larger than any
-// single scratch file the pipeline writes, so cleaning one up cannot by itself
-// reopen the gate.
-export const RESUME_MARGIN_GB_DEFAULT = 2;
-export const RESUME_MARGIN_GB_MAX = 1000;
-
-export const PARALLEL_TRANSCRIPTIONS_MAX = 16;
-export const PARALLEL_TRANSCRIPTIONS_DEFAULT = 2;
-
-// Passive UI auto-refresh cadence. 0 is a sentinel meaning "disabled"; any other
-// value is clamped into [MIN, MAX] seconds.
-export const AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS = 5;
-export const AUTO_REFRESH_INTERVAL_MIN_SECONDS = 1;
-export const AUTO_REFRESH_INTERVAL_MAX_SECONDS = 600;
-
-// Global snapshot-scheduler debounce presets. `debounceMs` is the quiet-period
-// window after the last report-changing action; `maxWaitMs` caps the total
-// delay under continuous activity (null = no cap, fire purely on the quiet
-// period). Consumed by common/jobs/snapshotScheduler.ts and surfaced in the
-// Settings form.
-export type ReportDebouncePreset = "fast" | "balanced" | "lazy";
-
-export const REPORT_DEBOUNCE_PRESETS: Record<
- ReportDebouncePreset,
- { debounceMs: number; maxWaitMs: number | null }
-> = {
- fast: { debounceMs: 1000, maxWaitMs: null },
- balanced: { debounceMs: 3000, maxWaitMs: 30000 },
- lazy: { debounceMs: 10000, maxWaitMs: 60000 },
-};
-
-export const DEFAULT_REPORT_DEBOUNCE_PRESET: ReportDebouncePreset = "fast";
-
-export function isReportDebouncePreset(v: unknown): v is ReportDebouncePreset {
- return v === "fast" || v === "balanced" || v === "lazy";
-}
-
-export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024;
-export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024;
-export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024;
-
-export const DEFAULT_ADMIN_TITLE = "Transcript Browser Admin";
-
-// Sync-scheduler bounds + defaults. Default cadence is daily; concurrency is
-// conservative so a tick doesn't fan out into the source provider all at once.
-export const SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES = 1440;
-export const SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT = 2;
-export const SYNC_SCHEDULER_MAX_CONCURRENT_MAX = 16;
-export const SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES = 30;
-export const SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES = 1440;
-export const KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES = 1440;
-// Full-sweep defaults. Daily: a sweep is one full enumeration of the channel,
-// far more expensive than the 50-entry page an ordinary sync fetches. The
-// confirm cap keeps an unattended sweep from fanning out into hundreds of
-// per-video probes when a channel's listing changes wholesale.
-export const FULL_SWEEP_DEFAULT_INTERVAL_MINUTES = 1440;
-export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT = 25;
-export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX = 10000;
-// Shrink-guard default: a listing that has lost more than a tenth of its
-// entries (and more than SHRINK_ABS_FLOOR of them) needs a second opinion.
-export const FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT = 10;
-export const FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX = 100;
-export const SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES = 1440;
-
-// Internal-heartbeat cadence bounds. 0 means "off" (use an external cron
-// heartbeat); any other value is clamped into [MIN, MAX] seconds. The floor
-// keeps the in-process timer from busy-looping; the ceiling is one hour.
-export const SYNC_HEARTBEAT_DEFAULT_SECONDS = 0;
-export const SYNC_HEARTBEAT_MIN_SECONDS = 15;
-export const SYNC_HEARTBEAT_MAX_SECONDS = 3600;
-
-export function defaultSyncScheduler(): SyncSchedulerSettings {
- return {
- enabled: false,
- defaultIntervalMinutes: SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES,
- maxConcurrentSyncs: SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT,
- quietHoursStart: null,
- quietHoursEnd: null,
- backoffBaseMinutes: SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES,
- backoffMaxMinutes: SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES,
- heartbeatSeconds: SYNC_HEARTBEAT_DEFAULT_SECONDS,
- keepLatestCheckIntervalMinutes: KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES,
- fullSweepIntervalMinutes: FULL_SWEEP_DEFAULT_INTERVAL_MINUTES,
- fullSweepConfirmMaxSuspects: FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT,
- fullSweepShrinkGuardPercent: FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT,
- };
-}
-
-// Clamp an internal-heartbeat cadence: 0 (off) passes through; any positive
-// value is clamped up into [MIN, MAX]; junk falls back to the default.
-export function clampHeartbeatSeconds(value: unknown): number {
- if (typeof value !== "number" || !Number.isFinite(value)) {
- return SYNC_HEARTBEAT_DEFAULT_SECONDS;
+ clampParallelTranscriptions,
+ defaultStorage,
+ normalizeSocialSvg,
+ parseSocialLinks,
+ sanitizeTranscriptionApps,
+ siteSettingsSchema,
+ type SiteSettings,
+ type SocialLink,
+} from "./settingsSchema";
+
+export * from "./settingsSchema";
+
+type RawSettings = Record<string, unknown>;
+
+// The file as JSON, or `undefined` when it is missing or not JSON. Never
+// throws: a settings read is on every request path.
+function readRawSettings(file: string): unknown {
+ try {
+ return JSON.parse(fs.readFileSync(file, "utf8"));
+ } catch {
+ return undefined;
}
- const n = Math.floor(value);
- if (n <= 0) return 0;
- if (n < SYNC_HEARTBEAT_MIN_SECONDS) return SYNC_HEARTBEAT_MIN_SECONDS;
- if (n > SYNC_HEARTBEAT_MAX_SECONDS) return SYNC_HEARTBEAT_MAX_SECONDS;
- return n;
-}
-
-function clampHourOrNull(value: unknown): number | null {
- if (typeof value !== "number" || !Number.isFinite(value)) return null;
- const n = Math.floor(value);
- if (n < 0 || n > 23) return null;
- return n;
-}
-
-// Like clampPositiveInt, but 0 survives as a sentinel ("off"/"never"). Used by
-// the cadences whose disabled state is expressed as a zero rather than a
-// separate boolean.
-function clampIntAllowZero(value: unknown, fallback: number, max: number): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? Math.floor(value)
- : fallback;
- if (n <= 0) return 0;
- if (n > max) return max;
- return n;
}
-function clampPositiveInt(value: unknown, fallback: number, max: number): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? Math.floor(value)
- : fallback;
- if (n < 1) return 1;
- if (n > max) return max;
- return n;
+// Only a plain object is a settings file. `null`, `[]`, `3` and a truncated
+// write all read as the empty file — every field its default — rather than as
+// an exception. (Before slice 4a a file containing `null` threw here.)
+function rawObject(raw: unknown): RawSettings {
+ return raw && typeof raw === "object" && !Array.isArray(raw)
+ ? (raw as RawSettings)
+ : {};
}
-// Coerce a raw settings.syncScheduler value into a clean SyncSchedulerSettings,
-// falling back to defaults for missing/ill-typed fields. Quiet hours are only
-// honored when BOTH endpoints are valid hours; otherwise the window is cleared.
-export function sanitizeSyncScheduler(value: unknown): SyncSchedulerSettings {
- const d = defaultSyncScheduler();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const start = clampHourOrNull(r.quietHoursStart);
- const end = clampHourOrNull(r.quietHoursEnd);
- const backoffBase = clampPositiveInt(
- r.backoffBaseMinutes,
- d.backoffBaseMinutes,
- SYNC_INTERVAL_MAX_MINUTES,
- );
- return {
- enabled: r.enabled === true,
- defaultIntervalMinutes: clampPositiveInt(
- r.defaultIntervalMinutes,
- d.defaultIntervalMinutes,
- SYNC_INTERVAL_MAX_MINUTES,
- ),
- maxConcurrentSyncs: clampPositiveInt(
- r.maxConcurrentSyncs,
- d.maxConcurrentSyncs,
- SYNC_SCHEDULER_MAX_CONCURRENT_MAX,
- ),
- quietHoursStart: start !== null && end !== null ? start : null,
- quietHoursEnd: start !== null && end !== null ? end : null,
- backoffBaseMinutes: backoffBase,
- // Cap can't sit below the base, or backoff would never grow.
- backoffMaxMinutes: Math.max(
- backoffBase,
- clampPositiveInt(
- r.backoffMaxMinutes,
- d.backoffMaxMinutes,
- SYNC_INTERVAL_MAX_MINUTES,
- ),
- ),
- heartbeatSeconds: clampHeartbeatSeconds(r.heartbeatSeconds),
- keepLatestCheckIntervalMinutes: clampPositiveInt(
- r.keepLatestCheckIntervalMinutes,
- d.keepLatestCheckIntervalMinutes,
- SYNC_INTERVAL_MAX_MINUTES,
- ),
- fullSweepIntervalMinutes: clampIntAllowZero(
- r.fullSweepIntervalMinutes,
- d.fullSweepIntervalMinutes,
- SYNC_INTERVAL_MAX_MINUTES,
- ),
- fullSweepConfirmMaxSuspects: clampIntAllowZero(
- r.fullSweepConfirmMaxSuspects,
- d.fullSweepConfirmMaxSuspects,
- FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX,
- ),
- fullSweepShrinkGuardPercent: clampIntAllowZero(
- r.fullSweepShrinkGuardPercent,
- d.fullSweepShrinkGuardPercent,
- FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX,
- ),
- };
-}
-
-export function defaultSavedVideoBackup(): SavedVideoBackupSettings {
- return {
- enabled: false,
- dest: "",
- intervalMinutes: SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES,
- };
-}
-
-// Coerce a raw settings.savedVideoBackup value into a clean
-// SavedVideoBackupSettings. A missing destination forces enabled off, since a
-// backup with nowhere to go is meaningless.
-export function sanitizeSavedVideoBackup(
- value: unknown,
-): SavedVideoBackupSettings {
- const d = defaultSavedVideoBackup();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const dest = typeof r.dest === "string" ? r.dest.trim() : "";
- return {
- enabled: dest !== "" && r.enabled === true,
- dest,
- intervalMinutes: clampPositiveInt(
- r.intervalMinutes,
- d.intervalMinutes,
- SYNC_INTERVAL_MAX_MINUTES,
- ),
- };
-}
-
-// Where relocated channel media goes: the named locations.
-//
-// This used to be ONE FIELD, `mediaRoot` — a single absolute string, the cold
-// drive, typed once. It grew into a list of entities because a root alone
-// cannot answer the two questions the operator actually has: is that disk here,
-// and if it came up somewhere else, how do I point the channels at it without
-// ssh and hand edits? A location carries an id, a label, the root, an opt-in
-// `autoRepoint`, and the volume identity learned at its last probe.
-//
-// Still NOT a policy: a channel on a location is not thereby deprioritized, and
-// nothing auto-relocates anything because a location exists.
+// THE TWO ABSENCE-KEYED MIGRATIONS THAT REWRITE AN INPUT BLOCK, applied to the
+// raw object BEFORE the parse. Each is handed the raw file, never a parsed one,
+// because "the file does not spell this key" is the whole trigger.
//
-// AVAILABILITY IS NEVER STORED HERE. A refresh that wrote "available" would
-// rewrite settings.json — and so bump the pulse revision — every few seconds.
-// The probe (common/lib/storageVolumes.ts) is computed per request; only the
-// `volume` identity is ever written back, and only when it changed.
+// - autoQueue: `migrateSweepsToLanes` fills `autoQueue.digest` / `.backfill`
+// from the retired sweep fields when — and only when — the file does not
+// already carry that lane. It never enables a lane the sweep flag did not.
+// See lib/laneMigration.ts.
+// - storage: `storage.mediaRoot` (one absolute string) becomes a one-entry
+// location list when the file has no `locations` key. An absent `storage`
+// block is the default block, which has a `locations` key and so does not
+// migrate. See lib/storageLocations.ts.
//
-// The types live in lib/storageLocations.ts, which is pure: a `"use client"`
-// file may import them, and must not reach storageVolumes.ts (execa).
-export type { StorageLocation, StorageVolume, StorageSettings };
-
-export function defaultStorage(): StorageSettings {
- return { locations: [], defaultLocationId: "" };
-}
-
-const LOCATION_ID_RE = /^[a-z0-9][a-z0-9-]{0,63}$/;
-
-// "internal" IS TAKEN. It is the synthetic /storage row for the corpus volume
-// (INTERNAL_LOCATION_ID), and the regex above admits it — so a hand-edited
-// settings.json, or an operator typing the obvious word into the New location
-// form, could store a real location under the one id the page assembles for
-// itself. The row would then be built twice, the rollup would count channels
-// into whichever assembled last, and `locationOfDataDir` would start matching
-// unrelocated channels against it.
-function isReservedLocationId(id: string): boolean {
- return id === INTERNAL_LOCATION_ID;
-}
-
-function sanitizeVolume(value: unknown): StorageVolume | undefined {
- if (!value || typeof value !== "object") return undefined;
- const v = value as Record<string, unknown>;
- const uuid = typeof v.uuid === "string" ? v.uuid.trim() : "";
- const mountpoint =
- typeof v.mountpoint === "string" ? v.mountpoint.trim() : "";
- // No uuid is no identity, and no mountpoint means `root === join(mountpoint,
- // relPath)` cannot hold — either way the record is not usable for finding the
- // volume again, so it is dropped rather than half-kept.
- if (!uuid || !mountpoint) return undefined;
- const relPath = typeof v.relPath === "string" ? v.relPath.trim() : "";
- const fstype = typeof v.fstype === "string" ? v.fstype.trim() : "";
- const label = typeof v.label === "string" ? v.label.trim() : "";
+// Both return RAW values; the schema's sanitizers then normalize them exactly as
+// they normalize a hand-edited file.
+function premigrateRaw(raw: RawSettings): RawSettings {
return {
- uuid,
- ...(fstype ? { fstype } : {}),
- ...(label ? { label } : {}),
- mountpoint,
- relPath,
+ ...raw,
+ autoQueue: migrateSweepsToLanes(raw),
+ storage: migrateMediaRootToLocations(raw.storage ?? defaultStorage()),
};
}
-// Coerce a raw settings.storage value into a clean StorageSettings.
-//
-// EXISTENCE IS NOT CHECKED, on purpose: the whole point of a cold location is
-// that it is a drive that may not be mounted when settings are read, and a
-// sanitizer that dropped the root on an unmounted platter would silently erase
-// the operator's choice on the next save.
+// THE TWO ABSENCE-KEYED MIGRATIONS THAT NEED THE PARSED RESULT, in this order:
//
-// ABSOLUTENESS *IS* checked, and a location with a relative root is DROPPED
-// rather than resolved. Resolving it would anchor the location to whatever cwd
-// the reader booted in — a different directory under docker, under a worktree,
-// and under `pnpm dev` — so the same settings.json would name three different
-// drives. The location form rejects a relative path with a message before it
-// ever gets here; this is the last line, not the only one.
-//
-// NESTED ROOTS ARE ALLOWED. "/mnt/platter" and "/mnt/platter/archive" may both
-// be locations; `locationOfDataDir` resolves a channel to the LONGEST matching
-// root. Nothing here rejects the nesting, because the operator who arranges a
-// disk that way means it.
-//
-// A STALE `mediaRoot` SITTING BESIDE `locations` IS IGNORED — it is not merged
-// back in as an extra location. `migrateMediaRootToLocations` reads it exactly
-// once, when `locations` is absent; after that the list is the whole truth, and
-// resurrecting a root the operator deleted would be a bug, not a kindness.
-//
-// ROLLBACK: an older binary sanitizes this block to `{ mediaRoot: "" }` — the
-// locations are dropped and the single cold root comes back blank. One string
-// lost, nothing on disk moved. `cp settings.json settings.json.pre-storage-
-// locations` before the upgrade and a downgrade is a file copy.
-export function sanitizeStorage(value: unknown): StorageSettings {
- const d = defaultStorage();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const rawList = Array.isArray(r.locations) ? r.locations : [];
- const locations: StorageLocation[] = [];
- const seen = new Set<string>();
- for (const entry of rawList) {
- if (!entry || typeof entry !== "object") continue;
- const e = entry as Record<string, unknown>;
- const id = typeof e.id === "string" ? e.id.trim() : "";
- if (!LOCATION_ID_RE.test(id) || isReservedLocationId(id) || seen.has(id)) {
- continue;
- }
- const rawRoot = typeof e.root === "string" ? e.root.trim() : "";
- if (!path.isAbsolute(rawRoot)) continue;
- // "/mnt/platter/" and "/mnt/platter" are one root; "/" stays "/".
- const stripped = rawRoot.replace(/\/+$/, "");
- const root = stripped === "" ? "/" : stripped;
- const label = typeof e.label === "string" ? e.label.trim() : "";
- const volume = sanitizeVolume(e.volume);
- seen.add(id);
- locations.push({
- id,
- label: label || id,
- root,
- autoRepoint: e.autoRepoint === true,
- ...(volume ? { volume } : {}),
- });
- }
- const wanted =
- typeof r.defaultLocationId === "string" ? r.defaultLocationId.trim() : "";
- // A default naming a location that is gone falls back to the first one, not
- // to "": with a location configured, "no default" is never the answer the
- // operator wanted, and a blank default silently disables every prefill.
- const defaultLocationId = locations.some((l) => l.id === wanted)
- ? wanted
- : (locations[0]?.id ?? "");
- // THE SAVED-VIDEO STORE'S LOCATION IS NOT FALLEN BACK, and the asymmetry
- // with `defaultLocationId` above is deliberate. That one is a PREFERENCE, so
- // picking another location when the named one is gone is helpful. This one is
- // a RECORD OF WHERE BYTES ARE: pointing it at a different location because
- // the recorded one was deleted would claim the store had moved when nothing
- // had. A dangling id sanitizes to "" — "in place" — which is what the disk
- // says as soon as anybody looks, and the symlink (if any) keeps working
- // regardless, because the store is reached through it and not through this.
- const savedWanted =
- typeof r.savedVideosLocationId === "string"
- ? r.savedVideosLocationId.trim()
- : "";
- const savedVideosLocationId = locations.some((l) => l.id === savedWanted)
- ? savedWanted
- : "";
- return {
- locations,
- defaultLocationId,
- ...(savedVideosLocationId ? { savedVideosLocationId } : {}),
- };
-}
-
-// 4 hours. Measured: videos over this are 8.2% of the corpus by count but hold
-// 46% of all transcript tokens, so they are where a sweep's wall-clock actually
-// goes and where chunk-seam bugs live.
-export const DIGEST_LONG_TAIL_DEFAULT_SECONDS = 4 * 3600;
-export const DIGEST_LONG_TAIL_MAX_SECONDS = 24 * 3600;
-
-export function defaultDigest(): DigestSettings {
- return {
- // OFF. The metered lane is built but never the default — see PLAN.md.
- remoteEnabled: false,
- longTailSeconds: DIGEST_LONG_TAIL_DEFAULT_SECONDS,
- localAppId: DEFAULT_DIGEST_APP_ID,
- remoteAppId: CLAUDE_DIGEST_APP_ID,
- // Empty on purpose: every per-app knob falls through to its own default
- // constant (resolveNumCtx -> DEFAULT_DIGEST_NUM_CTX, now 8192, and
- // maxCuesForContext sizes the chunk to it). Seeding a copy of those values
- // here would give the same number two homes and let them drift.
- apps: {},
- // ON. Real GPU contention with the transcription engine is a genuine cost
- // (re-priced: 11.2 s/chunk idle against 24.9 s/chunk on a contended box), so
- // the safe default is to step aside; turning it off is the deliberate choice.
- yieldToTranscription: true,
- // OFF. A CPU-pinned worker is not GPU contention, and treating it as such
- // stalled the digest lane for nothing. See DigestSettings.yieldToCpuWorkers.
- yieldToCpuWorkers: false,
- spendCapUsd: 0,
- sections: ["chapters"],
- timestampMode: DEFAULT_DIGEST_TIMESTAMP_MODE,
- promptVariant: "",
- };
-}
-
-// Coerce a raw settings.digest.apps value into a clean keyed map of
-// DigestAppConfig. Mirrors sanitizeTranscriptionApps — INCLUDING its
-// Array.isArray guard, without which a JSON array would pass the typeof check and
-// produce numeric-keyed garbage.
-export function sanitizeDigestApps(
- value: unknown,
-): Record<string, DigestAppConfig> {
- if (!value || typeof value !== "object" || Array.isArray(value)) return {};
- const out: Record<string, DigestAppConfig> = {};
- for (const [id, raw] of Object.entries(value as Record<string, unknown>)) {
- if (!raw || typeof raw !== "object") continue;
- const r = raw as Record<string, unknown>;
- const cfg: DigestAppConfig = {};
- if (typeof r.bin === "string" && r.bin.trim()) cfg.bin = r.bin.trim();
- if (typeof r.baseUrl === "string" && r.baseUrl.trim()) {
- cfg.baseUrl = r.baseUrl.trim();
- }
- if (typeof r.model === "string" && r.model.trim()) cfg.model = r.model.trim();
- if (typeof r.numCtx === "number" && r.numCtx > 0) {
- cfg.numCtx = Math.floor(r.numCtx);
- }
- if (typeof r.temperature === "number" && r.temperature >= 0) {
- cfg.temperature = r.temperature;
- }
- if (typeof r.timeoutMs === "number" && r.timeoutMs > 0) {
- cfg.timeoutMs = Math.floor(r.timeoutMs);
- }
- // Only carried when explicitly set — see DigestAppConfig.think.
- if (typeof r.think === "boolean") cfg.think = r.think;
- out[id] = cfg;
- }
- return out;
-}
-
-export function sanitizeDigest(value: unknown): DigestSettings {
- const d = defaultDigest();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const sections = Array.isArray(r.sections)
- ? (r.sections.filter(isDigestSectionKind) as DigestSectionKind[])
- : [];
- return {
- remoteEnabled: r.remoteEnabled === true,
- longTailSeconds: clampPositiveInt(
- r.longTailSeconds,
- d.longTailSeconds,
- DIGEST_LONG_TAIL_MAX_SECONDS,
- ),
- // Unknown app ids are not rejected here: getDigestApp() is total and falls
- // back to the local default, so a stale id degrades rather than breaking.
- localAppId:
- typeof r.localAppId === "string" && r.localAppId.trim()
- ? r.localAppId.trim()
- : d.localAppId,
- remoteAppId:
- typeof r.remoteAppId === "string" && r.remoteAppId.trim()
- ? r.remoteAppId.trim()
- : d.remoteAppId,
- apps: sanitizeDigestApps(r.apps),
- // Defaults to ON when absent — `=== false` rather than `!== true`, so a
- // settings file written before this field existed keeps the GPU-safe
- // behaviour instead of silently opting into contention.
- yieldToTranscription: r.yieldToTranscription !== false,
- // The OPPOSITE idiom, and deliberately so: `=== true`, so absence falls to
- // OFF. The field's absence means a settings file written before the CPU-worker
- // bug was found, and for those files OFF is the FIXED behaviour, not a silent
- // change of intent — nobody ever asked to stall the digest lane for a CPU
- // transcription. `yieldToTranscription` still gates the whole thing, so the
- // GPU-safe default is untouched.
- yieldToCpuWorkers: r.yieldToCpuWorkers === true,
- spendCapUsd:
- typeof r.spendCapUsd === "number" && r.spendCapUsd > 0
- ? Math.round(r.spendCapUsd * 100) / 100
- : 0,
- // An empty/garbage list would silently generate nothing, so fall back to the
- // default rather than honoring it.
- sections: sections.length > 0 ? sections : d.sections,
- timestampMode: isDigestTimestampMode(r.timestampMode)
- ? r.timestampMode
- : d.timestampMode,
- // Trimmed and length-capped: it goes into provenance on every record, and a
- // runaway value would bloat 119k sidecars.
- promptVariant:
- typeof r.promptVariant === "string"
- ? r.promptVariant.trim().slice(0, 40)
- : d.promptVariant,
- };
-}
-
-// Every known section kind, for the settings UI's checkbox list.
-export const DIGEST_SECTION_OPTIONS = DIGEST_SECTION_KINDS;
-export const DIGEST_TIMESTAMP_MODE_OPTIONS = DIGEST_TIMESTAMP_MODES;
-
-export const BUILD_MAX_PARALLEL_DEFAULT = 2;
-export const BUILD_MAX_PARALLEL_MAX = 16;
-export const DEFAULT_BUILD_IMAGE = "yt-dlp-transcript-browser-build";
-export const DEFAULT_BUILD_DOCKERFILE = "Dockerfile.build";
-
-export function isBuildMode(v: unknown): v is BuildMode {
- return v === "basic" || v === "docker";
-}
-
-export function defaultBuildPipeline(): BuildPipelineSettings {
- return {
- mode: "basic",
- maxParallelBuilds: BUILD_MAX_PARALLEL_DEFAULT,
- dockerImage: DEFAULT_BUILD_IMAGE,
- dockerfile: DEFAULT_BUILD_DOCKERFILE,
- };
-}
-
-// Coerce a raw settings.buildPipeline value into a clean BuildPipelineSettings,
-// falling back to defaults for missing/ill-typed fields.
-export function sanitizeBuildPipeline(value: unknown): BuildPipelineSettings {
- const d = defaultBuildPipeline();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const dockerImage =
- typeof r.dockerImage === "string" && r.dockerImage.trim()
- ? r.dockerImage.trim()
- : d.dockerImage;
- const dockerfile =
- typeof r.dockerfile === "string" && r.dockerfile.trim()
- ? r.dockerfile.trim()
- : d.dockerfile;
- return {
- mode: isBuildMode(r.mode) ? r.mode : d.mode,
- maxParallelBuilds: clampPositiveInt(
- r.maxParallelBuilds,
- d.maxParallelBuilds,
- BUILD_MAX_PARALLEL_MAX,
- ),
- dockerImage,
- dockerfile,
- };
-}
-
-function defaults(): SiteSettings {
- return {
- adminTitle: DEFAULT_ADMIN_TITLE,
- maxTranscriptPageBytes: TRANSCRIPT_PAGE_DEFAULT_BYTES,
- transcriptionApp: DEFAULT_TRANSCRIPTION_APP_ID,
- transcriptionApps: {},
- workers: [],
- cookiesFromBrowser: "",
- cookieMode: DEFAULT_COOKIE_MODE,
- sleepBetweenDownloadsSeconds: SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS,
- downloadFormat: "auto",
- minFreeDiskGB: MIN_FREE_DISK_GB_DEFAULT,
- resumeMarginGB: RESUME_MARGIN_GB_DEFAULT,
- parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT,
- inlineTranscribeOnFallback: false,
- skipLiveDownloads: true,
- verifyAvailabilityBeforeClean: true,
- buildArchives: true,
- archiveStorage: { bucket: "", publicBaseUrl: "" },
- reportDebouncePreset: DEFAULT_REPORT_DEBOUNCE_PRESET,
- autoRefreshIntervalSeconds: AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS,
- syncScheduler: defaultSyncScheduler(),
- autoQueue: defaultAutoQueue(),
- channelPriority: defaultChannelPriority(),
- socialLinks: [],
- homepageUrl: "",
- savedVideoBackup: defaultSavedVideoBackup(),
- storage: defaultStorage(),
- buildPipeline: defaultBuildPipeline(),
- // Must be listed here or the allowlist loop in getSettings() drops the key
- // entirely and the whole section is never read from disk.
- digest: defaultDigest(),
- diarization: defaultDiarization(),
- backfill: defaultBackfill(),
- attribution: defaultAttribution(),
- };
-}
-
-// The whole default settings object, without touching disk. Exported so a test
-// (or any caller that needs a settings-SHAPED value rather than the operator's
-// actual configuration) can build one without a settings.json.
-export function defaultSiteSettings(): SiteSettings {
- return defaults();
-}
-
-export function defaultBackfill(): BackfillSettings {
- return {
- concurrency: 1,
- // See BackfillSettings.allowRedownload — this one holds disk.
- allowRedownload: false,
- };
-}
-
-export function sanitizeBackfill(value: unknown): BackfillSettings {
- const d = defaultBackfill();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- return {
- // Clamped rather than rejected: a hand-edited 5 means "as much as possible",
- // and reading it as 0 would be the opposite of the intent.
- concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16),
- allowRedownload: r.allowRedownload === true,
- };
-}
-
-export function defaultAttribution(): AttributionSettings {
- return {
- // OFF, and both lanes OFF under it. See AttributionSettings.
- enabled: false,
- appId: DEFAULT_DIGEST_APP_ID,
- model: "",
- diarizedEnabled: false,
- textOnlyEnabled: false,
- promptVersion: ATTRIBUTION_PROMPT_VERSION,
- };
-}
-
-export function sanitizeAttribution(value: unknown): AttributionSettings {
- const d = defaultAttribution();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const str = (v: unknown, fallback: string) =>
- typeof v === "string" && v.trim() ? v.trim() : fallback;
- return {
- enabled: r.enabled === true,
- appId: str(r.appId, d.appId),
- // NOT `str(...)` with a fallback: empty is a meaningful value here ("use the
- // app's own model"), so an empty string must survive rather than reverting
- // to a default that is also empty by coincidence.
- model: typeof r.model === "string" ? r.model.trim() : d.model,
- diarizedEnabled: r.diarizedEnabled === true,
- textOnlyEnabled: r.textOnlyEnabled === true,
- // FLOORED at the shipped constant, never merely defaulted. A hand-edited
- // value below it would pin freshness to a superseded prompt generation and
- // freeze its output into the corpus — see AttributionSettings.promptVersion.
- promptVersion:
- typeof r.promptVersion === "number" && Number.isFinite(r.promptVersion)
- ? Math.max(ATTRIBUTION_PROMPT_VERSION, Math.floor(r.promptVersion))
- : d.promptVersion,
- };
-}
-
-export function defaultDiarization(): DiarizationSettings {
- return {
- // OFF. Capture is opt-in: turning it on makes the cleanup sweep start
- // refusing to delete audio for transcribed-but-undiarized videos, which is
- // correct but is a disk-pressure decision an operator should make.
- enabled: false,
- // OFF — see DiarizationSettings.inlineAfterTranscribe. Diarization is slower
- // than the transcription it would follow, so inline is the exception.
- inlineAfterTranscribe: false,
- // Measured, not sherpa-onnx's 0.5 — see DiarizationSettings.threshold. The
- // constant lives in lib/diarization.ts because isDiarizationFresh needs it
- // to normalize an absent recorded threshold; importing it keeps the default
- // and the comparator from drifting apart.
- threshold: DEFAULT_DIARIZATION_THRESHOLD,
- threads: 4,
- // The engine every sidecar on disk was produced by. Switching is an explicit
- // decision that restates the freshness identity — see DiarizationSettings.
- engine: DEFAULT_DIARIZATION_ENGINE,
- // Only consulted when engine is "sortformer". Defaulting to the GPU is safe
- // because the lane yields the card to transcription rather than sharing it.
- backend: "vulkan",
- python: "python3",
- segModel: "",
- embModel: "",
- sortformerBin: "",
- sortformerModel: "",
- concurrency: 1,
- // OFF, because windowing made it unnecessary — which is what it was always
- // for. It shipped at 4 hours as a stopgap while long recordings were being
- // OOM-killed; the engine now processes them in windows and the 6h12m file
- // that was being killed at 10.5 GB completes at ~2 GB. The knob stays, and
- // stays honest about what it does, for a machine smaller than this one or a
- // recording longer than anything measured here.
- maxAudioHours: 0,
- };
-}
-
-export function sanitizeDiarization(value: unknown): DiarizationSettings {
- const d = defaultDiarization();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const str = (v: unknown, fallback: string) =>
- typeof v === "string" && v.trim() ? v.trim() : fallback;
- return {
- enabled: r.enabled === true,
- inlineAfterTranscribe: r.inlineAfterTranscribe === true,
- threshold:
- typeof r.threshold === "number" &&
- Number.isFinite(r.threshold) &&
- r.threshold > 0
- ? r.threshold
- : d.threshold,
- threads: clampPositiveInt(r.threads, d.threads, 64),
- // An unknown engine falls back to the default rather than disabling the lane:
- // a typo in settings.json must not silently stop diarization, and the default
- // is the one every existing sidecar already matches.
- engine: DIARIZATION_ENGINE_IDS.includes(r.engine as DiarizationEngineId)
- ? (r.engine as DiarizationEngineId)
- : d.engine,
- backend: DIARIZATION_BACKENDS.includes(r.backend as DiarizationBackend)
- ? (r.backend as DiarizationBackend)
- : d.backend,
- python: str(r.python, d.python),
- segModel: str(r.segModel, d.segModel),
- embModel: str(r.embModel, d.embModel),
- sortformerBin: str(r.sortformerBin, d.sortformerBin),
- sortformerModel: str(r.sortformerModel, d.sortformerModel),
- concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16),
- // 0 is meaningful here (cap off), so this cannot use clampPositiveInt.
- // Fractional hours are allowed — the knob is a duration, not a count.
- maxAudioHours:
- typeof r.maxAudioHours === "number" &&
- Number.isFinite(r.maxAudioHours) &&
- r.maxAudioHours >= 0
- ? r.maxAudioHours
- : d.maxAudioHours,
- };
-}
-
-const SOCIAL_URL_RE = /^(https?:\/\/|mailto:|\/)/i;
-
-// Normalize the family hub URL into a trailing-slash-free absolute http(s) URL.
-// Returns "" for anything that isn't a usable absolute URL (the "no hub" state).
-// Kept local to settings.ts to avoid a settings<->homepage import cycle; mirrors
-// parseHomepageUrl() in homepage.ts.
-export function normalizeHomepageUrl(input: unknown): string {
- if (typeof input !== "string") return "";
- const trimmed = input.trim().replace(/\/+$/, "");
- return /^https?:\/\/\S+/i.test(trimmed) ? trimmed : "";
-}
-
-export function parseSocialLinks(input: unknown): SocialLink[] {
- if (!Array.isArray(input)) return [];
- const out: SocialLink[] = [];
- for (const raw of input) {
- if (!raw || typeof raw !== "object") continue;
- const r = raw as Record<string, unknown>;
- const label = typeof r.label === "string" ? r.label.trim() : "";
- const url = typeof r.url === "string" ? r.url.trim() : "";
- const svg = typeof r.svg === "string" ? r.svg : "";
- if (!label || !url || !svg) continue;
- if (!SOCIAL_URL_RE.test(url)) continue;
- out.push({ label, url, svg });
- }
- return out;
-}
-
-// Normalize an admin-provided SVG snippet for inline use in the export
-// footer. Returns null on anything that looks unsafe or unrenderable.
-// Steps: trim, allowlist-check, strip width/height, force fill="currentColor"
-// + aria-hidden on the root <svg>. Requires a viewBox so the icon scales.
-export function normalizeSocialSvg(raw: string): string | null {
- if (typeof raw !== "string") return null;
- const trimmed = raw.trim();
- if (!trimmed.startsWith("<svg") || !trimmed.endsWith("</svg>")) return null;
- if (/<script\b/i.test(trimmed)) return null;
- if (/<foreignObject\b/i.test(trimmed)) return null;
- if (/<iframe\b/i.test(trimmed)) return null;
- if (/javascript:/i.test(trimmed)) return null;
- if (/\son[a-z]+\s*=/i.test(trimmed)) return null;
- if (/<\?|<!ENTITY/i.test(trimmed)) return null;
-
- const openEnd = trimmed.indexOf(">");
- if (openEnd < 0) return null;
- let opening = trimmed.slice(0, openEnd);
- const rest = trimmed.slice(openEnd);
-
- if (!/\sviewBox\s*=\s*"/i.test(opening)) return null;
-
- opening = opening.replace(/\s(width|height)\s*=\s*"[^"]*"/gi, "");
- opening = opening.replace(/\s(width|height)\s*=\s*'[^']*'/gi, "");
-
- if (!/\sfill\s*=/i.test(opening)) {
- opening = opening.replace(/^<svg/i, '<svg fill="currentColor"');
- }
- if (!/\saria-hidden\s*=/i.test(opening)) {
- opening = opening.replace(/^<svg/i, '<svg aria-hidden="true"');
- }
- return opening + rest;
-}
-
-// normalizeSocialSvg() deliberately STRIPS width/height so the icon scales to its
-// wrapper. The cost is that a viewBox-only <svg> has no intrinsic size, so before
-// the stylesheet loads on a static host it paints at the replaced-element default
-// (huge) — the "flash of giant social icons" FOUC. sizeSocialSvg() re-injects an
-// intrinsic pixel size at RENDER time (existing site.json files already have the
-// attributes stripped, so this must run on read, not just on write). The size is
-// an *attribute*, not inline style, so a wrapper's `w-*`/`h-*` utilities still win
-// once CSS loads — it only governs the pre-CSS first paint.
-export function sizeSocialSvg(svg: string, px = 20): string {
- if (typeof svg !== "string") return svg;
- if (/^<svg[^>]*\swidth\s*=/i.test(svg)) return svg; // already sized
- return svg.replace(/^<svg\b/i, `<svg width="${px}" height="${px}"`);
-}
-
-export function clampSleepBetweenDownloadsSeconds(value: unknown): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? Math.floor(value)
- : SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS;
- if (n < 0) return 0;
- if (n > SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS) {
- return SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS;
- }
- return n;
-}
-
-export function clampMinFreeDiskGB(value: unknown): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? Math.floor(value)
- : MIN_FREE_DISK_GB_DEFAULT;
- if (n < 0) return 0;
- if (n > MIN_FREE_DISK_GB_MAX) return MIN_FREE_DISK_GB_MAX;
- return n;
-}
-
-export function clampResumeMarginGB(value: unknown): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? Math.floor(value)
- : RESUME_MARGIN_GB_DEFAULT;
- if (n < 0) return 0;
- if (n > RESUME_MARGIN_GB_MAX) return RESUME_MARGIN_GB_MAX;
- return n;
-}
-
-export function clampParallelTranscriptions(value: unknown): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? Math.floor(value)
- : PARALLEL_TRANSCRIPTIONS_DEFAULT;
- if (n < 1) return 1;
- if (n > PARALLEL_TRANSCRIPTIONS_MAX) return PARALLEL_TRANSCRIPTIONS_MAX;
- return n;
-}
-
-// 0 means "disabled" and is preserved as-is. Anything else is clamped into the
-// [MIN, MAX] window; a non-finite value falls back to the default cadence.
-export function clampAutoRefreshIntervalSeconds(value: unknown): number {
- if (typeof value !== "number" || !Number.isFinite(value)) {
- return AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS;
- }
- const n = Math.floor(value);
- if (n <= 0) return 0;
- if (n < AUTO_REFRESH_INTERVAL_MIN_SECONDS) {
- return AUTO_REFRESH_INTERVAL_MIN_SECONDS;
- }
- if (n > AUTO_REFRESH_INTERVAL_MAX_SECONDS) {
- return AUTO_REFRESH_INTERVAL_MAX_SECONDS;
- }
- return n;
-}
-
-export function getSettings(): SiteSettings {
- const file = getPaths().settingsFile;
- let parsed: Partial<SiteSettings> = {};
- try {
- parsed = JSON.parse(fs.readFileSync(file, "utf8")) as Partial<SiteSettings>;
- } catch {
- parsed = {};
- }
- // Pick only known operational keys — a pre-multi-site settings.json may still
- // carry siteTitle/groups/socialLinks, which now live per-site in site.json.
- const merged: SiteSettings = { ...defaults() };
- const knownKeys = Object.keys(merged) as (keyof SiteSettings)[];
- for (const key of knownKeys) {
- if (parsed[key] !== undefined) {
- (merged as Record<string, unknown>)[key] = parsed[key];
- }
- }
- if (typeof merged.adminTitle !== "string" || !merged.adminTitle.trim()) {
- merged.adminTitle = DEFAULT_ADMIN_TITLE;
- }
- merged.maxTranscriptPageBytes = clampPageBytes(merged.maxTranscriptPageBytes);
- merged.transcriptionApps = sanitizeTranscriptionApps(merged.transcriptionApps);
- // Migrate a pre-multi-app settings.json (transcribeBin/transcribeArgs/
- // transcribeModel, with no transcriptionApp key) onto the app registry.
- if (parsed.transcriptionApp === undefined) {
- migrateLegacyTranscription(parsed as LegacyTranscribeFields, merged);
- }
- if (
- typeof merged.transcriptionApp !== "string" ||
- !TRANSCRIPTION_APPS[merged.transcriptionApp]
- ) {
- merged.transcriptionApp = DEFAULT_TRANSCRIPTION_APP_ID;
- }
- if (typeof merged.cookiesFromBrowser !== "string") {
- merged.cookiesFromBrowser = "";
- }
- // A settings.json predating cookieMode (or carrying junk) gets the default,
- // which preserves the historical retry-only behavior.
- if (!isCookieMode(merged.cookieMode)) {
- merged.cookieMode = DEFAULT_COOKIE_MODE;
- }
- merged.sleepBetweenDownloadsSeconds = clampSleepBetweenDownloadsSeconds(
- merged.sleepBetweenDownloadsSeconds,
- );
- if (!isDownloadFormatPreset(merged.downloadFormat)) {
- merged.downloadFormat = "auto";
- }
- merged.minFreeDiskGB = clampMinFreeDiskGB(merged.minFreeDiskGB);
- merged.resumeMarginGB = clampResumeMarginGB(merged.resumeMarginGB);
- merged.parallelTranscriptions = clampParallelTranscriptions(
- merged.parallelTranscriptions,
- );
- if (typeof merged.inlineTranscribeOnFallback !== "boolean") {
- merged.inlineTranscribeOnFallback = false;
- }
- if (typeof merged.skipLiveDownloads !== "boolean") {
- merged.skipLiveDownloads = true;
- }
- if (typeof merged.verifyAvailabilityBeforeClean !== "boolean") {
- merged.verifyAvailabilityBeforeClean = true;
- }
- if (typeof merged.buildArchives !== "boolean") {
- merged.buildArchives = true;
- }
- {
- const s = merged.archiveStorage;
- merged.archiveStorage = {
- bucket: s && typeof s.bucket === "string" ? s.bucket : "",
- publicBaseUrl:
- s && typeof s.publicBaseUrl === "string" ? s.publicBaseUrl : "",
- };
- }
- if (!isReportDebouncePreset(merged.reportDebouncePreset)) {
- merged.reportDebouncePreset = DEFAULT_REPORT_DEBOUNCE_PRESET;
- }
- merged.autoRefreshIntervalSeconds = clampAutoRefreshIntervalSeconds(
- merged.autoRefreshIntervalSeconds,
- );
- merged.syncScheduler = sanitizeSyncScheduler(merged.syncScheduler);
- // THE SWEEPS' SCOPE, ON READ. `migrateSweepsToLanes` fills in
- // `autoQueue.digest` / `.backfill` from the retired sweep fields when — and
- // only when — the FILE does not already spell them, which is why it is handed
- // `parsed` rather than `merged`: `merged` has had defaults folded in and can
- // no longer tell "absent" from "default". It never enables a lane the sweep
- // flag did not. See lib/laneMigration.ts.
- merged.autoQueue = sanitizeAutoQueue(migrateSweepsToLanes(parsed));
- // No migration beside it: the legacy read (`channelPriorityFromLegacy`) needs
- // 68 config.json files and getSettings is synchronous and reads one. It is a
- // one-shot offline script instead, and an absent document sanitizes to the
- // empty one, which means today's behaviour.
- merged.channelPriority = sanitizeChannelPriority(merged.channelPriority);
- merged.socialLinks = parseSocialLinks(merged.socialLinks);
- merged.homepageUrl = normalizeHomepageUrl(merged.homepageUrl);
- merged.savedVideoBackup = sanitizeSavedVideoBackup(merged.savedVideoBackup);
- // THE COLD ROOT, ON READ. `storage.mediaRoot` — one absolute string — becomes
- // a one-entry location list. Handed `parsed.storage` rather than
- // `merged.storage` for the same reason `migrateSweepsToLanes` is handed
- // `parsed`: `merged` has had `defaultStorage()` folded in and can no longer
- // tell "the file has no locations key" from "the file has an empty list", and
- // the migration must only fire on the former. See lib/storageLocations.ts.
- merged.storage = sanitizeStorage(
- migrateMediaRootToLocations(
- (parsed as Record<string, unknown>).storage ?? merged.storage,
- ),
- );
- merged.buildPipeline = sanitizeBuildPipeline(merged.buildPipeline);
- merged.digest = sanitizeDigest(merged.digest);
- merged.diarization = sanitizeDiarization(merged.diarization);
- merged.backfill = sanitizeBackfill(merged.backfill);
- merged.attribution = sanitizeAttribution(merged.attribution);
- // Workers. When the file predates the worker model (no `workers` key),
- // synthesize a default list from the (now-settled) active app + per-app
- // configs so existing installs behave identically. Otherwise sanitize the
- // stored list.
- if (parsed.workers === undefined) {
- merged.workers = defaultWorkersFromApps(
- merged.transcriptionApp,
- merged.transcriptionApps,
- merged.parallelTranscriptions,
+// 1. A pre-multi-app file (transcribeBin/transcribeArgs/transcribeModel, no
+// `transcriptionApp` key) is migrated onto the app registry. It reads the
+// already-sanitized `transcriptionApps`, so it runs after the parse.
+// 2. A file predating the worker model (no `workers` key) gets a worker list
+// synthesized from the now-settled active app, so existing installs behave
+// identically. A file that HAS the key keeps its sanitized list, even an
+// empty one.
+function finishRawMigrations(
+ parsed: SiteSettings,
+ raw: RawSettings,
+): SiteSettings {
+ if (raw.transcriptionApp === undefined) {
+ migrateLegacyTranscription(raw as LegacyTranscribeFields, parsed);
+ }
+ if (raw.workers === undefined) {
+ parsed.workers = defaultWorkersFromApps(
+ parsed.transcriptionApp,
+ parsed.transcriptionApps,
+ parsed.parallelTranscriptions,
);
- } else {
- merged.workers = sanitizeWorkers(merged.workers);
}
- return merged;
+ return parsed;
}
-function clampPageBytes(value: unknown): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? value
- : TRANSCRIPT_PAGE_DEFAULT_BYTES;
- if (n < TRANSCRIPT_PAGE_MIN_BYTES) return TRANSCRIPT_PAGE_MIN_BYTES;
- if (n > TRANSCRIPT_PAGE_HARD_CAP_BYTES) return TRANSCRIPT_PAGE_HARD_CAP_BYTES;
- return Math.floor(n);
+export function getSettings(): SiteSettings {
+ const raw = rawObject(readRawSettings(getPaths().settingsFile));
+ return finishRawMigrations(siteSettingsSchema.parse(premigrateRaw(raw)), raw);
}
type LegacyTranscribeFields = {
@@ -1602,20 +132,6 @@ type LegacyTranscribeFields = {
transcribeModel?: unknown;
};
-// Coerce a raw settings.transcriptionApps value into a clean keyed map of
-// AppInstanceConfig, dropping unknown/ill-typed fields.
-export function sanitizeTranscriptionApps(
- value: unknown,
-): Record<string, AppInstanceConfig> {
- if (!value || typeof value !== "object" || Array.isArray(value)) return {};
- const out: Record<string, AppInstanceConfig> = {};
- for (const [id, raw] of Object.entries(value as Record<string, unknown>)) {
- if (!raw || typeof raw !== "object") continue;
- out[id] = sanitizeWorkerConfig(raw);
- }
- return out;
-}
-
function argsAreDefault(args: string[]): boolean {
return (
args.length === DEFAULT_TRANSCRIBE_ARGS.length &&
@@ -1659,10 +175,16 @@ function migrateLegacyTranscription(
merged.transcriptionApps = { ...merged.transcriptionApps, [appId]: cfg };
}
-export async function writeSettings(next: SiteSettings): Promise<void> {
- // Workers are the source of truth. A caller that still sets only the
- // deprecated transcriptionApp/transcriptionApps (no `workers`) gets a list
- // synthesized from them, so old call sites keep working during the transition.
+// THE WORKER SHADOW, derived on every save.
+//
+// Workers are the source of truth. A caller that still sets only the deprecated
+// transcriptionApp/transcriptionApps (no `workers`) gets a list synthesized from
+// them, so old call sites keep working. The list is then VALIDATED — this throws
+// — and the deprecated pair is rewritten as a faithful shadow of it (used only
+// for rollback to a pre-worker build; a file with a `workers` key is never
+// re-migrated): the active app is the first enabled local worker, and the per-app
+// map mirrors each local worker's config.
+function deriveWorkerShadow(next: SiteSettings): SiteSettings {
let workers = sanitizeWorkers(next.workers);
if (workers.length === 0) {
const fallbackAppId =
@@ -1679,14 +201,10 @@ export async function writeSettings(next: SiteSettings): Promise<void> {
const workersErr = validateWorkers(workers);
if (workersErr) throw new Error(workersErr);
- // Keep the deprecated transcriptionApp/transcriptionApps as a faithful shadow
- // of the workers (used only for rollback to a pre-worker build — a file with a
- // `workers` key is never re-migrated). The active app is the first enabled
- // local worker; the per-app map mirrors each local worker's config.
const firstLocal =
workers.find((w) => w.enabled && w.kind === "local") ??
workers.find((w) => w.kind === "local");
- const appId =
+ const transcriptionApp =
firstLocal?.appId && TRANSCRIPTION_APPS[firstLocal.appId]
? firstLocal.appId
: DEFAULT_TRANSCRIPTION_APP_ID;
@@ -1696,87 +214,36 @@ export async function writeSettings(next: SiteSettings): Promise<void> {
transcriptionApps[w.appId] = w.config ?? {};
}
}
- const socialLinks: SocialLink[] = [];
- for (const link of parseSocialLinks(next.socialLinks)) {
+ return { ...next, workers, transcriptionApp, transcriptionApps };
+}
+
+// Every social link's SVG normalized for inline use, or a THROW naming the
+// first one that is not safe to inline. The schema's own `parseSocialLinks`
+// only checks shape; this is the write-side half.
+function validatedSocialLinks(value: unknown): SocialLink[] {
+ const out: SocialLink[] = [];
+ for (const link of parseSocialLinks(value)) {
const svg = normalizeSocialSvg(link.svg);
if (svg === null) {
throw new Error(`Social link "${link.label}" has an invalid SVG`);
}
- socialLinks.push({ ...link, svg });
+ out.push({ ...link, svg });
}
- const file = getPaths().settingsFile;
- // Build the output from ONLY the known operational fields. Do NOT spread
- // `next`: getSettings() spreads the raw file, so a settings.json still
- // carrying pre-multi-site keys (siteTitle/groups/socialLinks) would otherwise
- // smuggle those stale keys back onto disk on every save.
- //
- // THIS IS ALSO WHAT RETIRES A FIELD. The four legacy pause flags
- // (`transcriptionsPaused`, `downloadsPaused`, `digest.digestsPaused` and the
- // inverted `backfill.enabled`) are gone from the type, from the sanitizers
- // and from this literal, so a settings.json that still spells one is read
- // past on load and loses it on the next write. The gate is
+ return out;
+}
+
+export async function writeSettings(next: SiteSettings): Promise<void> {
+ // THE SCHEMA IS ALSO WHAT RETIRES A FIELD. It names only the known
+ // operational fields and strips the rest, so a settings.json still carrying
+ // pre-multi-site keys (siteTitle/groups) or the four retired pause flags
+ // (`transcriptionsPaused`, `downloadsPaused`, `digest.digestsPaused`, the
+ // inverted `backfill.enabled`) loses them on this write. The gate is
// `autoQueue[lane].held` and nothing else — see lib/pauseGates.ts.
- const merged: SiteSettings = {
- adminTitle:
- typeof next.adminTitle === "string" && next.adminTitle.trim()
- ? next.adminTitle.trim()
- : DEFAULT_ADMIN_TITLE,
- maxTranscriptPageBytes: clampPageBytes(next.maxTranscriptPageBytes),
- transcriptionApp: appId,
- transcriptionApps,
- workers,
- cookiesFromBrowser:
- typeof next.cookiesFromBrowser === "string"
- ? next.cookiesFromBrowser.trim()
- : "",
- cookieMode: isCookieMode(next.cookieMode)
- ? next.cookieMode
- : DEFAULT_COOKIE_MODE,
- sleepBetweenDownloadsSeconds: clampSleepBetweenDownloadsSeconds(
- next.sleepBetweenDownloadsSeconds,
- ),
- downloadFormat: isDownloadFormatPreset(next.downloadFormat)
- ? next.downloadFormat
- : "auto",
- minFreeDiskGB: clampMinFreeDiskGB(next.minFreeDiskGB),
- resumeMarginGB: clampResumeMarginGB(next.resumeMarginGB),
- parallelTranscriptions: clampParallelTranscriptions(
- next.parallelTranscriptions,
- ),
- inlineTranscribeOnFallback: next.inlineTranscribeOnFallback === true,
- skipLiveDownloads: next.skipLiveDownloads !== false,
- verifyAvailabilityBeforeClean:
- next.verifyAvailabilityBeforeClean !== false,
- buildArchives: next.buildArchives !== false,
- archiveStorage: {
- bucket:
- typeof next.archiveStorage?.bucket === "string"
- ? next.archiveStorage.bucket.trim()
- : "",
- publicBaseUrl:
- typeof next.archiveStorage?.publicBaseUrl === "string"
- ? next.archiveStorage.publicBaseUrl.trim()
- : "",
- },
- reportDebouncePreset: isReportDebouncePreset(next.reportDebouncePreset)
- ? next.reportDebouncePreset
- : DEFAULT_REPORT_DEBOUNCE_PRESET,
- autoRefreshIntervalSeconds: clampAutoRefreshIntervalSeconds(
- next.autoRefreshIntervalSeconds,
- ),
- syncScheduler: sanitizeSyncScheduler(next.syncScheduler),
- autoQueue: sanitizeAutoQueue(next.autoQueue),
- channelPriority: sanitizeChannelPriority(next.channelPriority),
- socialLinks,
- homepageUrl: normalizeHomepageUrl(next.homepageUrl),
- savedVideoBackup: sanitizeSavedVideoBackup(next.savedVideoBackup),
- storage: sanitizeStorage(next.storage),
- buildPipeline: sanitizeBuildPipeline(next.buildPipeline),
- digest: sanitizeDigest(next.digest),
- diarization: sanitizeDiarization(next.diarization),
- backfill: sanitizeBackfill(next.backfill),
- attribution: sanitizeAttribution(next.attribution),
- };
+ const merged = siteSettingsSchema.parse({
+ ...deriveWorkerShadow(next),
+ socialLinks: validatedSocialLinks(next.socialLinks),
+ });
+ const file = getPaths().settingsFile;
const tmp = `${file}.tmp-${process.pid}`;
await fs.promises.writeFile(tmp, JSON.stringify(merged, null, 2) + "\n");
await fs.promises.rename(tmp, file);
diff --git a/common/lib/settingsDocs.test.ts b/common/lib/settingsDocs.test.ts
@@ -0,0 +1,63 @@
+import { readFileSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { renderSettingsExample, renderSettingsMarkdown } from "./settingsDocs";
+
+// Run with: node_modules/.bin/tsx --test common/lib/settingsDocs.test.ts
+//
+// settings.json.example and SETTINGS.md are GENERATED from the settings schema
+// (common/bin/settings-example.ts). This is what keeps them generated: a hand
+// edit to either file, or a schema change without a regenerate, fails here.
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+for (const [name, render] of [
+ ["settings.json.example", renderSettingsExample],
+ ["SETTINGS.md", renderSettingsMarkdown],
+] as const) {
+ test(`${name} is what the schema generates`, () => {
+ const committed = readFileSync(path.join(REPO, name), "utf8");
+ assert.equal(
+ committed,
+ render(),
+ `${name} is stale: run pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts`,
+ );
+ });
+}
+
+test("the example parses back to the defaults", async () => {
+ const { siteSettingsSchema, defaultSiteSettings } = await import("./settingsSchema");
+ const parsed = siteSettingsSchema.parse(JSON.parse(renderSettingsExample()));
+ assert.deepEqual(parsed, defaultSiteSettings());
+});
+
+// EVERY FIELD, NOT ONLY THE 31 TOP-LEVEL ONES. The *_FIELD_DOCS records are
+// complete by type (FieldDocs<T> requires one entry per key); these two check
+// the wiring — that every object-valued block has a key table, and that a
+// block's table names every key its default actually carries.
+test("every object-valued block has a nested key table", async () => {
+ const { defaultSiteSettings } = await import("./settingsSchema");
+ const { blockTables } = await import("./settingsDocs");
+ const d = defaultSiteSettings() as Record<string, unknown>;
+ const tables = blockTables(defaultSiteSettings()) as Record<string, unknown[]>;
+ for (const [key, value] of Object.entries(d)) {
+ if (value === null || typeof value !== "object") continue;
+ assert.ok((tables[key]?.length ?? 0) > 0, `${key} has no key table`);
+ }
+});
+
+test("a block table documents every key its default carries", async () => {
+ const { defaultSiteSettings } = await import("./settingsSchema");
+ const { blockTables } = await import("./settingsDocs");
+ const d = defaultSiteSettings() as Record<string, unknown>;
+ for (const [key, list] of Object.entries(blockTables(defaultSiteSettings()))) {
+ const first = list?.[0];
+ if (!first?.defaults || key === "autoQueue") continue;
+ const value = d[key] as Record<string, unknown>;
+ for (const field of Object.keys(value)) {
+ assert.ok(field in first.docs, `${key}.${field} is undocumented`);
+ }
+ }
+});
diff --git a/common/lib/settingsDocs.ts b/common/lib/settingsDocs.ts
@@ -0,0 +1,305 @@
+// THE TWO FILES GENERATED FROM THE SETTINGS SCHEMA: settings.json.example and
+// the key table SETTINGS.md, both at the repo root.
+//
+// Pure renderers — `common/bin/settings-example.ts` writes them, and
+// `settingsDocs.test.ts` asserts the committed files are byte-identical to what
+// these return, so neither can be edited by hand without the test failing.
+//
+// Everything comes from `siteSettingsSchema` (./settingsSchema.ts): the keys and
+// their order from its shape, the defaults from `defaultSiteSettings()`, the
+// prose from each field's `.describe()`. Changing a default or a description is
+// a schema edit followed by regenerating, never an edit here.
+
+import {
+ ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS,
+ ATTRIBUTION_SETTINGS_FIELD_DOCS,
+ BACKFILL_SETTINGS_FIELD_DOCS,
+ BUILD_PIPELINE_SETTINGS_FIELD_DOCS,
+ DIARIZATION_SETTINGS_FIELD_DOCS,
+ DIGEST_SETTINGS_FIELD_DOCS,
+ SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS,
+ SOCIAL_LINK_FIELD_DOCS,
+ SYNC_SCHEDULER_SETTINGS_FIELD_DOCS,
+ defaultSiteSettings,
+ siteSettingsSchema,
+ type SiteSettings,
+} from "./settingsSchema";
+import {
+ AUTO_QUEUE_MATCH_FIELD_DOCS,
+ AUTO_QUEUE_NODE_FIELD_DOCS,
+ AUTO_QUEUE_POLICY_FIELD_DOCS,
+ LANES,
+} from "./autoQueueTypes";
+import {
+ CHANNEL_AUTO_PAUSE_FIELD_DOCS,
+ CHANNEL_FOCUS_FIELD_DOCS,
+ CHANNEL_PRIORITY_ENTRY_FIELD_DOCS,
+ CHANNEL_PRIORITY_FIELD_DOCS,
+} from "./channelPriority";
+import {
+ LLM_WORKER_CONFIG_FIELD_DOCS,
+ REMOTE_WORKER_CONFIG_FIELD_DOCS,
+ WORKER_FIELD_DOCS,
+} from "./workers";
+import { APP_INSTANCE_CONFIG_FIELD_DOCS } from "./transcriptionApps";
+import { DIGEST_APP_CONFIG_FIELD_DOCS } from "./digest";
+import {
+ STORAGE_LOCATION_FIELD_DOCS,
+ STORAGE_SETTINGS_FIELD_DOCS,
+ STORAGE_VOLUME_FIELD_DOCS,
+} from "./storageLocations";
+
+// `workers` IS LEFT OUT OF THE EXAMPLE, and that is the one place the example
+// is not the literal default object. Its default is `[]`, and a settings.json
+// that SPELLS `workers: []` READS as "no transcription workers" — until the
+// next save, when writeSettings' worker shadow synthesizes one from
+// `transcriptionApp`; in between, auto-transcribe does nothing, silently. A file
+// that does not name the key gets that worker list synthesized on read (see
+// getSettings), which is what a template copied to settings.json should give.
+export const EXAMPLE_OMITTED_KEYS = ["workers"] as const;
+
+export function renderSettingsExample(): string {
+ const d = defaultSiteSettings() as Record<string, unknown>;
+ for (const key of EXAMPLE_OMITTED_KEYS) delete d[key];
+ return JSON.stringify(d, null, 2) + "\n";
+}
+
+function isScalar(v: unknown): boolean {
+ return v === null || typeof v !== "object";
+}
+
+// The default as a table cell: a scalar inline, an empty container inline,
+// anything larger by reference to its section.
+function defaultCell(v: unknown): string {
+ if (isScalar(v)) return "`" + JSON.stringify(v) + "`";
+ const json = JSON.stringify(v);
+ if (json === "[]" || json === "{}") return "`" + json + "`";
+ return Array.isArray(v) ? "list — see below" : "object — see below";
+}
+
+// A description inside a table cell: one line, pipes escaped, paragraphs kept.
+function cell(text: string): string {
+ return text.replace(/\|/g, "\\|").replace(/\n\n/g, "<br><br>").replace(/\n/g, " ");
+}
+
+// One nested key table. `defaults(key)` answers the Default column; a table of
+// per-entry fields (list items, map values, tree nodes) has no defaults — each
+// entry spells its own — and says so.
+type KeyTable = {
+ path: string;
+ docs: Readonly<Record<string, string>>;
+ defaults?: (key: string) => string;
+};
+
+function fromObject(obj: unknown): (key: string) => string {
+ const r = (obj ?? {}) as Record<string, unknown>;
+ return (key) => (key in r ? defaultCell(r[key]) : "absent");
+}
+
+// A lane-policy field's default can differ per lane (`held`, `order`, `root`),
+// and that difference is exactly what a reader needs to see.
+function perLane(d: SiteSettings): (key: string) => string {
+ return (key) => {
+ const cells = LANES.map((lane) => {
+ const policy = d.autoQueue[lane] as Record<string, unknown>;
+ return key in policy ? defaultCell(policy[key]) : "absent";
+ });
+ if (cells.every((c) => c === cells[0])) return cells[0];
+ return LANES.map((lane, i) => `${lane} ${cells[i]}`).join("<br>");
+ };
+}
+
+export function blockTables(d: SiteSettings): Partial<Record<keyof SiteSettings, KeyTable[]>> {
+ return {
+ transcriptionApps: [
+ { path: "transcriptionApps.<appId>", docs: APP_INSTANCE_CONFIG_FIELD_DOCS },
+ ],
+ workers: [
+ { path: "workers[]", docs: WORKER_FIELD_DOCS },
+ { path: "workers[].config", docs: APP_INSTANCE_CONFIG_FIELD_DOCS },
+ { path: "workers[].remote", docs: REMOTE_WORKER_CONFIG_FIELD_DOCS },
+ { path: "workers[].llm", docs: LLM_WORKER_CONFIG_FIELD_DOCS },
+ ],
+ archiveStorage: [
+ {
+ path: "archiveStorage",
+ docs: ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.archiveStorage),
+ },
+ ],
+ syncScheduler: [
+ {
+ path: "syncScheduler",
+ docs: SYNC_SCHEDULER_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.syncScheduler),
+ },
+ ],
+ autoQueue: [
+ {
+ path: "autoQueue.<lane>",
+ docs: AUTO_QUEUE_POLICY_FIELD_DOCS,
+ defaults: perLane(d),
+ },
+ { path: "autoQueue.<lane>.root (tree nodes)", docs: AUTO_QUEUE_NODE_FIELD_DOCS },
+ { path: "autoQueue.<lane>.root … .match", docs: AUTO_QUEUE_MATCH_FIELD_DOCS },
+ ],
+ channelPriority: [
+ {
+ path: "channelPriority",
+ docs: CHANNEL_PRIORITY_FIELD_DOCS,
+ defaults: fromObject(d.channelPriority),
+ },
+ { path: "channelPriority.focus", docs: CHANNEL_FOCUS_FIELD_DOCS },
+ { path: "channelPriority.channels.<slug>", docs: CHANNEL_PRIORITY_ENTRY_FIELD_DOCS },
+ {
+ path: "channelPriority.channels.<slug>.autoPaused",
+ docs: CHANNEL_AUTO_PAUSE_FIELD_DOCS,
+ },
+ ],
+ socialLinks: [{ path: "socialLinks[]", docs: SOCIAL_LINK_FIELD_DOCS }],
+ savedVideoBackup: [
+ {
+ path: "savedVideoBackup",
+ docs: SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.savedVideoBackup),
+ },
+ ],
+ storage: [
+ {
+ path: "storage",
+ docs: STORAGE_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.storage),
+ },
+ { path: "storage.locations[]", docs: STORAGE_LOCATION_FIELD_DOCS },
+ { path: "storage.locations[].volume", docs: STORAGE_VOLUME_FIELD_DOCS },
+ ],
+ buildPipeline: [
+ {
+ path: "buildPipeline",
+ docs: BUILD_PIPELINE_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.buildPipeline),
+ },
+ ],
+ digest: [
+ { path: "digest", docs: DIGEST_SETTINGS_FIELD_DOCS, defaults: fromObject(d.digest) },
+ { path: "digest.apps.<appId>", docs: DIGEST_APP_CONFIG_FIELD_DOCS },
+ ],
+ diarization: [
+ {
+ path: "diarization",
+ docs: DIARIZATION_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.diarization),
+ },
+ ],
+ backfill: [
+ {
+ path: "backfill",
+ docs: BACKFILL_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.backfill),
+ },
+ ],
+ attribution: [
+ {
+ path: "attribution",
+ docs: ATTRIBUTION_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.attribution),
+ },
+ ],
+ };
+}
+
+function renderTable(out: string[], table: KeyTable): void {
+ out.push(`#### \`${table.path}\``);
+ out.push("");
+ if (table.defaults) {
+ out.push("| Key | Default | Description |");
+ out.push("|---|---|---|");
+ for (const [key, text] of Object.entries(table.docs)) {
+ out.push(`| \`${key}\` | ${table.defaults(key)} | ${cell(text)} |`);
+ }
+ } else {
+ out.push("Per entry — each entry spells its own values.");
+ out.push("");
+ out.push("| Key | Description |");
+ out.push("|---|---|");
+ for (const [key, text] of Object.entries(table.docs)) {
+ out.push(`| \`${key}\` | ${cell(text)} |`);
+ }
+ }
+ out.push("");
+}
+
+export function renderSettingsMarkdown(): string {
+ const d = defaultSiteSettings();
+ const values = d as Record<string, unknown>;
+ const shape = siteSettingsSchema.shape as Record<
+ string,
+ { description?: string }
+ >;
+ const tables = blockTables(d);
+ const keys = Object.keys(shape);
+ const out: string[] = [];
+ out.push("# settings.json keys");
+ out.push("");
+ out.push(
+ "<!-- GENERATED by common/bin/settings-example.ts from common/lib/settingsSchema.ts and the *_FIELD_DOCS records beside each block type — do not edit by hand. -->",
+ );
+ out.push("");
+ out.push(
+ "Global operational settings shared by every site this editor powers, " +
+ "persisted to `settings.json` at the repo root (or `$SETTINGS_FILE`). " +
+ "Per-site presentation lives in `sites/<id>/site.json`. Every key is " +
+ "optional: a missing key reads as its default, an ill-typed one is " +
+ "coerced to its default or clamped, and an unknown one is dropped on the " +
+ "next save.",
+ );
+ out.push("");
+ out.push(
+ "Regenerate this file and `settings.json.example` with " +
+ "`pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts`.",
+ );
+ out.push("");
+ out.push(
+ "`settings.json.example` is the default object with one key left out, " +
+ "`workers`: a file that does not name it gets a worker list synthesized " +
+ "from `transcriptionApp` on read. A file that spells `workers: []` READS " +
+ "as no transcription at all — until the next save, when the writer " +
+ "synthesizes a worker the same way.",
+ );
+ out.push("");
+ out.push(
+ "A copied example PINS every default it spells — including each lane's " +
+ "`autoQueue.<lane>.held` — so a default changed in a later release will " +
+ "not reach that file. Delete any key you would rather have track the " +
+ "defaults.",
+ );
+ out.push("");
+ out.push("| Key | Default |");
+ out.push("|---|---|");
+ for (const key of keys) {
+ out.push(`| [\`${key}\`](#${key.toLowerCase()}) | ${defaultCell(values[key])} |`);
+ }
+ out.push("");
+ for (const key of keys) {
+ out.push(`## \`${key}\``);
+ out.push("");
+ out.push(shape[key].description ?? "");
+ out.push("");
+ const v = values[key];
+ if (isScalar(v)) {
+ out.push(`Default: \`${JSON.stringify(v)}\``);
+ out.push("");
+ } else {
+ for (const table of tables[key as keyof SiteSettings] ?? []) {
+ renderTable(out, table);
+ }
+ out.push("Default:");
+ out.push("");
+ out.push("```json");
+ out.push(JSON.stringify(v, null, 2));
+ out.push("```");
+ out.push("");
+ }
+ }
+ return out.join("\n");
+}
diff --git a/common/lib/settingsFieldSchemas.ts b/common/lib/settingsFieldSchemas.ts
@@ -0,0 +1,63 @@
+// THE ZOD SEAMS FOR THE THREE SANITIZERS THAT LIVE OUTSIDE settings.ts.
+//
+// `settings.json` has three blocks whose parsers have homes of their own and
+// callers of their own: `workers` (lib/workers.ts — the pool and the settings
+// form), `channelPriority` (lib/channelPriority.ts — storageWatch and the
+// channels actions call `sanitizeChannelPriority(value: unknown)` directly) and
+// `autoQueue` (lib/autoQueueSchema.ts — the picker's tests pin it). Each keeps
+// its home and its signature. What this file adds is ONE schema per block, so
+// `lib/settingsSchema.ts` can compose them with the other twenty-eight fields
+// the same way it composes everything: zod supplies the plumbing (the key, the
+// strip of unknown siblings), the existing sanitizer supplies the arithmetic —
+// and the totality. Nothing was re-implemented, so nothing could drift.
+//
+// WHY A SEPARATE FILE, and not a `workersSchema` export beside each sanitizer:
+// all three homes are imported AS VALUES by `"use client"` forms
+// (WorkersConfigForm, ChannelTierSelect, LadderRung, …). A module-level
+// `import { z } from "zod"` there would put zod in a browser bundle for no
+// reason. Only server code imports this file — lib/settingsSchema.ts, which
+// itself is only reached through lib/settings.ts (node:fs at module scope).
+
+import { z } from "zod";
+import { sanitizeWorkers, type Worker } from "./workers";
+import {
+ sanitizeChannelPriority,
+ type ChannelPriority,
+} from "./channelPriority";
+import { sanitizeAutoQueue } from "./autoQueueSchema";
+import type { AutoQueueSettings } from "./autoQueueTypes";
+
+// A settings FIELD: any JSON value in, a legal value out.
+//
+// TOTALITY COMES FROM THE COERCION, NOT FROM ZOD. `z.unknown()` accepts every
+// input, so its `.catch(undefined)` never fires, and zod does not guard the
+// transform: a coercion that threw would throw out of `parse`. What makes a
+// settings read never throw is that every coercion passed here — each clamp and
+// sanitizer — is itself total over `unknown`. The `.catch` is kept only so the
+// field keeps its shape if `z.unknown()` is ever swapped for a validating
+// schema.
+//
+// What zod does contribute: `z.unknown()` accepts a missing key, and zod 4
+// still runs the transform for it and emits the key — so an absent field is
+// DEFAULTED, not dropped, exactly as `defaults()` used to fill it — and unknown
+// sibling keys are stripped by the enclosing object.
+//
+// There is deliberately no `.default()` anywhere: a default applies only to
+// `undefined`, and every coercion here already decides that case itself — the
+// place "a zero limit is a hold" lives is inside the clamp, not in a fallback
+// zod would apply around it.
+export function settingsField<T>(coerce: (value: unknown) => T) {
+ return z.unknown().catch(undefined).transform((value): T => coerce(value));
+}
+
+export const workersSchema = settingsField(
+ (value): Worker[] => sanitizeWorkers(value),
+);
+
+export const channelPrioritySchema = settingsField(
+ (value): ChannelPriority => sanitizeChannelPriority(value),
+);
+
+export const autoQueueSchema = settingsField(
+ (value): AutoQueueSettings => sanitizeAutoQueue(value),
+);
diff --git a/common/lib/settingsSchema.test.ts b/common/lib/settingsSchema.test.ts
@@ -0,0 +1,425 @@
+// THE SETTINGS SCHEMA: its shape, its boundaries, its migrations, and the
+// promise that a read never throws.
+//
+// Run with: node_modules/.bin/tsx --test common/lib/settingsSchema.test.ts
+//
+// SETTINGS SEAM, same arrangement as ./settingsWrite.test.ts: `getPaths()`
+// memoizes at module scope, so SETTINGS_FILE is set before anything imports
+// ./settings, and the module is imported dynamically below.
+
+import { mkdtempSync, writeFileSync } from "node:fs";
+import { rm } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { test, after } from "node:test";
+import assert from "node:assert/strict";
+
+const ROOT = mkdtempSync(path.join(os.tmpdir(), "settings-schema-"));
+process.env.TRANSCRIPTS_DIR = ROOT;
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+
+import type { SiteSettings } from "./settings";
+import type { AppInstanceConfig } from "./transcriptionApps";
+import type { Worker } from "./workers";
+import type { AutoQueueSettings } from "./autoQueueTypes";
+import type { ChannelPriority } from "./channelPriority";
+import type { CookieMode } from "./cookiePolicy";
+import type { DownloadFormatPreset } from "../ytdlp/downloadFormat";
+import type { StorageSettings } from "./storageLocations";
+import type {
+ AttributionSettings,
+ BackfillSettings,
+ BuildPipelineSettings,
+ DiarizationSettings,
+ DigestSettings,
+ ReportDebouncePreset,
+ SavedVideoBackupSettings,
+ SocialLink,
+ SyncSchedulerSettings,
+} from "./settingsSchema";
+
+const S = await import("./settings");
+const { siteSettingsSchema, defaultSiteSettings, defaults, getSettings } = S;
+
+after(() => rm(ROOT, { recursive: true, force: true }));
+
+// ── THE SHAPE, PINNED ────────────────────────────────────────────────────────
+//
+// `SiteSettings` is `z.infer<typeof siteSettingsSchema>` now. This is the type
+// as it was hand-written before slice 4a, spelled out literally, so a schema
+// edit that changes what 200 importers see is a tsc error here — not a surprise
+// somewhere else.
+//
+// ONE DELIBERATE DIFFERENCE: `archiveStorage` was `archiveStorage?: {…}`. It was
+// never absent at runtime (defaults(), getSettings and writeSettings all
+// emitted it), and zod 4 cannot express "optional key that is always emitted",
+// so it is required now. Nothing in the repo relied on the `?`.
+type PreSchemaSiteSettings = {
+ adminTitle: string;
+ maxTranscriptPageBytes: number;
+ transcriptionApp: string;
+ transcriptionApps: Record<string, AppInstanceConfig>;
+ workers: Worker[];
+ cookiesFromBrowser: string;
+ cookieMode: CookieMode;
+ sleepBetweenDownloadsSeconds: number;
+ downloadFormat: DownloadFormatPreset;
+ minFreeDiskGB: number;
+ resumeMarginGB: number;
+ parallelTranscriptions: number;
+ inlineTranscribeOnFallback: boolean;
+ skipLiveDownloads: boolean;
+ verifyAvailabilityBeforeClean: boolean;
+ buildArchives: boolean;
+ archiveStorage: { bucket: string; publicBaseUrl: string };
+ reportDebouncePreset: ReportDebouncePreset;
+ autoRefreshIntervalSeconds: number;
+ syncScheduler: SyncSchedulerSettings;
+ autoQueue: AutoQueueSettings;
+ channelPriority: ChannelPriority;
+ socialLinks: SocialLink[];
+ homepageUrl: string;
+ savedVideoBackup: SavedVideoBackupSettings;
+ storage: StorageSettings;
+ buildPipeline: BuildPipelineSettings;
+ digest: DigestSettings;
+ diarization: DiarizationSettings;
+ backfill: BackfillSettings;
+ attribution: AttributionSettings;
+};
+
+// Bracketed so the conditional does not distribute (see commit 8c43231).
+type Same<A, B> = [A] extends [B] ? ([B] extends [A] ? true : false) : false;
+const shapeUnchanged: Same<SiteSettings, PreSchemaSiteSettings> = true;
+
+test("SiteSettings keeps its 31 fields, in file order", () => {
+ assert.equal(shapeUnchanged, true);
+ assert.deepEqual(Object.keys(siteSettingsSchema.shape), [
+ "adminTitle",
+ "maxTranscriptPageBytes",
+ "transcriptionApp",
+ "transcriptionApps",
+ "workers",
+ "cookiesFromBrowser",
+ "cookieMode",
+ "sleepBetweenDownloadsSeconds",
+ "downloadFormat",
+ "minFreeDiskGB",
+ "resumeMarginGB",
+ "parallelTranscriptions",
+ "inlineTranscribeOnFallback",
+ "skipLiveDownloads",
+ "verifyAvailabilityBeforeClean",
+ "buildArchives",
+ "archiveStorage",
+ "reportDebouncePreset",
+ "autoRefreshIntervalSeconds",
+ "syncScheduler",
+ "autoQueue",
+ "channelPriority",
+ "socialLinks",
+ "homepageUrl",
+ "savedVideoBackup",
+ "storage",
+ "buildPipeline",
+ "digest",
+ "diarization",
+ "backfill",
+ "attribution",
+ ]);
+ // A parsed object carries every key, in that order — writeSettings writes
+ // exactly this, so the order is the on-disk order.
+ assert.deepEqual(
+ Object.keys(defaults()),
+ Object.keys(siteSettingsSchema.shape),
+ );
+});
+
+test("every field has a description (SETTINGS.md is generated from them)", () => {
+ for (const [key, field] of Object.entries(siteSettingsSchema.shape)) {
+ assert.ok(
+ typeof field.description === "string" && field.description.length > 20,
+ `${key} has no .describe()`,
+ );
+ }
+});
+
+test("defaults() and defaultSiteSettings() are the schema's answer for an empty file", () => {
+ assert.deepEqual(defaultSiteSettings(), siteSettingsSchema.parse({}));
+ assert.deepEqual(defaults(), defaultSiteSettings());
+ const d = defaults();
+ assert.equal(d.adminTitle, S.DEFAULT_ADMIN_TITLE);
+ assert.equal(d.maxTranscriptPageBytes, S.TRANSCRIPT_PAGE_DEFAULT_BYTES);
+ assert.deepEqual(d.workers, []);
+ assert.deepEqual(d.archiveStorage, { bucket: "", publicBaseUrl: "" });
+ assert.equal(d.skipLiveDownloads, true);
+ assert.equal(d.inlineTranscribeOnFallback, false);
+});
+
+// ── THE CLAMP BOUNDARIES ─────────────────────────────────────────────────────
+
+type Case = [input: unknown, expected: number];
+
+function checkField(
+ key: keyof SiteSettings,
+ cases: Case[],
+): void {
+ for (const [input, expected] of cases) {
+ const got = siteSettingsSchema.parse({ [key]: input })[key];
+ assert.equal(got, expected, `${key}: ${JSON.stringify(input)} -> ${got}`);
+ }
+}
+
+// min / max / NaN / string / null / absent, for each clamped top-level field.
+test("maxTranscriptPageBytes clamps into [256 KiB, 20 MiB]", () => {
+ checkField("maxTranscriptPageBytes", [
+ [0, S.TRANSCRIPT_PAGE_MIN_BYTES],
+ [S.TRANSCRIPT_PAGE_MIN_BYTES - 1, S.TRANSCRIPT_PAGE_MIN_BYTES],
+ [S.TRANSCRIPT_PAGE_MIN_BYTES, S.TRANSCRIPT_PAGE_MIN_BYTES],
+ [S.TRANSCRIPT_PAGE_HARD_CAP_BYTES, S.TRANSCRIPT_PAGE_HARD_CAP_BYTES],
+ [S.TRANSCRIPT_PAGE_HARD_CAP_BYTES + 1, S.TRANSCRIPT_PAGE_HARD_CAP_BYTES],
+ [1_000_000.7, 1_000_000],
+ [NaN, S.TRANSCRIPT_PAGE_DEFAULT_BYTES],
+ ["9000000", S.TRANSCRIPT_PAGE_DEFAULT_BYTES],
+ [null, S.TRANSCRIPT_PAGE_DEFAULT_BYTES],
+ [undefined, S.TRANSCRIPT_PAGE_DEFAULT_BYTES],
+ ]);
+});
+
+test("sleepBetweenDownloadsSeconds: 0 is off, capped at the max", () => {
+ checkField("sleepBetweenDownloadsSeconds", [
+ [-1, 0],
+ [0, 0],
+ [S.SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS, S.SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS],
+ [S.SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS + 1, S.SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS],
+ [2.9, 2],
+ [NaN, S.SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS],
+ ["5", S.SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS],
+ [null, S.SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS],
+ ]);
+});
+
+test("minFreeDiskGB: 0 disables the gate, capped at the max", () => {
+ checkField("minFreeDiskGB", [
+ [-5, 0],
+ [0, 0],
+ [S.MIN_FREE_DISK_GB_MAX, S.MIN_FREE_DISK_GB_MAX],
+ [S.MIN_FREE_DISK_GB_MAX + 1, S.MIN_FREE_DISK_GB_MAX],
+ [NaN, S.MIN_FREE_DISK_GB_DEFAULT],
+ ["0", S.MIN_FREE_DISK_GB_DEFAULT],
+ [null, S.MIN_FREE_DISK_GB_DEFAULT],
+ ]);
+});
+
+test("resumeMarginGB: 0 disables the hysteresis, capped at the max", () => {
+ checkField("resumeMarginGB", [
+ [-1, 0],
+ [0, 0],
+ [S.RESUME_MARGIN_GB_MAX, S.RESUME_MARGIN_GB_MAX],
+ [S.RESUME_MARGIN_GB_MAX + 1, S.RESUME_MARGIN_GB_MAX],
+ [NaN, S.RESUME_MARGIN_GB_DEFAULT],
+ ["2", S.RESUME_MARGIN_GB_DEFAULT],
+ [null, S.RESUME_MARGIN_GB_DEFAULT],
+ ]);
+});
+
+test("parallelTranscriptions: floored at 1, capped at the max", () => {
+ checkField("parallelTranscriptions", [
+ [0, 1],
+ [-3, 1],
+ [1, 1],
+ [S.PARALLEL_TRANSCRIPTIONS_MAX, S.PARALLEL_TRANSCRIPTIONS_MAX],
+ [S.PARALLEL_TRANSCRIPTIONS_MAX + 1, S.PARALLEL_TRANSCRIPTIONS_MAX],
+ [NaN, S.PARALLEL_TRANSCRIPTIONS_DEFAULT],
+ ["4", S.PARALLEL_TRANSCRIPTIONS_DEFAULT],
+ [null, S.PARALLEL_TRANSCRIPTIONS_DEFAULT],
+ ]);
+});
+
+test("autoRefreshIntervalSeconds: 0 survives as off, the rest clamps", () => {
+ checkField("autoRefreshIntervalSeconds", [
+ [0, 0],
+ [-10, 0],
+ [0.5, 0],
+ [S.AUTO_REFRESH_INTERVAL_MIN_SECONDS, S.AUTO_REFRESH_INTERVAL_MIN_SECONDS],
+ [S.AUTO_REFRESH_INTERVAL_MAX_SECONDS, S.AUTO_REFRESH_INTERVAL_MAX_SECONDS],
+ [S.AUTO_REFRESH_INTERVAL_MAX_SECONDS + 1, S.AUTO_REFRESH_INTERVAL_MAX_SECONDS],
+ [NaN, S.AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS],
+ ["5", S.AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS],
+ [null, S.AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS],
+ ]);
+});
+
+test("syncScheduler.heartbeatSeconds: 0 is off, positive clamps into [MIN, MAX]", () => {
+ const hb = (v: unknown) =>
+ siteSettingsSchema.parse({ syncScheduler: { heartbeatSeconds: v } })
+ .syncScheduler.heartbeatSeconds;
+ assert.equal(hb(0), 0);
+ assert.equal(hb(-1), 0);
+ assert.equal(hb(1), S.SYNC_HEARTBEAT_MIN_SECONDS);
+ assert.equal(hb(S.SYNC_HEARTBEAT_MAX_SECONDS + 1), S.SYNC_HEARTBEAT_MAX_SECONDS);
+ assert.equal(hb(NaN), S.SYNC_HEARTBEAT_DEFAULT_SECONDS);
+ assert.equal(hb("60"), S.SYNC_HEARTBEAT_DEFAULT_SECONDS);
+ assert.equal(hb(null), S.SYNC_HEARTBEAT_DEFAULT_SECONDS);
+});
+
+test("sync cadences that allow zero keep it; the ones that do not floor at 1", () => {
+ const s = siteSettingsSchema.parse({
+ syncScheduler: {
+ fullSweepIntervalMinutes: 0,
+ fullSweepConfirmMaxSuspects: 0,
+ fullSweepShrinkGuardPercent: 0,
+ defaultIntervalMinutes: 0,
+ maxConcurrentSyncs: 0,
+ quietHoursStart: 24,
+ quietHoursEnd: 6,
+ },
+ }).syncScheduler;
+ assert.equal(s.fullSweepIntervalMinutes, 0);
+ assert.equal(s.fullSweepConfirmMaxSuspects, 0);
+ assert.equal(s.fullSweepShrinkGuardPercent, 0);
+ assert.equal(s.defaultIntervalMinutes, 1);
+ assert.equal(s.maxConcurrentSyncs, 1);
+ // An out-of-range hour clears BOTH ends of the window.
+ assert.equal(s.quietHoursStart, null);
+ assert.equal(s.quietHoursEnd, null);
+});
+
+test("enum fields fall back to their default on junk", () => {
+ const s = siteSettingsSchema.parse({
+ cookieMode: "sometimes",
+ downloadFormat: "best",
+ reportDebouncePreset: "instant",
+ transcriptionApp: "no-such-app",
+ });
+ const d = defaults();
+ assert.equal(s.cookieMode, d.cookieMode);
+ assert.equal(s.downloadFormat, "auto");
+ assert.equal(s.reportDebouncePreset, d.reportDebouncePreset);
+ assert.equal(s.transcriptionApp, d.transcriptionApp);
+});
+
+test("booleans: only an explicit value moves them off their default", () => {
+ const s = siteSettingsSchema.parse({
+ inlineTranscribeOnFallback: "yes",
+ skipLiveDownloads: "no",
+ verifyAvailabilityBeforeClean: 0,
+ buildArchives: false,
+ });
+ assert.equal(s.inlineTranscribeOnFallback, false);
+ assert.equal(s.skipLiveDownloads, true);
+ assert.equal(s.verifyAvailabilityBeforeClean, true);
+ assert.equal(s.buildArchives, false);
+});
+
+test("an unknown key is stripped, at the top level and inside a block", () => {
+ const s = siteSettingsSchema.parse({
+ siteTitle: "pre-multi-site",
+ transcriptionsPaused: true,
+ backfill: { enabled: true, concurrency: 2 },
+ }) as Record<string, unknown>;
+ assert.equal("siteTitle" in s, false);
+ assert.equal("transcriptionsPaused" in s, false);
+ assert.deepEqual(s.backfill, { concurrency: 2, allowRedownload: false });
+});
+
+// ── THE LANE GATE AND THE ZERO HOLD ──────────────────────────────────────────
+
+test("held defaults to [false, false, false, true] — backfill ships held", () => {
+ const aq = defaults().autoQueue;
+ assert.deepEqual(
+ [aq.transcription.held, aq.download.held, aq.digest.held, aq.backfill.held],
+ [false, false, false, true],
+ );
+ // And through a file that names the lanes but no gate.
+ const bare = siteSettingsSchema.parse({
+ autoQueue: { transcription: {}, download: {}, digest: {}, backfill: {} },
+ }).autoQueue;
+ assert.deepEqual(
+ [bare.transcription.held, bare.download.held, bare.digest.held, bare.backfill.held],
+ [false, false, false, true],
+ );
+});
+
+// ── getSettings: THE RAW MIGRATIONS AND THE NEVER-THROW ──────────────────────
+
+function withFile(contents: string): SiteSettings {
+ writeFileSync(process.env.SETTINGS_FILE!, contents);
+ return getSettings();
+}
+
+test("getSettings never throws on a file that is not a settings object", () => {
+ // THE EMPTY FILE is the reference, not bare defaults(): an empty file still
+ // runs the absence-keyed migrations (workers synthesized, the two sweep lanes
+ // built by migrateSweepsToLanes), and a non-object file must read as exactly
+ // that. Before slice 4a, a file containing `null` threw.
+ const empty = withFile("{}");
+ assert.equal(empty.workers.length, S.PARALLEL_TRANSCRIPTIONS_DEFAULT);
+ for (const contents of ["{", "null", "[]", "3", "", '"text"']) {
+ let got: SiteSettings | undefined;
+ assert.doesNotThrow(() => {
+ got = withFile(contents);
+ }, `contents ${JSON.stringify(contents)}`);
+ assert.deepEqual(got, empty, `contents ${JSON.stringify(contents)}`);
+ }
+});
+
+test("workers are synthesized ONLY when the key is absent", () => {
+ const absent = withFile(JSON.stringify({ parallelTranscriptions: 3 }));
+ assert.equal(absent.workers.length, 3);
+ assert.ok(absent.workers.every((w) => w.enabled && w.kind === "local"));
+ const empty = withFile(JSON.stringify({ workers: [] }));
+ assert.deepEqual(empty.workers, []);
+});
+
+test("legacy transcribe* fields migrate ONLY when transcriptionApp is absent", () => {
+ const legacy = {
+ transcribeBin: "/opt/chough",
+ transcribeModel: "/models/x.bin",
+ };
+ const migrated = withFile(JSON.stringify(legacy));
+ assert.equal(migrated.transcriptionApp, "chough");
+ assert.equal(migrated.transcriptionApps.chough?.bin, "/opt/chough");
+ assert.equal(migrated.workers[0].appId, "chough");
+
+ const named = withFile(
+ JSON.stringify({ ...legacy, transcriptionApp: "whisper-cpp" }),
+ );
+ assert.equal(named.transcriptionApp, "whisper-cpp");
+ assert.equal(named.transcriptionApps.chough, undefined);
+});
+
+test("storage.mediaRoot migrates ONLY when locations is absent", () => {
+ const migrated = withFile(
+ JSON.stringify({ storage: { mediaRoot: "/mnt/cold" } }),
+ );
+ assert.deepEqual(migrated.storage.locations.map((l) => l.root), ["/mnt/cold"]);
+ assert.equal(migrated.storage.defaultLocationId, "default");
+
+ const listed = withFile(
+ JSON.stringify({ storage: { mediaRoot: "/mnt/cold", locations: [] } }),
+ );
+ assert.deepEqual(listed.storage.locations, []);
+
+ const none = withFile(JSON.stringify({}));
+ assert.deepEqual(none.storage, { locations: [], defaultLocationId: "" });
+});
+
+test("the retired sweep fields migrate onto a lane ONLY when the lane is absent", () => {
+ const migrated = withFile(
+ JSON.stringify({ digest: { sweepEnabled: true, sweepChannels: ["a"] } }),
+ );
+ assert.equal(migrated.autoQueue.digest.enabled, true);
+ // Absence was the trigger, not the default: a file that spells the lane
+ // keeps its own answer even though the sweep flag says otherwise.
+ const spelled = withFile(
+ JSON.stringify({
+ digest: { sweepEnabled: true },
+ autoQueue: { digest: { enabled: false } },
+ }),
+ );
+ assert.equal(spelled.autoQueue.digest.enabled, false);
+ // No sweep flag, no lane: the migration never enables anything.
+ assert.equal(withFile("{}").autoQueue.digest.enabled, false);
+ assert.equal(withFile("{}").autoQueue.backfill.enabled, false);
+});
diff --git a/common/lib/settingsSchema.ts b/common/lib/settingsSchema.ts
@@ -0,0 +1,1557 @@
+// THE SETTINGS SCHEMA — one definition of settings.json, used by the reader,
+// the writer, the example file and the key table.
+//
+// one-core phase 3 slice 4a. What used to be three copies of the same list —
+// the `SiteSettings` type, the `defaults()` literal, and the two field-by-field
+// sanitizing literals in getSettings and writeSettings — is `siteSettingsSchema`
+// below. `SiteSettings` is inferred from it; `defaults()` is `parse({})`;
+// getSettings and writeSettings (lib/settings.ts) both parse through it; and
+// `common/bin/settings-example.ts` generates settings.json.example and SETTINGS.md
+// from it, including each field's `.describe()` text — which is where the
+// comments that used to sit on the `SiteSettings` type now live, so they have one
+// home and cannot drift from the key they describe.
+//
+// ZOD SUPPLIES THE PLUMBING, NOT THE ARITHMETIC. Every field is
+// `settingsField(coerce)` — `z.unknown().catch(undefined).transform(coerce)` —
+// and every `coerce` is the clamp or sanitizer that already existed, reused, so
+// no boundary moved. Each is total over `unknown`, and that — not zod, whose
+// `.catch` cannot fire on `z.unknown()` — is what makes a read never throw.
+// The nested blocks' own fields are documented in the `*_FIELD_DOCS` record
+// beside each block type (here and in workers.ts, channelPriority.ts,
+// autoQueueTypes.ts, storageLocations.ts, transcriptionApps.ts, digest.ts),
+// type-checked complete. Unknown keys are dropped by zod's default strip (never
+// `.passthrough()`), which is what retires a field: a key the schema does not
+// name cannot survive a read or a save.
+//
+// SERVER-ONLY IN PRACTICE: it imports node:path (sanitizeStorage) and is reached
+// through lib/settings.ts, which imports node:fs at module scope. A `"use
+// client"` form that needs a constant imports the constant — never this schema.
+//
+// NOT HERE: the three migrations keyed on a field's ABSENCE in the raw file
+// (sweeps → lanes, mediaRoot → locations, legacy transcribe* → app registry and
+// synthesized workers), because a parsed object cannot tell "absent" from
+// "default". They run in getSettings around the parse. See lib/settings.ts.
+
+import path from "node:path";
+import { z } from "zod";
+// The auto-queue, workers and channel-priority blocks keep their own
+// sanitizers in their own homes; these are their zod seams.
+import {
+ autoQueueSchema,
+ channelPrioritySchema,
+ settingsField,
+ workersSchema,
+} from "./settingsFieldSchemas";
+import { SYNC_INTERVAL_MAX_MINUTES } from "./channelConfig";
+import {
+ isDownloadFormatPreset,
+ type DownloadFormatPreset,
+} from "../ytdlp/downloadFormat";
+import {
+ type AppInstanceConfig,
+ DEFAULT_TRANSCRIPTION_APP_ID,
+ TRANSCRIPTION_APPS,
+} from "./transcriptionApps";
+import { sanitizeWorkerConfig } from "./workers";
+import {
+ INTERNAL_LOCATION_ID,
+ type StorageLocation,
+ type StorageSettings,
+ type StorageVolume,
+} from "./storageLocations";
+import {
+ DEFAULT_DIARIZATION_ENGINE,
+ DEFAULT_DIARIZATION_THRESHOLD,
+ DIARIZATION_BACKENDS,
+ DIARIZATION_ENGINE_IDS,
+ type DiarizationBackend,
+ type DiarizationEngineId,
+} from "./diarization";
+import { ATTRIBUTION_PROMPT_VERSION } from "./attribution";
+import {
+ DEFAULT_COOKIE_MODE,
+ isCookieMode,
+ type CookieMode,
+} from "./cookiePolicy";
+// From the CLIENT-SAFE digest module, deliberately — digestApps.ts imports execa,
+// and settings.ts must stay reachable from anywhere.
+import {
+ CLAUDE_DIGEST_APP_ID,
+ DEFAULT_DIGEST_APP_ID,
+ DEFAULT_DIGEST_TIMESTAMP_MODE,
+ DIGEST_SECTION_KINDS,
+ DIGEST_TIMESTAMP_MODES,
+ isDigestSectionKind,
+ isDigestTimestampMode,
+ type DigestAppConfig,
+ type DigestSectionKind,
+ type DigestTimestampMode,
+} from "./digest";
+import type { FieldDocs } from "./fieldDocs";
+
+export type { Worker } from "./workers";
+export type { AutoQueueSettings } from "./autoQueueTypes";
+export type { ChannelPriority } from "./channelPriority";
+
+// Transcribe placeholder/arg helpers now live with the whisper-cpp app in
+// transcriptionApps.ts. Re-exported here so existing import sites keep working.
+export {
+ type AppInstanceConfig,
+ TRANSCRIBE_PLACEHOLDER_AUDIO,
+ TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE,
+ TRANSCRIBE_PLACEHOLDER_MODEL,
+ TRANSCRIBE_KNOWN_PLACEHOLDERS,
+ DEFAULT_TRANSCRIBE_ARGS,
+ validateTranscribeArgs,
+} from "./transcriptionApps";
+
+// Configuration for speaker attribution — putting names to the speaker turns.
+//
+// OFF by default, and that default is doing real work rather than being
+// cautious. The text-only lane costs roughly one model call per transcript
+// CHUNK, which on this corpus is ~194,000 calls, the same order as the digest
+// sweep — and the digest sweep has completed 0.17% of its own. Arming both at
+// once puts two multi-week lanes on the same 8 GB card with nothing to arbitrate
+// between them (the backfill lane's yield deliberately watches only the
+// transcription lane). Nothing here arms anything; a pilot decides whether the
+// corpus-wide text-only pass is worth 25-55 GPU-days at all.
+// Each field is documented in ATTRIBUTION_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
+export type AttributionSettings = {
+ enabled: boolean;
+ appId: string;
+ model: string;
+ diarizedEnabled: boolean;
+ textOnlyEnabled: boolean;
+ promptVersion: number;
+};
+
+export const ATTRIBUTION_SETTINGS_FIELD_DOCS: FieldDocs<AttributionSettings> = {
+ enabled:
+ "Master switch. Off means the backfill registry reports no attribution " +
+ "work at all — the feature gate every Operation has.",
+ appId:
+ "Which digest app runs the naming. Attribution IS a digest-app workload" +
+ " — constrained JSON decoding over transcript text — so it reuses that " +
+ "registry and that per-app config (settings.digest.apps[appId]) rather " +
+ "than growing a second copy of the ollama URL, context size and " +
+ "timeout.",
+ model:
+ "Model override. Empty = the app's configured model, then its default. " +
+ "It is separate from the digest's because the two workloads may want " +
+ "different sizes, and because it is part of the freshness identity: " +
+ "sharing the digest's model field would make a digest bake-off " +
+ "invalidate every attribution record on disk as a side effect.",
+ diarizedEnabled:
+ "The lanes, separately. Both default OFF even when `enabled` is on, so " +
+ "turning the feature on to look at it cannot start a corpus sweep.\n\n" +
+ "They are not a fallback pair. `diarized` is one call per video and " +
+ "grounded in acoustic clustering; `textOnly` is ~30 calls and guesses " +
+ "at identity across chunk seams. An operator may reasonably want the " +
+ "first forever and the second never.",
+ textOnlyEnabled:
+ "The text-only attribution lane: names speakers from the transcript alone (~30 model calls per video, guessing identity across chunk seams). Default OFF even when `enabled` is on. See `diarizedEnabled` — the two are separate lanes, not a fallback pair.",
+ promptVersion:
+ "The prompt generation a record must match to count as fresh.\n\n" +
+ "Defaults to (and is floored at) ATTRIBUTION_PROMPT_VERSION, the " +
+ "shipped constant. Raising it forces a corpus-wide regeneration without" +
+ " a code change, which is the honest way to redo everything after a " +
+ "prompt tweak. It cannot be set BELOW the shipped constant, and that " +
+ "floor is the lesson from digestPrompt.ts's version 1 -> 2 note: " +
+ "pinning freshness to an older generation freezes output from a " +
+ "superseded prompt into the corpus, looking identical to output from " +
+ "the current one.",
+};
+
+// Configuration for the backfill lane — the generic answer to "a derived-data
+// feature landed and 77,000 existing videos do not have it".
+//
+// WHAT THE LANE GETS is its own queueKey (concurrency with transcription) plus a
+// limit() that returns 0 to stand aside — the same mechanism the digest yield
+// uses, which fails OPEN so a bad read costs contention rather than a deadlock.
+// There is no priority system to join: the registry submits every named queue at
+// concurrency 1 and SchedulerTier only orders work within a single key.
+//
+// THE SHARE IS `concurrency` AND THE LANE'S `autoQueue.backfill.maxWorkers`; the
+// yield is the operation's declared `contendsFor`. Slice 1.3 retired the
+// `weight` scalar that used to mean both — see backfillLimit().
+// Each field is documented in BACKFILL_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
+export type BackfillSettings = {
+ concurrency: number;
+ allowRedownload: boolean;
+};
+
+export const BACKFILL_SETTINGS_FIELD_DOCS: FieldDocs<BackfillSettings> = {
+ concurrency:
+ "Slots the lane may use when it is not standing aside. Kept at 1 by " +
+ "default for the same reason diarization.concurrency is: this is CPU-" +
+ "bound work competing with GPU feeding and the digest sweep for the " +
+ "same 8 threads.",
+ allowRedownload:
+ "Re-acquire media for videos whose input is GONE (audio deleted after " +
+ "transcription). OFF by default and deliberately so: measured on this " +
+ "corpus, 836 videos still have media and ~76,270 would need a re-" +
+ "download — 91x the reachable work, against 45 GB free at 97% full. " +
+ "When on, each re-fetched file is removed in a `finally` as soon as the" +
+ " backfill has used it, unless the video is marked do-not-clean, or " +
+ "unless the auto-transcribe policy would replace its auto-captions " +
+ "(`replaceAutoSubs`, or a leaf on `downloadedAutoSubsOnly`), in which " +
+ "case the audio is kept for that runner.\n\n" +
+ "WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: " +
+ "\"youtube\"` channel — which normally only fetches subtitles — the re-" +
+ "acquire applies a PER-VIDEO transcribe override so yt-dlp lands audio " +
+ "a diarizer can read; the channel's stored config is not changed. " +
+ "Without that override the fetch re-downloads the captions the video " +
+ "already has and lands nothing, which is what happened to ~16,000 " +
+ "videos on eight channels in 2026-08.",
+};
+
+// Configuration for the speaker-diarization capture lane.
+//
+// This exists because AUDIO IS THE ONE PERISHABLE INPUT in the pipeline:
+// cleanAudioFromTranscribed deletes it once a video is transcribed, so
+// diarization has to happen while the audio is still there or not at all. The
+// capture half is deliberately all that ships here — attribution, LLM speaker
+// naming, viewer badges and quote filtering can all be redone later from the
+// saved JSON, whereas the audio cannot.
+// Each field is documented in DIARIZATION_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
+export type DiarizationSettings = {
+ enabled: boolean;
+ inlineAfterTranscribe: boolean;
+ threshold: number;
+ threads: number;
+ engine: DiarizationEngineId;
+ backend: DiarizationBackend;
+ python: string;
+ segModel: string;
+ embModel: string;
+ sortformerBin: string;
+ sortformerModel: string;
+ concurrency: number;
+ maxAudioHours: number;
+};
+
+export const DIARIZATION_SETTINGS_FIELD_DOCS: FieldDocs<DiarizationSettings> = {
+ enabled:
+ "Master switch. OFF by default so a transcription batch can start " +
+ "before this lands, with diarization backfilled over the retained audio" +
+ " afterwards.\n\n" +
+ "Turning it ON also arms the cleanup guard: the Clean-audio sweep stops" +
+ " deleting audio for a transcribed video that has no diarization.json " +
+ "yet. That is the point — it is what keeps the perishable input alive " +
+ "long enough to be captured — but it means enabling this holds disk.",
+ inlineAfterTranscribe:
+ "Run diarization inline in the post-transcribe hook.\n\n" +
+ "OFF by default, and that default is a MEASURED decision, not caution. " +
+ "Measured on this box: GPU transcription runs at 221 s/audio-hour " +
+ "(16.3x realtime, over 3,602 real videos), CPU diarization at ~500-680 " +
+ "s/audio-hour. Diarization is therefore ~2-3x SLOWER than the " +
+ "transcription it follows, so running it inline drops whole-pipeline " +
+ "throughput by roughly 3-4x and leaves the GPU idle while the CPU " +
+ "catches up.\n\n" +
+ "The intended sequence for a large batch is the opposite: leave this " +
+ "off, let the batch transcribe at full GPU speed with `enabled` holding" +
+ " the audio, and diarize afterwards with the backfill pass. Turn it on " +
+ "for steady state, once the arrival rate is a few videos a day rather " +
+ "than a corpus.",
+ threshold:
+ "Clustering threshold — the single most consequential knob, since it " +
+ "decides how many speakers come out. Larger merges more aggressively.\n\n" +
+ "The default is 0.9, NOT sherpa-onnx's own 0.5, and that is measured on" +
+ " this corpus. On a 6-minute excerpt of a two-person interview (known " +
+ "ground truth: 2 speakers), sherpa's default produced 22 clusters; 0.9 " +
+ "produced 6, with the top two at 40%/40% of talk time — recognizably " +
+ "the two hosts. Sweep on the same clip: 0.4→23, 0.5→22, 0.6→17, 0.7→12," +
+ " 0.8→10, 0.9→6.\n\n" +
+ "It still over-splits, which is why this is a capture lane and not an " +
+ "answer: the turns are recorded with the threshold that produced them, " +
+ "so a later attribution pass can re-cluster or re-run without needing " +
+ "the audio back.",
+ threads:
+ "Engine threads per diarize run.",
+ engine:
+ "Which engine runs. \"sherpa-onnx\" is the shipped default and what every" +
+ " sidecar on disk was produced by; \"sortformer\" is the ggml engine " +
+ "built by scripts/build-sortformer.sh.\n\n" +
+ "CHANGING THIS RESTATES THE FRESHNESS IDENTITY (see diarizationTarget)," +
+ " so every sidecar written by the other engine becomes stale and the " +
+ "backfill lane offers to redo it. That is intended — the two disagree " +
+ "about how many speakers exist, and a corpus half-diarized by each is " +
+ "not one corpus — but on the retained audio it is weeks of work, not a " +
+ "toggle.\n\n" +
+ "Why anyone would: on the same file, sherpa at its tuned threshold " +
+ "returns 13 speakers and sortformer returns 4, agreeing on the dominant" +
+ " speaker's share to within half a point (73.1% vs 73.5%). On the " +
+ "corpus's worst case sherpa returns 35 and sortformer 4. Over-splitting" +
+ " is the failure mode this lane has always had, and sortformer is end-" +
+ "to-end rather than clustered, so it does not have it. The cost is a " +
+ "hard ceiling of 4 speakers and ~1.8x the wall clock.",
+ backend:
+ "Compute device for the sortformer engine; ignored by sherpa-onnx, " +
+ "which has no Vulkan compute path on Linux.\n\n" +
+ "\"vulkan\" is 1.5x faster than a thread-tuned CPU run (894 vs 1305 " +
+ "s/audio-hour, measured on this box) and holds 558 MB resident instead " +
+ "of 4.84 GB by keeping weights and activations in VRAM. It also takes " +
+ "~4.4 GB of an 8 GB card, which is why the lane YIELDS to transcription" +
+ " rather than sharing — see controller/digestYield.ts.",
+ python:
+ "Python interpreter for the default sherpa-onnx engine. sherpa-onnx " +
+ "ships wheels only up to cp313, and this box's system python is 3.14 — " +
+ "so this usually points at a dedicated venv rather than `python3`.",
+ segModel:
+ "ONNX model paths for the default engine. Empty = the lane cannot run, " +
+ "which is reported as a skip rather than a failure.",
+ embModel:
+ "ONNX speaker-embedding model path for the sherpa-onnx engine. Empty = the lane cannot run, reported as a `not-configured` skip rather than a failure (same as `segModel`).",
+ sortformerBin:
+ "Binary and model for the sortformer engine, both produced by " +
+ "scripts/build-sortformer.sh. Empty = that engine cannot run, reported " +
+ "as the same \"not-configured\" skip as an unset segModel/embModel.",
+ sortformerModel:
+ "Model for the sortformer engine, produced by scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the same `not-configured` skip as an unset `sortformerBin`.",
+ concurrency:
+ "How many diarize runs may execute at once in the backfill pass. Kept " +
+ "low by default: diarization is CPU-bound and competes with GPU feeding" +
+ " and the digest sweep for the same 8 threads.",
+ maxAudioHours:
+ "Videos longer than this are DEFERRED rather than diarized: reported as" +
+ " a third number that is never summed into reachable work, so a capped " +
+ "corpus can never read as finished.\n\n" +
+ "THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering " +
+ "holds a pairwise distance matrix over speech-segment embeddings — " +
+ "O(n^2) in SEGMENT count — and speaker-turn density varies 40x across " +
+ "this corpus (33-1364 turns/hour), so duration does not actually " +
+ "predict the blowup: a sparse 7h42m video completed while a dense 6h12m" +
+ " one was OOM-killed. Duration is merely the only predictor available " +
+ "for free, from metadata already on disk, BEFORE spending 45 minutes to" +
+ " find out. n^2 at 30k segments is 6.7 GiB and at 40k is 11.9 GiB, " +
+ "which brackets the 10.6 GB and 9.6 GB peaks measured on this 16 GB " +
+ "box.\n\n" +
+ "0 disables the cap. That is where this goes once windowed diarization " +
+ "lands: windowing divides per-window n by the window count, so the " +
+ "matrix falls by its square, and the cap stops being needed rather than" +
+ " being tuned.",
+};
+
+// Configuration for the derived-corpus digest layer. Local-first by decision:
+// `remoteEnabled` gates the metered lane and defaults to false, so nothing here
+// can spend money until it is explicitly turned on.
+// Each field is documented in DIGEST_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
+export type DigestSettings = {
+ remoteEnabled: boolean;
+ longTailSeconds: number;
+ localAppId: string;
+ remoteAppId: string;
+ apps: Record<string, DigestAppConfig>;
+ yieldToTranscription: boolean;
+ yieldToCpuWorkers: boolean;
+ spendCapUsd: number;
+ sections: DigestSectionKind[];
+ timestampMode: DigestTimestampMode;
+ promptVariant: string;
+};
+
+export const DIGEST_SETTINGS_FIELD_DOCS: FieldDocs<DigestSettings> = {
+ remoteEnabled:
+ "Master switch for the metered (remote-api) lane. OFF by default — an " +
+ "opt-in overflow for the long tail or a channel where local quality is " +
+ "poor, never the default path.",
+ longTailSeconds:
+ "Videos longer than this are \"long tail\": 8.2% of the corpus by count, " +
+ "46% of all transcript tokens. The batch's duration-aware ordering and " +
+ "the optional remote overflow both key off it.",
+ localAppId:
+ "The engine each lane uses (ids from common/lib/digestApps.ts).",
+ remoteAppId:
+ "The engine the metered (remote-api) lane uses — an id from common/lib/digestApps.ts. Unknown ids degrade to the default app rather than failing.",
+ apps:
+ "Per-app config, keyed by app id — the same id-keyed sub-record shape " +
+ "as transcriptionApps.",
+ yieldToTranscription:
+ "Yield the GPU to the transcription lane: while transcription is " +
+ "working, the digest batch's limit() returns 0 and the pool idle-waits." +
+ " ON by default, because `digest:local` is deliberately on a different " +
+ "queue from TRANSCRIPTION_QUEUE and so would otherwise run ollama and " +
+ "the transcription engine on the same 8 GB card. See " +
+ "controller/digestYield.ts.",
+ yieldToCpuWorkers:
+ "Whether a busy worker pinned to `device: \"cpu\"` counts as GPU " +
+ "contention.\n\n" +
+ "OFF by default, which is the FIX for a real bug: the yield originally " +
+ "tested only `kind === \"local\"`, so on a box with one GPU worker and " +
+ "two CPU-pinned ones (this box, at parallelTranscriptions 2) the digest" +
+ " lane stopped dead for transcription that competes for zero GPU " +
+ "shaders.\n\n" +
+ "Only an EXPLICIT \"cpu\" is treated as non-contending. A worker with no " +
+ "device set is using the engine binary's own default, which may be the " +
+ "GPU, so it still triggers the yield — the unknown case fails safe.\n\n" +
+ "Composes with `yieldToTranscription`: that is the master switch, this " +
+ "only narrows which workers it reacts to.",
+ spendCapUsd:
+ "Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. " +
+ "Only ever consulted for a metered app.",
+ sections:
+ "Which sections a sweep generates.\n\n" +
+ "Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, " +
+ "and that is not a contradiction: a tag call sends the same transcript " +
+ "as the chapter call before it, so it hits the engine's cached prefix " +
+ "and pays essentially no prefill (+0.4 s across 4 extra calls, against " +
+ "22.4 s for the first 4). All it pays is decode, and a tag list is ~30 " +
+ "output tokens where a chapter list is ~200–290.\n\n" +
+ "The corollary matters more than the number: run them in the SAME pass." +
+ " Tags generated later, on their own, pay full prefill again — measured" +
+ " at 44% of a whole chapters pass, i.e. 3–9× the marginal cost of just " +
+ "including them now.",
+ timestampMode:
+ "How each chunk's transcript markers are numbered — see " +
+ "DigestTimestampMode. Was a scored variable in the bake-off rather than" +
+ " a pre-applied fix; the measurement is in and \"chunk-local\" is now the" +
+ " shipped default.",
+ promptVariant:
+ "A free-text label for a non-default prompt shape, folded into the " +
+ "recorded provenance by digestPromptVariant(). Setting it invalidates " +
+ "every digest generated under a different label, which is exactly what " +
+ "makes a bake-off round re-run its sample instead of skipping it as " +
+ "fresh. Empty = default.",
+};
+
+// "basic" — `pnpm run build` in export/, serialized on the build queue (shared
+// output tree → no safe parallelism).
+// "docker" — isolated per-site container builds (follow-up); enables real
+// parallel multi-site builds capped by maxParallelBuilds.
+export type BuildMode = "basic" | "docker";
+
+// Each field is documented in BUILD_PIPELINE_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
+export type BuildPipelineSettings = {
+ mode: BuildMode;
+ maxParallelBuilds: number;
+ dockerImage: string;
+ dockerfile: string;
+};
+
+export const BUILD_PIPELINE_SETTINGS_FIELD_DOCS: FieldDocs<BuildPipelineSettings> = {
+ mode:
+ "\"basic\" — `pnpm run build` in export/, serialized on the build queue (shared output tree, no safe parallelism). \"docker\" — isolated per-site container builds, parallel up to `maxParallelBuilds`.",
+ maxParallelBuilds:
+ "Cap on concurrent per-site container builds in docker mode. Ignored in" +
+ " basic mode (which is always serial). Clamped to [1, " +
+ "BUILD_MAX_PARALLEL_MAX].",
+ dockerImage:
+ "Tag of the reusable build image (built once, reused for every site).",
+ dockerfile:
+ "Dockerfile path relative to the monorepo root, used to (re)build the " +
+ "image.",
+};
+
+// Each field is documented in SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
+export type SavedVideoBackupSettings = {
+ enabled: boolean;
+ dest: string;
+ intervalMinutes: number;
+};
+
+export const SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS: FieldDocs<SavedVideoBackupSettings> = {
+ enabled:
+ "Master switch for the scheduled backup. A backup can still be run " +
+ "manually when this is false, as long as a destination is set.",
+ dest:
+ "Destination root the store is mirrored into (a local path or any rsync" +
+ " target). Empty disables both scheduled and manual backups.",
+ intervalMinutes:
+ "Cadence (minutes) for the scheduled backup when enabled. Clamped into " +
+ "the sync-interval window; default daily.",
+};
+
+// Each field is documented in SYNC_SCHEDULER_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
+export type SyncSchedulerSettings = {
+ enabled: boolean;
+ defaultIntervalMinutes: number;
+ maxConcurrentSyncs: number;
+ quietHoursStart: number | null;
+ quietHoursEnd: number | null;
+ backoffBaseMinutes: number;
+ backoffMaxMinutes: number;
+ heartbeatSeconds: number;
+ keepLatestCheckIntervalMinutes: number;
+ fullSweepIntervalMinutes: number;
+ fullSweepConfirmMaxSuspects: number;
+ fullSweepShrinkGuardPercent: number;
+};
+
+export const SYNC_SCHEDULER_SETTINGS_FIELD_DOCS: FieldDocs<SyncSchedulerSettings> = {
+ enabled:
+ "Master switch. When false, a tick selects nothing (manual sync still " +
+ "works).",
+ defaultIntervalMinutes:
+ "Fallback cadence (minutes) for channels with no per-channel override.",
+ maxConcurrentSyncs:
+ "Cap on sync jobs running/queued at once. A tick queues at most (cap - " +
+ "currently-active) channels; the rest roll to the next tick. This is " +
+ "also the stagger mechanism that keeps a big due-batch from hitting the" +
+ " source all at once.",
+ quietHoursStart:
+ "Optional local-clock quiet window during which auto-sync is " +
+ "suppressed. Both null = always allowed. The window may wrap past " +
+ "midnight (e.g. start=22, end=6). Hours are [0,23]; the window is " +
+ "[start, end).",
+ quietHoursEnd:
+ "End hour of the quiet window, [0,23], exclusive. See `quietHoursStart`: both must be valid hours or the window is cleared (null = always allowed).",
+ backoffBaseMinutes:
+ "Failure backoff bounds. After N consecutive failed scheduled syncs a " +
+ "channel waits min(base * 2^(N-1), max) minutes before it's eligible " +
+ "again.",
+ backoffMaxMinutes:
+ "Ceiling on the failure backoff (see `backoffBaseMinutes`): a channel waits min(base * 2^(N-1), max) minutes after N consecutive failures. Never below the base.",
+ heartbeatSeconds:
+ "Cadence (seconds) for the editor's in-process heartbeat — the internal" +
+ " timer armed by the instrumentation hook (editor/instrumentation.ts) " +
+ "that calls the scheduler tick directly, so no external cron is needed." +
+ " 0 = off: rely on the external `pnpm sync:tick` heartbeat instead. Any" +
+ " positive value is clamped to [SYNC_HEARTBEAT_MIN_SECONDS, " +
+ "SYNC_HEARTBEAT_MAX_SECONDS]. The env var SYNC_HEARTBEAT_SECONDS " +
+ "overrides this at runtime. See SCHEDULED_SYNC.md.",
+ keepLatestCheckIntervalMinutes:
+ "Cadence (minutes) for the scheduled keep-latest deletion check. For " +
+ "each channel with ChannelConfig.keepLatest > 0, the tick re-probes the" +
+ " kept window for source deletion (checkKeptDeletedAction) at most this" +
+ " often and pins any gone videos as do-not-clean. Clamped into the " +
+ "sync-interval window; default daily. The check shares the same " +
+ "concurrency cap and quiet-hours window as scheduled syncs. See " +
+ "editor/app/scheduler/runTick.ts.",
+ fullSweepIntervalMinutes:
+ "Default cadence (minutes) for the sync FULL SWEEP — the deep pass that" +
+ " re-enumerates a channel's whole listing in one yt-dlp spawn, " +
+ "refreshes the stored `playlist` file, and flags videos that have left " +
+ "the listing into maybe-missing.json. Ordinary syncs stay on the cheap " +
+ "newest-first paged walk; a sync only upgrades itself to a sweep when " +
+ "this interval has elapsed since the channel's lastFullSweepAt. Per-" +
+ "channel override: ChannelConfig.fullSweepIntervalMinutes. 0 = never " +
+ "sweep. Default daily. See common/jobs/deepSync.ts.",
+ fullSweepConfirmMaxSuspects:
+ "Upper bound on how many maybe-missing suspects a full sweep will " +
+ "resolve in-line with the per-video availability probe (deleted vs " +
+ "private vs unlisted). At or under the cap the sweep runs the targeted " +
+ "check itself, so \"Sync all\" surfaces upstream deletions with no extra " +
+ "clicks; over it, the suspects are flagged and left for a manual check " +
+ "rather than firing hundreds of probes inside a sync. 0 = never auto-" +
+ "confirm.",
+ fullSweepShrinkGuardPercent:
+ "Shrink guard: how far a fresh listing may fall below the stored one " +
+ "before it is treated as suspect rather than acted on. Expressed as a " +
+ "percentage of the previous count, floored at SHRINK_ABS_FLOOR entries " +
+ "so ordinary churn on a small channel doesn't trip it. A suspect " +
+ "listing does not rewrite `playlist` or maybe-missing.json and does not" +
+ " count as a sweep — but a SECOND enumeration reporting a similar count" +
+ " confirms it and is accepted, so a genuine mass deletion costs at most" +
+ " one cadence period. 0 = off (the empty-listing rejection still " +
+ "applies). See controller/acceptListing.ts.",
+};
+
+// Each field is documented in SOCIAL_LINK_FIELD_DOCS below (rendered into SETTINGS.md).
+export type SocialLink = {
+ label: string;
+ url: string;
+ svg: string;
+};
+
+export const SOCIAL_LINK_FIELD_DOCS: FieldDocs<SocialLink> = {
+ label:
+ "Visible name, also the accessible label of the icon.",
+ url:
+ "Link target: http(s), mailto: or a site-relative path.",
+ svg:
+ "Inline SVG markup. Normalized on save (width/height stripped, " +
+ "fill=\"currentColor\", aria-hidden) and rejected when unsafe (script, " +
+ "foreignObject, event handlers, javascript: URLs) or when it has no " +
+ "viewBox.",
+};
+
+// Each field is documented in ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
+export type ArchiveStorageSettings = {
+ bucket: string;
+ publicBaseUrl: string;
+};
+
+export const ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS: FieldDocs<ArchiveStorageSettings> = {
+ bucket:
+ "Cloudflare R2 bucket an oversize archive zip is uploaded to on deploy " +
+ "(`wrangler r2 object put`, keyed `<siteId>/archives/<file>`). Blank = " +
+ "no overflow.",
+ publicBaseUrl:
+ "Public base URL of that bucket; the Downloads page links " +
+ "`<publicBaseUrl>/<key>`. Both fields must be set for overflow to " +
+ "happen.",
+};
+
+export const SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS = 600;
+export const SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS = 10;
+
+export const MIN_FREE_DISK_GB_DEFAULT = 5;
+export const MIN_FREE_DISK_GB_MAX = 100000;
+
+// Hysteresis margin for the low-disk gate. 2 GB is deliberately larger than any
+// single scratch file the pipeline writes, so cleaning one up cannot by itself
+// reopen the gate.
+export const RESUME_MARGIN_GB_DEFAULT = 2;
+export const RESUME_MARGIN_GB_MAX = 1000;
+
+export const PARALLEL_TRANSCRIPTIONS_MAX = 16;
+export const PARALLEL_TRANSCRIPTIONS_DEFAULT = 2;
+
+// Passive UI auto-refresh cadence. 0 is a sentinel meaning "disabled"; any other
+// value is clamped into [MIN, MAX] seconds.
+export const AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS = 5;
+export const AUTO_REFRESH_INTERVAL_MIN_SECONDS = 1;
+export const AUTO_REFRESH_INTERVAL_MAX_SECONDS = 600;
+
+// Global snapshot-scheduler debounce presets. `debounceMs` is the quiet-period
+// window after the last report-changing action; `maxWaitMs` caps the total
+// delay under continuous activity (null = no cap, fire purely on the quiet
+// period). Consumed by common/jobs/snapshotScheduler.ts and surfaced in the
+// Settings form.
+export type ReportDebouncePreset = "fast" | "balanced" | "lazy";
+
+export const REPORT_DEBOUNCE_PRESETS: Record<
+ ReportDebouncePreset,
+ { debounceMs: number; maxWaitMs: number | null }
+> = {
+ fast: { debounceMs: 1000, maxWaitMs: null },
+ balanced: { debounceMs: 3000, maxWaitMs: 30000 },
+ lazy: { debounceMs: 10000, maxWaitMs: 60000 },
+};
+
+export const DEFAULT_REPORT_DEBOUNCE_PRESET: ReportDebouncePreset = "fast";
+
+export function isReportDebouncePreset(v: unknown): v is ReportDebouncePreset {
+ return v === "fast" || v === "balanced" || v === "lazy";
+}
+
+export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024;
+export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024;
+export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024;
+
+export const DEFAULT_ADMIN_TITLE = "Transcript Browser Admin";
+
+// Sync-scheduler bounds + defaults. Default cadence is daily; concurrency is
+// conservative so a tick doesn't fan out into the source provider all at once.
+export const SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES = 1440;
+export const SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT = 2;
+export const SYNC_SCHEDULER_MAX_CONCURRENT_MAX = 16;
+export const SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES = 30;
+export const SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES = 1440;
+export const KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES = 1440;
+// Full-sweep defaults. Daily: a sweep is one full enumeration of the channel,
+// far more expensive than the 50-entry page an ordinary sync fetches. The
+// confirm cap keeps an unattended sweep from fanning out into hundreds of
+// per-video probes when a channel's listing changes wholesale.
+export const FULL_SWEEP_DEFAULT_INTERVAL_MINUTES = 1440;
+export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT = 25;
+export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX = 10000;
+// Shrink-guard default: a listing that has lost more than a tenth of its
+// entries (and more than SHRINK_ABS_FLOOR of them) needs a second opinion.
+export const FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT = 10;
+export const FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX = 100;
+export const SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES = 1440;
+
+// Internal-heartbeat cadence bounds. 0 means "off" (use an external cron
+// heartbeat); any other value is clamped into [MIN, MAX] seconds. The floor
+// keeps the in-process timer from busy-looping; the ceiling is one hour.
+export const SYNC_HEARTBEAT_DEFAULT_SECONDS = 0;
+export const SYNC_HEARTBEAT_MIN_SECONDS = 15;
+export const SYNC_HEARTBEAT_MAX_SECONDS = 3600;
+
+export function defaultSyncScheduler(): SyncSchedulerSettings {
+ return {
+ enabled: false,
+ defaultIntervalMinutes: SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES,
+ maxConcurrentSyncs: SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT,
+ quietHoursStart: null,
+ quietHoursEnd: null,
+ backoffBaseMinutes: SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES,
+ backoffMaxMinutes: SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES,
+ heartbeatSeconds: SYNC_HEARTBEAT_DEFAULT_SECONDS,
+ keepLatestCheckIntervalMinutes: KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES,
+ fullSweepIntervalMinutes: FULL_SWEEP_DEFAULT_INTERVAL_MINUTES,
+ fullSweepConfirmMaxSuspects: FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT,
+ fullSweepShrinkGuardPercent: FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT,
+ };
+}
+
+// Clamp an internal-heartbeat cadence: 0 (off) passes through; any positive
+// value is clamped up into [MIN, MAX]; junk falls back to the default.
+export function clampHeartbeatSeconds(value: unknown): number {
+ if (typeof value !== "number" || !Number.isFinite(value)) {
+ return SYNC_HEARTBEAT_DEFAULT_SECONDS;
+ }
+ const n = Math.floor(value);
+ if (n <= 0) return 0;
+ if (n < SYNC_HEARTBEAT_MIN_SECONDS) return SYNC_HEARTBEAT_MIN_SECONDS;
+ if (n > SYNC_HEARTBEAT_MAX_SECONDS) return SYNC_HEARTBEAT_MAX_SECONDS;
+ return n;
+}
+
+function clampHourOrNull(value: unknown): number | null {
+ if (typeof value !== "number" || !Number.isFinite(value)) return null;
+ const n = Math.floor(value);
+ if (n < 0 || n > 23) return null;
+ return n;
+}
+
+// Like clampPositiveInt, but 0 survives as a sentinel ("off"/"never"). Used by
+// the cadences whose disabled state is expressed as a zero rather than a
+// separate boolean.
+function clampIntAllowZero(value: unknown, fallback: number, max: number): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? Math.floor(value)
+ : fallback;
+ if (n <= 0) return 0;
+ if (n > max) return max;
+ return n;
+}
+
+function clampPositiveInt(value: unknown, fallback: number, max: number): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? Math.floor(value)
+ : fallback;
+ if (n < 1) return 1;
+ if (n > max) return max;
+ return n;
+}
+
+// Coerce a raw settings.syncScheduler value into a clean SyncSchedulerSettings,
+// falling back to defaults for missing/ill-typed fields. Quiet hours are only
+// honored when BOTH endpoints are valid hours; otherwise the window is cleared.
+export function sanitizeSyncScheduler(value: unknown): SyncSchedulerSettings {
+ const d = defaultSyncScheduler();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const start = clampHourOrNull(r.quietHoursStart);
+ const end = clampHourOrNull(r.quietHoursEnd);
+ const backoffBase = clampPositiveInt(
+ r.backoffBaseMinutes,
+ d.backoffBaseMinutes,
+ SYNC_INTERVAL_MAX_MINUTES,
+ );
+ return {
+ enabled: r.enabled === true,
+ defaultIntervalMinutes: clampPositiveInt(
+ r.defaultIntervalMinutes,
+ d.defaultIntervalMinutes,
+ SYNC_INTERVAL_MAX_MINUTES,
+ ),
+ maxConcurrentSyncs: clampPositiveInt(
+ r.maxConcurrentSyncs,
+ d.maxConcurrentSyncs,
+ SYNC_SCHEDULER_MAX_CONCURRENT_MAX,
+ ),
+ quietHoursStart: start !== null && end !== null ? start : null,
+ quietHoursEnd: start !== null && end !== null ? end : null,
+ backoffBaseMinutes: backoffBase,
+ // Cap can't sit below the base, or backoff would never grow.
+ backoffMaxMinutes: Math.max(
+ backoffBase,
+ clampPositiveInt(
+ r.backoffMaxMinutes,
+ d.backoffMaxMinutes,
+ SYNC_INTERVAL_MAX_MINUTES,
+ ),
+ ),
+ heartbeatSeconds: clampHeartbeatSeconds(r.heartbeatSeconds),
+ keepLatestCheckIntervalMinutes: clampPositiveInt(
+ r.keepLatestCheckIntervalMinutes,
+ d.keepLatestCheckIntervalMinutes,
+ SYNC_INTERVAL_MAX_MINUTES,
+ ),
+ fullSweepIntervalMinutes: clampIntAllowZero(
+ r.fullSweepIntervalMinutes,
+ d.fullSweepIntervalMinutes,
+ SYNC_INTERVAL_MAX_MINUTES,
+ ),
+ fullSweepConfirmMaxSuspects: clampIntAllowZero(
+ r.fullSweepConfirmMaxSuspects,
+ d.fullSweepConfirmMaxSuspects,
+ FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX,
+ ),
+ fullSweepShrinkGuardPercent: clampIntAllowZero(
+ r.fullSweepShrinkGuardPercent,
+ d.fullSweepShrinkGuardPercent,
+ FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX,
+ ),
+ };
+}
+
+export function defaultSavedVideoBackup(): SavedVideoBackupSettings {
+ return {
+ enabled: false,
+ dest: "",
+ intervalMinutes: SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES,
+ };
+}
+
+// Coerce a raw settings.savedVideoBackup value into a clean
+// SavedVideoBackupSettings. A missing destination forces enabled off, since a
+// backup with nowhere to go is meaningless.
+export function sanitizeSavedVideoBackup(
+ value: unknown,
+): SavedVideoBackupSettings {
+ const d = defaultSavedVideoBackup();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const dest = typeof r.dest === "string" ? r.dest.trim() : "";
+ return {
+ enabled: dest !== "" && r.enabled === true,
+ dest,
+ intervalMinutes: clampPositiveInt(
+ r.intervalMinutes,
+ d.intervalMinutes,
+ SYNC_INTERVAL_MAX_MINUTES,
+ ),
+ };
+}
+
+// Where relocated channel media goes: the named locations.
+//
+// This used to be ONE FIELD, `mediaRoot` — a single absolute string, the cold
+// drive, typed once. It grew into a list of entities because a root alone
+// cannot answer the two questions the operator actually has: is that disk here,
+// and if it came up somewhere else, how do I point the channels at it without
+// ssh and hand edits? A location carries an id, a label, the root, an opt-in
+// `autoRepoint`, and the volume identity learned at its last probe.
+//
+// Still NOT a policy: a channel on a location is not thereby deprioritized, and
+// nothing auto-relocates anything because a location exists.
+//
+// AVAILABILITY IS NEVER STORED HERE. A refresh that wrote "available" would
+// rewrite settings.json — and so bump the pulse revision — every few seconds.
+// The probe (common/lib/storageVolumes.ts) is computed per request; only the
+// `volume` identity is ever written back, and only when it changed.
+//
+// The types live in lib/storageLocations.ts, which is pure: a `"use client"`
+// file may import them, and must not reach storageVolumes.ts (execa).
+export type { StorageLocation, StorageVolume, StorageSettings };
+
+export function defaultStorage(): StorageSettings {
+ return { locations: [], defaultLocationId: "" };
+}
+
+const LOCATION_ID_RE = /^[a-z0-9][a-z0-9-]{0,63}$/;
+
+// "internal" IS TAKEN. It is the synthetic /storage row for the corpus volume
+// (INTERNAL_LOCATION_ID), and the regex above admits it — so a hand-edited
+// settings.json, or an operator typing the obvious word into the New location
+// form, could store a real location under the one id the page assembles for
+// itself. The row would then be built twice, the rollup would count channels
+// into whichever assembled last, and `locationOfDataDir` would start matching
+// unrelocated channels against it.
+function isReservedLocationId(id: string): boolean {
+ return id === INTERNAL_LOCATION_ID;
+}
+
+function sanitizeVolume(value: unknown): StorageVolume | undefined {
+ if (!value || typeof value !== "object") return undefined;
+ const v = value as Record<string, unknown>;
+ const uuid = typeof v.uuid === "string" ? v.uuid.trim() : "";
+ const mountpoint =
+ typeof v.mountpoint === "string" ? v.mountpoint.trim() : "";
+ // No uuid is no identity, and no mountpoint means `root === join(mountpoint,
+ // relPath)` cannot hold — either way the record is not usable for finding the
+ // volume again, so it is dropped rather than half-kept.
+ if (!uuid || !mountpoint) return undefined;
+ const relPath = typeof v.relPath === "string" ? v.relPath.trim() : "";
+ const fstype = typeof v.fstype === "string" ? v.fstype.trim() : "";
+ const label = typeof v.label === "string" ? v.label.trim() : "";
+ return {
+ uuid,
+ ...(fstype ? { fstype } : {}),
+ ...(label ? { label } : {}),
+ mountpoint,
+ relPath,
+ };
+}
+
+// Coerce a raw settings.storage value into a clean StorageSettings.
+//
+// EXISTENCE IS NOT CHECKED, on purpose: the whole point of a cold location is
+// that it is a drive that may not be mounted when settings are read, and a
+// sanitizer that dropped the root on an unmounted platter would silently erase
+// the operator's choice on the next save.
+//
+// ABSOLUTENESS *IS* checked, and a location with a relative root is DROPPED
+// rather than resolved. Resolving it would anchor the location to whatever cwd
+// the reader booted in — a different directory under docker, under a worktree,
+// and under `pnpm dev` — so the same settings.json would name three different
+// drives. The location form rejects a relative path with a message before it
+// ever gets here; this is the last line, not the only one.
+//
+// NESTED ROOTS ARE ALLOWED. "/mnt/platter" and "/mnt/platter/archive" may both
+// be locations; `locationOfDataDir` resolves a channel to the LONGEST matching
+// root. Nothing here rejects the nesting, because the operator who arranges a
+// disk that way means it.
+//
+// A STALE `mediaRoot` SITTING BESIDE `locations` IS IGNORED — it is not merged
+// back in as an extra location. `migrateMediaRootToLocations` reads it exactly
+// once, when `locations` is absent; after that the list is the whole truth, and
+// resurrecting a root the operator deleted would be a bug, not a kindness.
+//
+// ROLLBACK: an older binary sanitizes this block to `{ mediaRoot: "" }` — the
+// locations are dropped and the single cold root comes back blank. One string
+// lost, nothing on disk moved. `cp settings.json settings.json.pre-storage-
+// locations` before the upgrade and a downgrade is a file copy.
+export function sanitizeStorage(value: unknown): StorageSettings {
+ const d = defaultStorage();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const rawList = Array.isArray(r.locations) ? r.locations : [];
+ const locations: StorageLocation[] = [];
+ const seen = new Set<string>();
+ for (const entry of rawList) {
+ if (!entry || typeof entry !== "object") continue;
+ const e = entry as Record<string, unknown>;
+ const id = typeof e.id === "string" ? e.id.trim() : "";
+ if (!LOCATION_ID_RE.test(id) || isReservedLocationId(id) || seen.has(id)) {
+ continue;
+ }
+ const rawRoot = typeof e.root === "string" ? e.root.trim() : "";
+ if (!path.isAbsolute(rawRoot)) continue;
+ // "/mnt/platter/" and "/mnt/platter" are one root; "/" stays "/".
+ const stripped = rawRoot.replace(/\/+$/, "");
+ const root = stripped === "" ? "/" : stripped;
+ const label = typeof e.label === "string" ? e.label.trim() : "";
+ const volume = sanitizeVolume(e.volume);
+ seen.add(id);
+ locations.push({
+ id,
+ label: label || id,
+ root,
+ autoRepoint: e.autoRepoint === true,
+ ...(volume ? { volume } : {}),
+ });
+ }
+ const wanted =
+ typeof r.defaultLocationId === "string" ? r.defaultLocationId.trim() : "";
+ // A default naming a location that is gone falls back to the first one, not
+ // to "": with a location configured, "no default" is never the answer the
+ // operator wanted, and a blank default silently disables every prefill.
+ const defaultLocationId = locations.some((l) => l.id === wanted)
+ ? wanted
+ : (locations[0]?.id ?? "");
+ // THE SAVED-VIDEO STORE'S LOCATION IS NOT FALLEN BACK, and the asymmetry
+ // with `defaultLocationId` above is deliberate. That one is a PREFERENCE, so
+ // picking another location when the named one is gone is helpful. This one is
+ // a RECORD OF WHERE BYTES ARE: pointing it at a different location because
+ // the recorded one was deleted would claim the store had moved when nothing
+ // had. A dangling id sanitizes to "" — "in place" — which is what the disk
+ // says as soon as anybody looks, and the symlink (if any) keeps working
+ // regardless, because the store is reached through it and not through this.
+ const savedWanted =
+ typeof r.savedVideosLocationId === "string"
+ ? r.savedVideosLocationId.trim()
+ : "";
+ const savedVideosLocationId = locations.some((l) => l.id === savedWanted)
+ ? savedWanted
+ : "";
+ return {
+ locations,
+ defaultLocationId,
+ ...(savedVideosLocationId ? { savedVideosLocationId } : {}),
+ };
+}
+
+// 4 hours. Measured: videos over this are 8.2% of the corpus by count but hold
+// 46% of all transcript tokens, so they are where a sweep's wall-clock actually
+// goes and where chunk-seam bugs live.
+export const DIGEST_LONG_TAIL_DEFAULT_SECONDS = 4 * 3600;
+export const DIGEST_LONG_TAIL_MAX_SECONDS = 24 * 3600;
+
+export function defaultDigest(): DigestSettings {
+ return {
+ // OFF. The metered lane is built but never the default — see PLAN.md.
+ remoteEnabled: false,
+ longTailSeconds: DIGEST_LONG_TAIL_DEFAULT_SECONDS,
+ localAppId: DEFAULT_DIGEST_APP_ID,
+ remoteAppId: CLAUDE_DIGEST_APP_ID,
+ // Empty on purpose: every per-app knob falls through to its own default
+ // constant (resolveNumCtx -> DEFAULT_DIGEST_NUM_CTX, now 8192, and
+ // maxCuesForContext sizes the chunk to it). Seeding a copy of those values
+ // here would give the same number two homes and let them drift.
+ apps: {},
+ // ON. Real GPU contention with the transcription engine is a genuine cost
+ // (re-priced: 11.2 s/chunk idle against 24.9 s/chunk on a contended box), so
+ // the safe default is to step aside; turning it off is the deliberate choice.
+ yieldToTranscription: true,
+ // OFF. A CPU-pinned worker is not GPU contention, and treating it as such
+ // stalled the digest lane for nothing. See DigestSettings.yieldToCpuWorkers.
+ yieldToCpuWorkers: false,
+ spendCapUsd: 0,
+ sections: ["chapters"],
+ timestampMode: DEFAULT_DIGEST_TIMESTAMP_MODE,
+ promptVariant: "",
+ };
+}
+
+// Coerce a raw settings.digest.apps value into a clean keyed map of
+// DigestAppConfig. Mirrors sanitizeTranscriptionApps — INCLUDING its
+// Array.isArray guard, without which a JSON array would pass the typeof check and
+// produce numeric-keyed garbage.
+export function sanitizeDigestApps(
+ value: unknown,
+): Record<string, DigestAppConfig> {
+ if (!value || typeof value !== "object" || Array.isArray(value)) return {};
+ const out: Record<string, DigestAppConfig> = {};
+ for (const [id, raw] of Object.entries(value as Record<string, unknown>)) {
+ if (!raw || typeof raw !== "object") continue;
+ const r = raw as Record<string, unknown>;
+ const cfg: DigestAppConfig = {};
+ if (typeof r.bin === "string" && r.bin.trim()) cfg.bin = r.bin.trim();
+ if (typeof r.baseUrl === "string" && r.baseUrl.trim()) {
+ cfg.baseUrl = r.baseUrl.trim();
+ }
+ if (typeof r.model === "string" && r.model.trim()) cfg.model = r.model.trim();
+ if (typeof r.numCtx === "number" && r.numCtx > 0) {
+ cfg.numCtx = Math.floor(r.numCtx);
+ }
+ if (typeof r.temperature === "number" && r.temperature >= 0) {
+ cfg.temperature = r.temperature;
+ }
+ if (typeof r.timeoutMs === "number" && r.timeoutMs > 0) {
+ cfg.timeoutMs = Math.floor(r.timeoutMs);
+ }
+ // Only carried when explicitly set — see DigestAppConfig.think.
+ if (typeof r.think === "boolean") cfg.think = r.think;
+ out[id] = cfg;
+ }
+ return out;
+}
+
+export function sanitizeDigest(value: unknown): DigestSettings {
+ const d = defaultDigest();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const sections = Array.isArray(r.sections)
+ ? (r.sections.filter(isDigestSectionKind) as DigestSectionKind[])
+ : [];
+ return {
+ remoteEnabled: r.remoteEnabled === true,
+ longTailSeconds: clampPositiveInt(
+ r.longTailSeconds,
+ d.longTailSeconds,
+ DIGEST_LONG_TAIL_MAX_SECONDS,
+ ),
+ // Unknown app ids are not rejected here: getDigestApp() is total and falls
+ // back to the local default, so a stale id degrades rather than breaking.
+ localAppId:
+ typeof r.localAppId === "string" && r.localAppId.trim()
+ ? r.localAppId.trim()
+ : d.localAppId,
+ remoteAppId:
+ typeof r.remoteAppId === "string" && r.remoteAppId.trim()
+ ? r.remoteAppId.trim()
+ : d.remoteAppId,
+ apps: sanitizeDigestApps(r.apps),
+ // Defaults to ON when absent — `=== false` rather than `!== true`, so a
+ // settings file written before this field existed keeps the GPU-safe
+ // behaviour instead of silently opting into contention.
+ yieldToTranscription: r.yieldToTranscription !== false,
+ // The OPPOSITE idiom, and deliberately so: `=== true`, so absence falls to
+ // OFF. The field's absence means a settings file written before the CPU-worker
+ // bug was found, and for those files OFF is the FIXED behaviour, not a silent
+ // change of intent — nobody ever asked to stall the digest lane for a CPU
+ // transcription. `yieldToTranscription` still gates the whole thing, so the
+ // GPU-safe default is untouched.
+ yieldToCpuWorkers: r.yieldToCpuWorkers === true,
+ spendCapUsd:
+ typeof r.spendCapUsd === "number" && r.spendCapUsd > 0
+ ? Math.round(r.spendCapUsd * 100) / 100
+ : 0,
+ // An empty/garbage list would silently generate nothing, so fall back to the
+ // default rather than honoring it.
+ sections: sections.length > 0 ? sections : d.sections,
+ timestampMode: isDigestTimestampMode(r.timestampMode)
+ ? r.timestampMode
+ : d.timestampMode,
+ // Trimmed and length-capped: it goes into provenance on every record, and a
+ // runaway value would bloat 119k sidecars.
+ promptVariant:
+ typeof r.promptVariant === "string"
+ ? r.promptVariant.trim().slice(0, 40)
+ : d.promptVariant,
+ };
+}
+
+// Every known section kind, for the settings UI's checkbox list.
+export const DIGEST_SECTION_OPTIONS = DIGEST_SECTION_KINDS;
+export const DIGEST_TIMESTAMP_MODE_OPTIONS = DIGEST_TIMESTAMP_MODES;
+
+export const BUILD_MAX_PARALLEL_DEFAULT = 2;
+export const BUILD_MAX_PARALLEL_MAX = 16;
+export const DEFAULT_BUILD_IMAGE = "yt-dlp-transcript-browser-build";
+export const DEFAULT_BUILD_DOCKERFILE = "Dockerfile.build";
+
+export function isBuildMode(v: unknown): v is BuildMode {
+ return v === "basic" || v === "docker";
+}
+
+export function defaultBuildPipeline(): BuildPipelineSettings {
+ return {
+ mode: "basic",
+ maxParallelBuilds: BUILD_MAX_PARALLEL_DEFAULT,
+ dockerImage: DEFAULT_BUILD_IMAGE,
+ dockerfile: DEFAULT_BUILD_DOCKERFILE,
+ };
+}
+
+// Coerce a raw settings.buildPipeline value into a clean BuildPipelineSettings,
+// falling back to defaults for missing/ill-typed fields.
+export function sanitizeBuildPipeline(value: unknown): BuildPipelineSettings {
+ const d = defaultBuildPipeline();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const dockerImage =
+ typeof r.dockerImage === "string" && r.dockerImage.trim()
+ ? r.dockerImage.trim()
+ : d.dockerImage;
+ const dockerfile =
+ typeof r.dockerfile === "string" && r.dockerfile.trim()
+ ? r.dockerfile.trim()
+ : d.dockerfile;
+ return {
+ mode: isBuildMode(r.mode) ? r.mode : d.mode,
+ maxParallelBuilds: clampPositiveInt(
+ r.maxParallelBuilds,
+ d.maxParallelBuilds,
+ BUILD_MAX_PARALLEL_MAX,
+ ),
+ dockerImage,
+ dockerfile,
+ };
+}
+
+export function defaultBackfill(): BackfillSettings {
+ return {
+ concurrency: 1,
+ // See BackfillSettings.allowRedownload — this one holds disk.
+ allowRedownload: false,
+ };
+}
+
+export function sanitizeBackfill(value: unknown): BackfillSettings {
+ const d = defaultBackfill();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ return {
+ // Clamped rather than rejected: a hand-edited 5 means "as much as possible",
+ // and reading it as 0 would be the opposite of the intent.
+ concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16),
+ allowRedownload: r.allowRedownload === true,
+ };
+}
+
+export function defaultAttribution(): AttributionSettings {
+ return {
+ // OFF, and both lanes OFF under it. See AttributionSettings.
+ enabled: false,
+ appId: DEFAULT_DIGEST_APP_ID,
+ model: "",
+ diarizedEnabled: false,
+ textOnlyEnabled: false,
+ promptVersion: ATTRIBUTION_PROMPT_VERSION,
+ };
+}
+
+export function sanitizeAttribution(value: unknown): AttributionSettings {
+ const d = defaultAttribution();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const str = (v: unknown, fallback: string) =>
+ typeof v === "string" && v.trim() ? v.trim() : fallback;
+ return {
+ enabled: r.enabled === true,
+ appId: str(r.appId, d.appId),
+ // NOT `str(...)` with a fallback: empty is a meaningful value here ("use the
+ // app's own model"), so an empty string must survive rather than reverting
+ // to a default that is also empty by coincidence.
+ model: typeof r.model === "string" ? r.model.trim() : d.model,
+ diarizedEnabled: r.diarizedEnabled === true,
+ textOnlyEnabled: r.textOnlyEnabled === true,
+ // FLOORED at the shipped constant, never merely defaulted. A hand-edited
+ // value below it would pin freshness to a superseded prompt generation and
+ // freeze its output into the corpus — see AttributionSettings.promptVersion.
+ promptVersion:
+ typeof r.promptVersion === "number" && Number.isFinite(r.promptVersion)
+ ? Math.max(ATTRIBUTION_PROMPT_VERSION, Math.floor(r.promptVersion))
+ : d.promptVersion,
+ };
+}
+
+export function defaultDiarization(): DiarizationSettings {
+ return {
+ // OFF. Capture is opt-in: turning it on makes the cleanup sweep start
+ // refusing to delete audio for transcribed-but-undiarized videos, which is
+ // correct but is a disk-pressure decision an operator should make.
+ enabled: false,
+ // OFF — see DiarizationSettings.inlineAfterTranscribe. Diarization is slower
+ // than the transcription it would follow, so inline is the exception.
+ inlineAfterTranscribe: false,
+ // Measured, not sherpa-onnx's 0.5 — see DiarizationSettings.threshold. The
+ // constant lives in lib/diarization.ts because isDiarizationFresh needs it
+ // to normalize an absent recorded threshold; importing it keeps the default
+ // and the comparator from drifting apart.
+ threshold: DEFAULT_DIARIZATION_THRESHOLD,
+ threads: 4,
+ // The engine every sidecar on disk was produced by. Switching is an explicit
+ // decision that restates the freshness identity — see DiarizationSettings.
+ engine: DEFAULT_DIARIZATION_ENGINE,
+ // Only consulted when engine is "sortformer". Defaulting to the GPU is safe
+ // because the lane yields the card to transcription rather than sharing it.
+ backend: "vulkan",
+ python: "python3",
+ segModel: "",
+ embModel: "",
+ sortformerBin: "",
+ sortformerModel: "",
+ concurrency: 1,
+ // OFF, because windowing made it unnecessary — which is what it was always
+ // for. It shipped at 4 hours as a stopgap while long recordings were being
+ // OOM-killed; the engine now processes them in windows and the 6h12m file
+ // that was being killed at 10.5 GB completes at ~2 GB. The knob stays, and
+ // stays honest about what it does, for a machine smaller than this one or a
+ // recording longer than anything measured here.
+ maxAudioHours: 0,
+ };
+}
+
+export function sanitizeDiarization(value: unknown): DiarizationSettings {
+ const d = defaultDiarization();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const str = (v: unknown, fallback: string) =>
+ typeof v === "string" && v.trim() ? v.trim() : fallback;
+ return {
+ enabled: r.enabled === true,
+ inlineAfterTranscribe: r.inlineAfterTranscribe === true,
+ threshold:
+ typeof r.threshold === "number" &&
+ Number.isFinite(r.threshold) &&
+ r.threshold > 0
+ ? r.threshold
+ : d.threshold,
+ threads: clampPositiveInt(r.threads, d.threads, 64),
+ // An unknown engine falls back to the default rather than disabling the lane:
+ // a typo in settings.json must not silently stop diarization, and the default
+ // is the one every existing sidecar already matches.
+ engine: DIARIZATION_ENGINE_IDS.includes(r.engine as DiarizationEngineId)
+ ? (r.engine as DiarizationEngineId)
+ : d.engine,
+ backend: DIARIZATION_BACKENDS.includes(r.backend as DiarizationBackend)
+ ? (r.backend as DiarizationBackend)
+ : d.backend,
+ python: str(r.python, d.python),
+ segModel: str(r.segModel, d.segModel),
+ embModel: str(r.embModel, d.embModel),
+ sortformerBin: str(r.sortformerBin, d.sortformerBin),
+ sortformerModel: str(r.sortformerModel, d.sortformerModel),
+ concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16),
+ // 0 is meaningful here (cap off), so this cannot use clampPositiveInt.
+ // Fractional hours are allowed — the knob is a duration, not a count.
+ maxAudioHours:
+ typeof r.maxAudioHours === "number" &&
+ Number.isFinite(r.maxAudioHours) &&
+ r.maxAudioHours >= 0
+ ? r.maxAudioHours
+ : d.maxAudioHours,
+ };
+}
+
+const SOCIAL_URL_RE = /^(https?:\/\/|mailto:|\/)/i;
+
+// Normalize the family hub URL into a trailing-slash-free absolute http(s) URL.
+// Returns "" for anything that isn't a usable absolute URL (the "no hub" state).
+// Kept local to settings.ts to avoid a settings<->homepage import cycle; mirrors
+// parseHomepageUrl() in homepage.ts.
+export function normalizeHomepageUrl(input: unknown): string {
+ if (typeof input !== "string") return "";
+ const trimmed = input.trim().replace(/\/+$/, "");
+ return /^https?:\/\/\S+/i.test(trimmed) ? trimmed : "";
+}
+
+export function parseSocialLinks(input: unknown): SocialLink[] {
+ if (!Array.isArray(input)) return [];
+ const out: SocialLink[] = [];
+ for (const raw of input) {
+ if (!raw || typeof raw !== "object") continue;
+ const r = raw as Record<string, unknown>;
+ const label = typeof r.label === "string" ? r.label.trim() : "";
+ const url = typeof r.url === "string" ? r.url.trim() : "";
+ const svg = typeof r.svg === "string" ? r.svg : "";
+ if (!label || !url || !svg) continue;
+ if (!SOCIAL_URL_RE.test(url)) continue;
+ out.push({ label, url, svg });
+ }
+ return out;
+}
+
+// Normalize an admin-provided SVG snippet for inline use in the export
+// footer. Returns null on anything that looks unsafe or unrenderable.
+// Steps: trim, allowlist-check, strip width/height, force fill="currentColor"
+// + aria-hidden on the root <svg>. Requires a viewBox so the icon scales.
+export function normalizeSocialSvg(raw: string): string | null {
+ if (typeof raw !== "string") return null;
+ const trimmed = raw.trim();
+ if (!trimmed.startsWith("<svg") || !trimmed.endsWith("</svg>")) return null;
+ if (/<script\b/i.test(trimmed)) return null;
+ if (/<foreignObject\b/i.test(trimmed)) return null;
+ if (/<iframe\b/i.test(trimmed)) return null;
+ if (/javascript:/i.test(trimmed)) return null;
+ if (/\son[a-z]+\s*=/i.test(trimmed)) return null;
+ if (/<\?|<!ENTITY/i.test(trimmed)) return null;
+
+ const openEnd = trimmed.indexOf(">");
+ if (openEnd < 0) return null;
+ let opening = trimmed.slice(0, openEnd);
+ const rest = trimmed.slice(openEnd);
+
+ if (!/\sviewBox\s*=\s*"/i.test(opening)) return null;
+
+ opening = opening.replace(/\s(width|height)\s*=\s*"[^"]*"/gi, "");
+ opening = opening.replace(/\s(width|height)\s*=\s*'[^']*'/gi, "");
+
+ if (!/\sfill\s*=/i.test(opening)) {
+ opening = opening.replace(/^<svg/i, '<svg fill="currentColor"');
+ }
+ if (!/\saria-hidden\s*=/i.test(opening)) {
+ opening = opening.replace(/^<svg/i, '<svg aria-hidden="true"');
+ }
+ return opening + rest;
+}
+
+// normalizeSocialSvg() deliberately STRIPS width/height so the icon scales to its
+// wrapper. The cost is that a viewBox-only <svg> has no intrinsic size, so before
+// the stylesheet loads on a static host it paints at the replaced-element default
+// (huge) — the "flash of giant social icons" FOUC. sizeSocialSvg() re-injects an
+// intrinsic pixel size at RENDER time (existing site.json files already have the
+// attributes stripped, so this must run on read, not just on write). The size is
+// an *attribute*, not inline style, so a wrapper's `w-*`/`h-*` utilities still win
+// once CSS loads — it only governs the pre-CSS first paint.
+export function sizeSocialSvg(svg: string, px = 20): string {
+ if (typeof svg !== "string") return svg;
+ if (/^<svg[^>]*\swidth\s*=/i.test(svg)) return svg; // already sized
+ return svg.replace(/^<svg\b/i, `<svg width="${px}" height="${px}"`);
+}
+
+export function clampSleepBetweenDownloadsSeconds(value: unknown): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? Math.floor(value)
+ : SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS;
+ if (n < 0) return 0;
+ if (n > SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS) {
+ return SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS;
+ }
+ return n;
+}
+
+export function clampMinFreeDiskGB(value: unknown): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? Math.floor(value)
+ : MIN_FREE_DISK_GB_DEFAULT;
+ if (n < 0) return 0;
+ if (n > MIN_FREE_DISK_GB_MAX) return MIN_FREE_DISK_GB_MAX;
+ return n;
+}
+
+export function clampResumeMarginGB(value: unknown): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? Math.floor(value)
+ : RESUME_MARGIN_GB_DEFAULT;
+ if (n < 0) return 0;
+ if (n > RESUME_MARGIN_GB_MAX) return RESUME_MARGIN_GB_MAX;
+ return n;
+}
+
+export function clampParallelTranscriptions(value: unknown): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? Math.floor(value)
+ : PARALLEL_TRANSCRIPTIONS_DEFAULT;
+ if (n < 1) return 1;
+ if (n > PARALLEL_TRANSCRIPTIONS_MAX) return PARALLEL_TRANSCRIPTIONS_MAX;
+ return n;
+}
+
+// 0 means "disabled" and is preserved as-is. Anything else is clamped into the
+// [MIN, MAX] window; a non-finite value falls back to the default cadence.
+export function clampAutoRefreshIntervalSeconds(value: unknown): number {
+ if (typeof value !== "number" || !Number.isFinite(value)) {
+ return AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS;
+ }
+ const n = Math.floor(value);
+ if (n <= 0) return 0;
+ if (n < AUTO_REFRESH_INTERVAL_MIN_SECONDS) {
+ return AUTO_REFRESH_INTERVAL_MIN_SECONDS;
+ }
+ if (n > AUTO_REFRESH_INTERVAL_MAX_SECONDS) {
+ return AUTO_REFRESH_INTERVAL_MAX_SECONDS;
+ }
+ return n;
+}
+
+export function clampPageBytes(value: unknown): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? value
+ : TRANSCRIPT_PAGE_DEFAULT_BYTES;
+ if (n < TRANSCRIPT_PAGE_MIN_BYTES) return TRANSCRIPT_PAGE_MIN_BYTES;
+ if (n > TRANSCRIPT_PAGE_HARD_CAP_BYTES) return TRANSCRIPT_PAGE_HARD_CAP_BYTES;
+ return Math.floor(n);
+}
+
+
+// Coerce a raw settings.transcriptionApps value into a clean keyed map of
+// AppInstanceConfig, dropping unknown/ill-typed fields.
+export function sanitizeTranscriptionApps(
+ value: unknown,
+): Record<string, AppInstanceConfig> {
+ if (!value || typeof value !== "object" || Array.isArray(value)) return {};
+ const out: Record<string, AppInstanceConfig> = {};
+ for (const [id, raw] of Object.entries(value as Record<string, unknown>)) {
+ if (!raw || typeof raw !== "object") continue;
+ out[id] = sanitizeWorkerConfig(raw);
+ }
+ return out;
+}
+
+
+// --- The schema -------------------------------------------------------------
+
+// Global, OPERATIONAL settings shared across every site this editor powers.
+// Per-site presentation (branding, social links, channel groups, membership)
+// lives in sites/<siteId>/site.json — see common/lib/site.ts.
+//
+// FIELD ORDER IS FILE ORDER: zod emits keys in the order they are declared, and
+// writeSettings writes what the schema emits, so reordering these reorders every
+// settings.json on its next save.
+export const siteSettingsSchema = z.object({
+ adminTitle: settingsField((v): string =>
+ typeof v === "string" && v.trim() ? v.trim() : DEFAULT_ADMIN_TITLE).describe(
+ "Title for the EDITOR admin shell only (the editor manages all sites and so is not tied to any one site's branding). Public sites get their own titles from site.json.",
+ ),
+ maxTranscriptPageBytes: settingsField((v): number => clampPageBytes(v)).describe(
+ "Target size (bytes) of one exported transcript page shard — the unit the export site fetches. Clamped into [TRANSCRIPT_PAGE_MIN_BYTES, TRANSCRIPT_PAGE_HARD_CAP_BYTES] (256 KiB – 20 MiB); default 8 MiB.",
+ ),
+ transcriptionApp: settingsField((v): string =>
+ typeof v === "string" && TRANSCRIPTION_APPS[v] ? v : DEFAULT_TRANSCRIPTION_APP_ID).describe(
+ "Active transcription app id (key into TRANSCRIPTION_APPS, e.g. \"whisper-cpp\" or \"chough\"). Selected globally; see common/lib/transcriptionApps.ts.",
+ ),
+ transcriptionApps: settingsField((v): Record<string, AppInstanceConfig> => sanitizeTranscriptionApps(v)).describe(
+ "Per-app configuration, keyed by app id. Each app reads only its own block; a missing block means \"use the app's defaults\". DEPRECATED in favor of `workers` (each local worker carries its own config); kept one release to drive migration and allow rollback. See common/lib/workers.ts.",
+ ),
+ workers: workersSchema.describe(
+ "Configured transcription workers (named processing slots). The scheduler distributes each video to the highest-priority free worker. A settings.json predating this field is migrated to a single enabled worker from the active app (see defaultWorkersFromApps). See common/lib/workers.ts.",
+ ),
+ cookiesFromBrowser: settingsField((v): string => (typeof v === "string" ? v.trim() : "")).describe(
+ "Browser spec (e.g. \"firefox\", \"chrome:Default\") passed to `yt-dlp --cookies-from-browser`. WHEN it is passed is governed by `cookieMode` below. Empty string = no cookies configured. Per-channel override available (ChannelConfig.cookiesFromBrowser).",
+ ),
+ cookieMode: settingsField((v): CookieMode => (isCookieMode(v) ? v : DEFAULT_COOKIE_MODE)).describe(
+ "How yt-dlp invocations use the configured cookies (see common/lib/cookiePolicy.ts): \"always\" passes them on every invocation, \"when-required\" (default; the historical behavior) only to retry an auth/age failure, \"defer\" never in normal runs — auth-gated videos are excluded from batches and collected into the per-channel \"Needs cookies\" bucket for a manual cookie run. Per-channel override available (ChannelConfig.cookieMode).",
+ ),
+ sleepBetweenDownloadsSeconds: settingsField((v): number => clampSleepBetweenDownloadsSeconds(v)).describe(
+ "Pause (seconds) inserted between per-video yt-dlp invocations in managed batch downloads. yt-dlp's own `-t sleep` only paces requests within one invocation, so without this the managed loop hammers the source IP back-to-back. 0 disables. Per-channel override available.",
+ ),
+ downloadFormat: settingsField((v): DownloadFormatPreset =>
+ isDownloadFormatPreset(v) ? v : "auto").describe(
+ "Default yt-dlp `-f` download format for every channel that doesn't set its own (ChannelConfig.downloadFormat). \"auto\" picks per-source: `original` for Odysee (whose HLS rungs are CDN-truncated), `bestaudio/worst` elsewhere. See common/ytdlp/downloadFormat.ts.",
+ ),
+ minFreeDiskGB: settingsField((v): number => clampMinFreeDiskGB(v)).describe(
+ "Minimum free disk space (GB) required on the transcripts data directory for downloads to run. When free space is below this floor, a download job is prevented from starting and a running batch stops launching new videos (the in-flight one finishes). 0 disables the gate. See common/lib/diskSpace.ts.",
+ ),
+ resumeMarginGB: settingsField((v): number => clampResumeMarginGB(v)).describe(
+ "Extra headroom (GB) above minFreeDiskGB that a stopped pipeline must see before it resumes. Resuming at the same number we stopped at flaps — the first restarted download pushes free space back under the floor. This is the hysteresis margin, so \"resumed\" means the operator actually freed something rather than a scratch file being cleaned up. 0 disables the hysteresis (resume at the floor). See diskGate() in common/lib/diskSpace.ts.",
+ ),
+ parallelTranscriptions: settingsField((v): number => clampParallelTranscriptions(v)).describe(
+ "Default number of videos transcribed in parallel when a \"Transcribe missing\" / bucket run doesn't specify its own concurrency. The per-run Concurrency input in the channel UI overrides this for a single run.",
+ ),
+ inlineTranscribeOnFallback: settingsField((v): boolean => v === true).describe(
+ "When true, the no-subs fallback in the managed downloader runs whisper inline immediately after the audio download succeeds. When false (default), audio is left for the next \"Transcribe missing\" pass so a batch download finishes faster and whisper can parallelize.",
+ ),
+ skipLiveDownloads: settingsField((v): boolean => v !== false).describe(
+ "When true (default), managed downloads skip videos that are currently live or scheduled/upcoming, decided from a metadata-only prefetch pass. Finished livestream VODs (was_live) are NOT skipped and download normally. A skip is recorded but not archived, so the next sync/download-missing retries the video once the stream ends. Per-channel override available (ChannelConfig.skipLiveDownloads).",
+ ),
+ verifyAvailabilityBeforeClean: settingsField((v): boolean => v !== false).describe(
+ "Whether the transcribed-audio cleanup sweep checks each candidate is still available upstream before deleting its audio, pinning (do-not-clean) any video found permanently gone. The delete is irreversible and a gone video's audio is irreplaceable, so this defaults to true. Turn it off for an offline or URL-less setup, where the check can never resolve and cleanup would otherwise never delete anything. See verifyBeforeClean.ts.",
+ ),
+ buildArchives: settingsField((v): boolean => v !== false).describe(
+ "Whether site builds generate downloadable transcript/live-chat archive zips (into public/archives, linked on the Downloads page). Global default; a site can opt out via site.json `archives: false`, and a single build can skip via the \"Skip archive zips\" build control. Opt-out: default true.",
+ ),
+ archiveStorage: settingsField((v): ArchiveStorageSettings => {
+ const r = (v && typeof v === "object" ? v : {}) as Record<string, unknown>;
+ return {
+ bucket: typeof r.bucket === "string" ? r.bucket.trim() : "",
+ publicBaseUrl:
+ typeof r.publicBaseUrl === "string" ? r.publicBaseUrl.trim() : "",
+ };
+ }).describe(
+ "Overflow object storage (Cloudflare R2) for archive zips that exceed the Pages per-file size cap (see Site.archiveMaxBytes). When both fields are set, an oversize archive is uploaded here on deploy — via `wrangler r2 object put`, keyed `<siteId>/archives/<file>` — instead of being dropped, and the Downloads page links to `<publicBaseUrl>/<key>`. Blank/absent → no overflow, so oversize archives stay unavailable (\"Too large to host\").",
+ ),
+ reportDebouncePreset: settingsField((v): ReportDebouncePreset =>
+ isReportDebouncePreset(v) ? v : DEFAULT_REPORT_DEBOUNCE_PRESET).describe(
+ "Debounce preset for the global snapshot scheduler: how long it waits after the last report-changing action before regenerating affected channel reports. See REPORT_DEBOUNCE_PRESETS. Default \"fast\" (~1s, no cap).",
+ ),
+ autoRefreshIntervalSeconds: settingsField((v): number => clampAutoRefreshIntervalSeconds(v)).describe(
+ "How often (seconds) the editor UI passively re-fetches the current page's server-rendered data via router.refresh(), so sidebar badges and reports stay live without a manual reload. Mounted globally; pauses while the tab is hidden. 0 disables passive refresh entirely. See AUTO_REFRESH_INTERVAL_*.",
+ ),
+ syncScheduler: settingsField((v): SyncSchedulerSettings => sanitizeSyncScheduler(v)).describe(
+ "Global configuration for the scheduled (cron-driven) channel sync system. The per-channel cadence lives on ChannelConfig.syncIntervalMinutes; this block holds the defaults and guard rails the scheduler applies across all channels. See common/jobs/syncScheduler.ts.",
+ ),
+ autoQueue: autoQueueSchema.describe(
+ "Configuration for the automatic priority-queue runners (auto-transcribe / auto-download). Each holds a tree policy that decides which channel's video to process next, cross-channel, by priority/round-robin/weighted-fair rules. Independent of syncScheduler (which decides staleness, not work order). See common/jobs/autoQueuePolicy.ts.",
+ ),
+ channelPriority: channelPrioritySchema.describe(
+ "THE OPERATOR-FACING PRIORITY MODEL: one tier per channel plus one corpus-wide focus selector. It is the SOURCE the four `autoQueue[lane].root` trees are compiled from (common/lib/channelPriority.ts), not a second mechanism beside them — and its `paused` tier is the one part that is not a tree shape, filtering the runner's channel list instead. An empty document (the default) is today's behaviour exactly: no focus, every channel normal, the stored trees stand.",
+ ),
+ socialLinks: settingsField((v): SocialLink[] => parseSocialLinks(v)).describe(
+ "Default social links applied to every site that doesn't define its own. A site inherits these unless its site.json carries an explicit `socialLinks` array — see Site.socialLinks / resolveSocialLinks in common/lib/site.ts. The one presentation field that lives globally so a shared footer doesn't have to be repeated per site.",
+ ),
+ homepageUrl: settingsField((v): string => normalizeHomepageUrl(v)).describe(
+ "Absolute public URL of the family hub/homepage (e.g. \"https://archilyzer.pages.dev\"). Every export site links back to it (\"the family\" backlink) when set. Empty = no hub link rendered. Normalized to a trailing-slash-free http(s) URL.",
+ ),
+ savedVideoBackup: settingsField((v): SavedVideoBackupSettings => sanitizeSavedVideoBackup(v)).describe(
+ "Backup configuration for the saved-video store (Phase 4 of the video-persistence feature). When enabled with a destination, the store is mirrored there (additively, no deletes) with a per-backup manifest, and the sync scheduler runs the backup on the configured cadence. See common/controller/backupSavedVideos.ts.",
+ ),
+ storage: settingsField((v): StorageSettings => sanitizeStorage(v)).describe(
+ "Where a channel's downloaded media goes when it is relocated off the corpus disk. A DEFAULT ONLY: the relocate controller never reads it and always takes an explicit root, so this is the value the per-channel Storage panel prefills and the /channels bulk move falls back to. Blank = no default. See StorageSettings.",
+ ),
+ buildPipeline: settingsField((v): BuildPipelineSettings => sanitizeBuildPipeline(v)).describe(
+ "How the static export is built: \"basic\" reuses the single export/ tree and serializes builds on one queue (the long-standing behavior); \"docker\" runs each site's build in an isolated container for safe parallelism. The Docker pipeline itself is a follow-up; this block persists the chosen mode plus the container/concurrency knobs the deploy page and the future orchestrator read.",
+ ),
+ digest: settingsField((v): DigestSettings => sanitizeDigest(v)).describe(
+ "AI digest generation (chapters + topic tags over the existing transcripts). Local-first: the metered lane is off by default. See DigestSettings.",
+ ),
+ diarization: settingsField((v): DiarizationSettings => sanitizeDiarization(v)).describe(
+ "Speaker diarization captured right after transcription, while the audio is still on disk. OFF by default. See DiarizationSettings.",
+ ),
+ backfill: settingsField((v): BackfillSettings => sanitizeBackfill(v)).describe(
+ "The generic catch-up lane for derived data the existing corpus predates. OFF by default, and idle-only when on. See BackfillSettings.",
+ ),
+ attribution: settingsField((v): AttributionSettings => sanitizeAttribution(v)).describe(
+ "Naming the speakers diarization found (or reconstructing them from the transcript when it found none). OFF by default. See AttributionSettings.",
+ ),
+});
+
+export type SiteSettings = z.infer<typeof siteSettingsSchema>;
+
+// The whole default settings object, without touching disk: the schema's answer
+// for an empty file. Not a second literal — a default that lived anywhere but in
+// the field's own coercion would be a second place to change it.
+export function defaults(): SiteSettings {
+ return siteSettingsSchema.parse({});
+}
+
+// Exported under this name too, because tests and e2e helpers already say it:
+// a caller that needs a settings-SHAPED value rather than the operator's actual
+// configuration builds one here without a settings.json.
+export function defaultSiteSettings(): SiteSettings {
+ return defaults();
+}
diff --git a/common/lib/storageLocations.ts b/common/lib/storageLocations.ts
@@ -1,4 +1,5 @@
import path from "node:path";
+import type { FieldDocs } from "./fieldDocs";
// STORAGE LOCATIONS — the named places a channel's media may live.
//
@@ -31,57 +32,88 @@ import path from "node:path";
export const INTERNAL_LOCATION_ID = "internal";
export const INTERNAL_LOCATION_LABEL = "Internal (in place)";
+// Each field is documented in STORAGE_VOLUME_FIELD_DOCS below (rendered into SETTINGS.md).
export type StorageVolume = {
- // Filesystem UUID, the one stable name a disk has across mountpoints. This is
- // what makes "the platter came up somewhere else" a recoverable situation.
uuid: string;
fstype?: string;
label?: string;
- // Where the volume was mounted at the last successful probe, and the path of
- // the location's root RELATIVE to that mountpoint. Invariant:
- // `root === join(mountpoint, relPath)`. Keeping the two halves is what lets a
- // probe compute a candidate root when the volume reappears elsewhere.
mountpoint: string;
relPath: string;
};
+export const STORAGE_VOLUME_FIELD_DOCS: FieldDocs<StorageVolume> = {
+ uuid:
+ "Filesystem UUID, the one stable name a disk has across mountpoints. " +
+ "This is what makes \"the platter came up somewhere else\" a recoverable " +
+ "situation.",
+ fstype:
+ "Filesystem type reported by the probe (e.g. \"ext4\"). Informational; omitted when unknown.",
+ label:
+ "Filesystem label reported by the probe. Informational; omitted when unknown.",
+ mountpoint:
+ "Where the volume was mounted at the last successful probe, and the " +
+ "path of the location's root RELATIVE to that mountpoint. Invariant: " +
+ "`root === join(mountpoint, relPath)`. Keeping the two halves is what " +
+ "lets a probe compute a candidate root when the volume reappears " +
+ "elsewhere.",
+ relPath:
+ "The location root's path RELATIVE to `mountpoint` (see there). Invariant: `root === join(mountpoint, relPath)`.",
+};
+
+// Each field is documented in STORAGE_LOCATION_FIELD_DOCS below (rendered into SETTINGS.md).
export type StorageLocation = {
- // /^[a-z0-9][a-z0-9-]{0,63}$/, unique within the list. Stable: it is what
- // `defaultLocationId` and every form and action refer to.
id: string;
- // Human name. Blank sanitizes to the id.
label: string;
- // Absolute directory, trailing "/" stripped. NEVER existence-checked on read
- // — the whole point of a cold location is a drive that may not be mounted
- // when settings are parsed.
root: string;
- // Opt-in: when the volume is found mounted somewhere else, re-point without
- // asking (if the preflight passes). Off by default — re-point rewrites every
- // channel symlink on the location, and that is not something to do silently
- // unless the operator asked for it.
autoRepoint: boolean;
- // Identity learned at the last successful probe. Optional because a location
- // may never have been probed, and because in a container block devices are
- // invisible and identity is permanently unknown.
volume?: StorageVolume;
};
+export const STORAGE_LOCATION_FIELD_DOCS: FieldDocs<StorageLocation> = {
+ id:
+ "/^[a-z0-9][a-z0-9-]{0,63}$/, unique within the list. Stable: it is " +
+ "what `defaultLocationId` and every form and action refer to.",
+ label:
+ "Human name. Blank sanitizes to the id.",
+ root:
+ "Absolute directory, trailing \"/\" stripped. NEVER existence-checked on " +
+ "read — the whole point of a cold location is a drive that may not be " +
+ "mounted when settings are parsed.",
+ autoRepoint:
+ "Opt-in: when the volume is found mounted somewhere else, re-point " +
+ "without asking (if the preflight passes). Off by default — re-point " +
+ "rewrites every channel symlink on the location, and that is not " +
+ "something to do silently unless the operator asked for it.",
+ volume:
+ "Identity learned at the last successful probe. Optional because a " +
+ "location may never have been probed, and because in a container block " +
+ "devices are invisible and identity is permanently unknown.",
+};
+
+// Each field is documented in STORAGE_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
export type StorageSettings = {
locations: StorageLocation[];
- // The location prefilled as the destination of a move. "" = no default.
defaultLocationId: string;
- // WHERE THE SAVED-VIDEO STORE IS, by location id. "" = in place, under the
- // corpus at `paths.savedVideosDir`.
- //
- // A RECORD OF WHAT IS ON DISK, never an intention — the same contract as a
- // channel's `config.dataDir`. It is written by the move, on success, after
- // the copy has verified and the symlink is in place; nothing else writes it,
- // and a reader that disagrees with the disk trusts the disk. Optional so an
- // older settings.json parses (and an older binary that drops it leaves a
- // store that still works, because the symlink is what every reader follows).
savedVideosLocationId?: string;
};
+export const STORAGE_SETTINGS_FIELD_DOCS: FieldDocs<StorageSettings> = {
+ locations:
+ "The named storage locations a channel's media may be relocated to — one entry per root, each with an id, label, root, `autoRepoint` and the learned volume identity. Order is display order. Managed on /storage.",
+ defaultLocationId:
+ "The location prefilled as the destination of a move. \"\" = no default.",
+ savedVideosLocationId:
+ "WHERE THE SAVED-VIDEO STORE IS, by location id. \"\" = in place, under " +
+ "the corpus at `paths.savedVideosDir`.\n\n" +
+ "A RECORD OF WHAT IS ON DISK, never an intention — the same contract as" +
+ " a channel's `config.dataDir`. It is written by the move, on success, " +
+ "after the copy has verified and the symlink is in place; nothing else " +
+ "writes it, and a reader that disagrees with the disk trusts the disk. " +
+ "Optional so an older settings.json parses (and an older binary that " +
+ "drops it leaves a store that still works, because the symlink is what " +
+ "every reader follows).",
+};
+
// Strip trailing slashes so "/mnt/platter/" and "/mnt/platter" are one root.
// The sanitizer does this on write too; this is here so a hand-edited
// settings.json still compares correctly.
diff --git a/common/lib/transcriptionApps.ts b/common/lib/transcriptionApps.ts
@@ -17,29 +17,45 @@ import {
createChoughProgressParser,
createParakeetProgressParser,
} from "../jobs/progressParsers";
+import type { FieldDocs } from "./fieldDocs";
export type TranscriptOutputFormat = "whisper-json" | "chough-json" | "vtt";
// Per-app configuration persisted under settings.transcriptionApps[id]. Every
// field is optional; an app falls back to its own defaults (defaultBin, env).
+// Each field is documented in APP_INSTANCE_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md).
export type AppInstanceConfig = {
- // Binary path/name override. Empty/undefined falls back to app.defaultBin().
bin?: string;
- // whisper.cpp model path (substituted for {model}); for chough this is the
- // optional CHOUGH_MODEL env (chough auto-downloads a model when unset).
model?: string;
- // chough remote server URL (CHOUGH_URL). Empty/undefined = local transcription.
remoteUrl?: string;
- // chough chunk size in seconds (-c). Undefined = chough's own default.
chunkSize?: number;
- // whisper.cpp custom argv template using the {audioFile}/{outputBase}/{model}
- // placeholders. Undefined = DEFAULT_TRANSCRIBE_ARGS.
customArgs?: string[];
- // parakeet compute device passed to parakeet-cli (--device / PARAKEET_DEVICE),
- // e.g. "cuda:0", "cpu". Undefined = parakeet-cli's default device.
device?: string;
};
+export const APP_INSTANCE_CONFIG_FIELD_DOCS: FieldDocs<AppInstanceConfig> = {
+ bin:
+ "Binary path/name override. Empty/undefined falls back to " +
+ "app.defaultBin().",
+ model:
+ "whisper.cpp model path (substituted for {model}); for chough this is " +
+ "the optional CHOUGH_MODEL env (chough auto-downloads a model when " +
+ "unset).",
+ remoteUrl:
+ "chough remote server URL (CHOUGH_URL). Empty/undefined = local " +
+ "transcription.",
+ chunkSize:
+ "chough chunk size in seconds (-c). Undefined = chough's own default.",
+ customArgs:
+ "whisper.cpp custom argv template using the " +
+ "{audioFile}/{outputBase}/{model} placeholders. Undefined = " +
+ "DEFAULT_TRANSCRIBE_ARGS.",
+ device:
+ "parakeet compute device passed to parakeet-cli (--device / " +
+ "PARAKEET_DEVICE), e.g. \"cuda:0\", \"cpu\". Undefined = parakeet-cli's " +
+ "default device.",
+};
+
export type TranscribeBuild = {
// Args passed after the binary.
argv: string[];
diff --git a/common/lib/workers.ts b/common/lib/workers.ts
@@ -19,6 +19,7 @@ import {
getTranscriptionApp,
validateTranscribeArgs,
} from "./transcriptionApps";
+import type { FieldDocs } from "./fieldDocs";
export type WorkerKind = "local" | "remote" | "llm";
@@ -36,32 +37,50 @@ export type WorkerKind = "local" | "remote" | "llm";
// "close enough": its output would be permanently-stale. The dispatcher
// verifies the tag against /api/tags before first use and degrades the worker
// when it is missing (controller/llmWorkers.ts).
+// Each field is documented in LLM_WORKER_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md).
export type LlmWorkerConfig = {
- baseUrl: string; // e.g. http://macbook.lan:11434
- // Concurrent generations to allow this endpoint. Defaults to 1 — one model
- // instance, one generation — unless the operator knows better.
+ baseUrl: string;
slots?: number;
};
+export const LLM_WORKER_CONFIG_FIELD_DOCS: FieldDocs<LlmWorkerConfig> = {
+ baseUrl:
+ "e.g. http://macbook.lan:11434",
+ slots:
+ "Concurrent generations to allow this endpoint. Defaults to 1 — one " +
+ "model instance, one generation — unless the operator knows better.",
+};
+
// Where a remote worker delegates. The remote runs its OWN worker pool and picks
// among ITS local workers — so the primary stores only how to reach it, not which
// engine to use.
+// Each field is documented in REMOTE_WORKER_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md).
export type RemoteWorkerConfig = {
- baseUrl: string; // e.g. http://gpu-box.lan:3011
- // Outbound bearer token sent with every /api/worker request to this remote.
- // The accepting side validates against its own WORKER_TOKEN env, never this.
+ baseUrl: string;
token?: string;
- // When true the remote shares the transcripts mount, so we send
- // {channelSlug, videoId} instead of uploading the audio bytes.
sharedFs?: boolean;
- // How many units this remote takes in parallel. The pool expands one remote
- // config into this many independently-schedulable slot entries at
- // reconfigure time (the defaultWorkersFromApps trick, applied live). Absent =
- // probed from the remote's /api/worker/health (its enabled worker count) —
- // see controller/remoteCapacity.ts; 1 until the probe answers.
slots?: number;
};
+export const REMOTE_WORKER_CONFIG_FIELD_DOCS: FieldDocs<RemoteWorkerConfig> = {
+ baseUrl:
+ "e.g. http://gpu-box.lan:3011",
+ token:
+ "Outbound bearer token sent with every /api/worker request to this " +
+ "remote. The accepting side validates against its own WORKER_TOKEN env," +
+ " never this.",
+ sharedFs:
+ "When true the remote shares the transcripts mount, so we send " +
+ "{channelSlug, videoId} instead of uploading the audio bytes.",
+ slots:
+ "How many units this remote takes in parallel. The pool expands one " +
+ "remote config into this many independently-schedulable slot entries at" +
+ " reconfigure time (the defaultWorkersFromApps trick, applied live). " +
+ "Absent = probed from the remote's /api/worker/health (its enabled " +
+ "worker count) — see controller/remoteCapacity.ts; 1 until the probe " +
+ "answers.",
+};
+
// A worker is ONE processing slot — one transcription at a time. To run N in
// parallel on the same engine, define N workers (the settings editor's "Copy"
// button duplicates one). This makes each slot independently togglable on the
@@ -71,34 +90,53 @@ export type RemoteWorkerConfig = {
// probed capacity): the pool expands it into N slot entries itself, because
// asking the operator to hand-copy a remote once per slot of a machine whose
// slot count the machine already reports would be busywork.
+// Each field is documented in WORKER_FIELD_DOCS below (rendered into SETTINGS.md).
export type Worker = {
- // Stable slug; used in settings, task ids, and logs.
id: string;
- // Human label shown in the UI.
name: string;
kind: WorkerKind;
enabled: boolean;
- // Lower = preferred. Ties broken by array order in the scheduler.
priority: number;
- // Capability routing. A tag is an OPERATION id from the backfill catalog
- // ("diarization", "attribution-text", …) or a contended RESOURCE
- // (WORKER_RESOURCE_TAGS). The scheduler consults them through workerMatches
- // below: an untagged worker takes anything, a tagged worker takes only work
- // whose requirement intersects its tags. Unknown tags are tolerated (they
- // match nothing and warn in the settings UI), never fatal.
tags?: string[];
- // LOCAL: an instance of a TRANSCRIPTION_APPS entry + its per-worker config.
appId?: string;
config?: AppInstanceConfig;
- // REMOTE: how to reach the delegate instance.
remote?: RemoteWorkerConfig;
- // LLM: how to reach the bare model endpoint.
llm?: LlmWorkerConfig;
};
+export const WORKER_FIELD_DOCS: FieldDocs<Worker> = {
+ id:
+ "Stable slug; used in settings, task ids, and logs.",
+ name:
+ "Human label shown in the UI.",
+ kind:
+ "\"local\" runs an app from TRANSCRIPTION_APPS on this machine (`appId` + `config`); \"remote\" delegates to another instance on the LAN (`remote`); \"llm\" is a bare ollama endpoint serving digest/attribution calls only (`llm`).",
+ enabled:
+ "Whether the scheduler may hand this slot work. Each worker is one slot, so parallelism is toggled per slot on the Workers page. Anything but an explicit `false` reads as enabled.",
+ priority:
+ "Lower = preferred. Ties broken by array order in the scheduler.",
+ tags:
+ "Capability routing. A tag is an OPERATION id from the backfill catalog" +
+ " (\"diarization\", \"attribution-text\", …) or a contended RESOURCE " +
+ "(WORKER_RESOURCE_TAGS). The scheduler consults them through " +
+ "workerMatches below: an untagged worker takes anything, a tagged " +
+ "worker takes only work whose requirement intersects its tags. Unknown " +
+ "tags are tolerated (they match nothing and warn in the settings UI), " +
+ "never fatal.",
+ appId:
+ "LOCAL: an instance of a TRANSCRIPTION_APPS entry + its per-worker " +
+ "config.",
+ config:
+ "LOCAL: the per-worker engine config (binary, model, device, …) — an AppInstanceConfig, see `transcriptionApps.<appId>`.",
+ remote:
+ "REMOTE: how to reach the delegate instance.",
+ llm:
+ "LLM: how to reach the bare model endpoint.",
+};
+
// The contended-resource half of the tag vocabulary — Lane.contendsFor's
// three values, restated here because this module must stay client-safe and the
// lane type lives in operations.ts, whose import graph reaches controllers.
diff --git a/common/package.json b/common/package.json
@@ -65,7 +65,8 @@
"recharts": "2.15.4",
"sonner": "^2.0.7",
"tailwind-merge": "^3.6.0",
- "tw-animate-css": "^1.4.0"
+ "tw-animate-css": "^1.4.0",
+ "zod": "^4.3.6"
},
"peerDependencies": {
"next": "16.2.3",
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **`settings.json` has one schema and one writer, and its key table is generated.** Every key, its default, its clamp and its documentation is now one zod schema (`common/lib/settingsSchema.ts`); `getSettings`/`writeSettings` both parse through it, and every settings form saves through one helper (`editor/app/settings/saveSettings.ts`) that merges only what the form changed. **`SETTINGS.md`** (new, repo root) lists every key with its default and what it does, and `settings.json.example` is now the full default object — both generated by `common/bin/settings-example.ts` and checked by a test, so neither can drift. Nothing an operator has configured reads differently. **Fixed:** adding or editing a storage location on `/storage` no longer erases the record of which location the saved-video store is on (`storage.savedVideosLocationId`).
- **Every live panel now polls one endpoint, `/api/view/<name>`, and the eight old addresses still answer.** The change token, the job head, workers, the operations board, the sync console and the widget's three strips were eight separate API routes that each did the same thing; they are one route serving eight named views (`pulse`, `activeJobs`, `workers`, `autoQueueStatus`, `schedulerStatus`, `widgetSync`, `widgetActionable`, `cleanable`), and the editor's own pages poll it there. `/api/pulse`, `/api/jobs/active`, `/api/workers`, `/api/auto-queue/status`, `/api/scheduler/status` and `/api/widget/{sync,actionable,cleanable}` are kept as **rewrites**, not redirects — same method, status, body and query string (`?rev=` included) — so a monitor widget pinned in a browser, or any script polling the old path, keeps working untouched. `/api/widget/presets` is unchanged. An unknown view name is a 404. **One behaviour change you might notice:** the operations board (every 3 s) and the sync console (every 5 s) now send their next poll only after the previous one answers, and abandon a poll that takes longer than 10–15 s — so a slow editor no longer piles requests up behind itself, and a hung request no longer stops the page updating.
- **A lane's pause is one key on the lane, and the four old pause fields are gone from `settings.json`.** Holding a lane has been `autoQueue.<lane>.held` since the runner work landed; until now the file also still carried the four flags that used to mean it — `transcriptionsPaused`, `downloadsPaused`, `digest.digestsPaused` and the backwards `backfill.enabled` (where *enabled* meant *not held*) — which were read only when a lane had no `held` yet, to carry an older file's pause across. Every lane now carries its own key, so those four are **deleted**: nothing reads them, no form writes them, and the next settings save drops them from the file. A settings.json that still spells one of them holds nothing with it, so a hand-edited file (or a very old backup restored over a newer one) can no longer resurrect a pause you had lifted, or lift one you had set. "Run the backfill lane" on the diarization page and the Hold/Pause buttons write the one key, as they already did. **UPGRADING: boot once on the release that writes `held` before taking this one.** That release is the one that moved the gate onto the lane and carried the old fields across on read; a single boot of it (any settings save, or just starting the editor and pausing/resuming anything) puts `autoQueue.<lane>.held` in your settings.json, after which **nothing you can see changes here** — the same buttons, the same labels, the same pauses. An install that jumps straight from an older release to this one has no `held` keys at all and **loses its pauses**: transcription, downloads and digests come up running, and the backfill lane comes up held. Re-set them from the dashboard, or add the keys by hand before starting.
- **A relocate job says how far it has got.** `rsync` has been printing its progress the whole time (`--info=progress2`) and every frame of it went into the job log as a carriage-return redraw of one line — so a 131 GB move and a 3 MB one looked identical from `/jobs`: a spinner. Now each frame is parsed into the **task bar** every other long job on that page already draws, reading `12.3 GB of 45.6 GB · 27 % · 110.50MB/s · ETA 5:32`, and the log gets **one line per 10 %** instead of several thousand frames of one. The percentage is against the tree the job already measured for its space check, not rsync's own — under incremental recursion that one is a percentage of what it has enumerated so far and walks backwards.
diff --git a/editor/app/channels/[slug]/incompleteTranscriptActions.ts b/editor/app/channels/[slug]/incompleteTranscriptActions.ts
@@ -7,7 +7,8 @@ import {
resolveQueueKey,
} from "yt-dlp-transcript-common/lib/queueKeys";
import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels";
-import { getSettings, writeSettings } from "yt-dlp-transcript-common/lib/settings";
+import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { saveSettings } from "../../settings/saveSettings";
import { startAutoRunner } from "yt-dlp-transcript-common/controller/autoRunner";
import {
runManagedFunction,
@@ -40,8 +41,7 @@ export async function enableAutoRunners(): Promise<void> {
!current.autoQueue.transcription.enabled ||
!current.autoQueue.download.enabled
) {
- await writeSettings({
- ...current,
+ await saveSettings({
autoQueue: {
...current.autoQueue,
transcription: { ...current.autoQueue.transcription, enabled: true },
diff --git a/editor/app/channels/actions.ts b/editor/app/channels/actions.ts
@@ -35,10 +35,8 @@ import {
siteChannelIndex,
type Site,
} from "yt-dlp-transcript-common/lib/site";
-import {
- getSettings,
- writeSettings,
-} from "yt-dlp-transcript-common/lib/settings";
+import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { saveSettings } from "../settings/saveSettings";
import {
LANES,
type AutoQueueKind,
@@ -944,7 +942,10 @@ export async function saveChannelPriorityAction(
autoQueue[lane] = { ...autoQueue[lane], enabled: true };
}
try {
- await writeSettings({ ...settings, channelPriority: next, autoQueue });
+ // BOTH BLOCKS IN ONE WRITE — the reason saveSettings takes a whole-settings
+ // patch rather than one block. `channelPriority` is a full document (its
+ // `channels` map replaces), and `autoQueue` carries all four lanes.
+ await saveSettings({ channelPriority: next, autoQueue });
} catch (e) {
return { error: (e as Error).message };
}
diff --git a/editor/app/jobs/actions.ts b/editor/app/jobs/actions.ts
@@ -2,10 +2,8 @@
import { revalidatePath } from "next/cache";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import {
- getSettings,
- writeSettings,
-} from "yt-dlp-transcript-common/lib/settings";
+import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { saveSettings } from "../settings/saveSettings";
import { laneRootFromScope } from "yt-dlp-transcript-common/lib/laneMigration";
import { isDefaultChannelPriority } from "yt-dlp-transcript-common/lib/channelPriority";
import { operationsForLane } from "yt-dlp-transcript-common/lib/operations";
@@ -216,8 +214,7 @@ export async function armLaneAction(
};
}
const policy = settings.autoQueue[lane];
- await writeSettings({
- ...settings,
+ await saveSettings({
autoQueue: {
...settings.autoQueue,
[lane]: {
@@ -276,8 +273,7 @@ export async function disarmLaneAction(
): Promise<ArmLaneResult> {
try {
const settings = getSettings();
- await writeSettings({
- ...settings,
+ await saveSettings({
autoQueue: {
...settings.autoQueue,
[lane]: { ...settings.autoQueue[lane], enabled: false },
diff --git a/editor/app/operations/actions.ts b/editor/app/operations/actions.ts
@@ -3,9 +3,9 @@
import { revalidatePath } from "next/cache";
import {
getSettings,
- writeSettings,
type SiteSettings,
} from "yt-dlp-transcript-common/lib/settings";
+import { saveSettings } from "../settings/saveSettings";
import {
isGateHeld,
withGateHeld,
@@ -94,8 +94,7 @@ export async function saveAutoQueueAction(
// in slice 1.4, is exactly what "forgotten here" would now mean. `undefined`
// is carried as undefined on purpose: that is what keeps a lane that has
// never been written falling back to its retired field.
- const next: SiteSettings = {
- ...current,
+ const next: Partial<SiteSettings> = {
autoQueue: {
...current.autoQueue,
[kind]: {
@@ -110,7 +109,7 @@ export async function saveAutoQueueAction(
},
};
try {
- await writeSettings(next);
+ await saveSettings(next);
} catch (e) {
return { ok: false, error: (e as Error).message };
}
@@ -153,15 +152,14 @@ export async function snoozeAutoQueueAction(
untilMs: number | null,
): Promise<SaveResult> {
const current = getSettings();
- const next: SiteSettings = {
- ...current,
+ const next: Partial<SiteSettings> = {
autoQueue: {
...current.autoQueue,
[kind]: { ...current.autoQueue[kind], snoozeUntil: untilMs },
},
};
try {
- await writeSettings(next);
+ await saveSettings(next);
} catch (e) {
return { ok: false, error: (e as Error).message };
}
@@ -199,7 +197,7 @@ async function setLaneHeld(
try {
const cur = getSettings();
if (isGateHeld(cur, lane) !== held) {
- await writeSettings(withGateHeld(cur, lane, held));
+ await saveSettings({ autoQueue: withGateHeld(cur, lane, held).autoQueue });
}
} catch (e) {
// REPORTED, not swallowed. The workers page's old best-effort persist
diff --git a/editor/app/operations/settingsActions.ts b/editor/app/operations/settingsActions.ts
@@ -6,7 +6,9 @@
// `*FormPresent` marker, because unchecked checkboxes are simply ABSENT from a
// FormData and a submit from a form lacking the block would read every switch
// as off. One form per block makes the marker unnecessary — the `...current`
-// spread at the top of each block is now the whole isolation story.
+// spread at the top of each block is now the whole isolation story, and
+// `saveSettings` (../settings/saveSettings.ts) merges each patch onto the
+// rest of the file.
//
// Every field name is the one the old single form used, and the persisted keys
// are unchanged: this is a move, not a redesign.
@@ -14,9 +16,9 @@
import { revalidatePath } from "next/cache";
import {
getSettings,
- writeSettings,
type SiteSettings,
} from "yt-dlp-transcript-common/lib/settings";
+import { saveSettings } from "../settings/saveSettings";
import {
isDigestSectionKind,
isDigestTimestampMode,
@@ -62,7 +64,7 @@ export async function saveDigestSettingsAction(
return { ok: false, error: "Digest app config payload is malformed" };
}
}
- // Filtered to KNOWN kinds here rather than leaning on writeSettings' sanitizer.
+ // Filtered to KNOWN kinds here rather than leaning on the settings schema's sanitizer.
// sanitizeDigest would drop an unknown value anyway, but typing it honestly is
// what lets the digest block below be checked against DigestSettings instead of
// cast — and the cast is what hid the dropped-fields bug.
@@ -73,8 +75,7 @@ export async function saveDigestSettingsAction(
const digestTimestampModeRaw = String(
formData.get("digestTimestampMode") ?? "",
).trim();
- const next: SiteSettings = {
- ...current,
+ const next: Partial<SiteSettings> = {
digest: {
// Spread the CURRENT block first. Every field this form does not render
// must survive a save untouched, and before this spread they did not:
@@ -119,7 +120,7 @@ export async function saveDigestSettingsAction(
},
};
try {
- await writeSettings(next);
+ await saveSettings(next);
} catch (e) {
return { ok: false, error: (e as Error).message };
}
@@ -139,8 +140,7 @@ export async function saveDiarizationSettingsAction(
): Promise<SaveResult> {
const current = getSettings();
const dDiar = current.diarization;
- const next: SiteSettings = {
- ...current,
+ const next: Partial<SiteSettings> = {
diarization: {
...dDiar,
enabled: formData.get("diarizationEnabled") === "on",
@@ -179,7 +179,7 @@ export async function saveDiarizationSettingsAction(
},
};
try {
- await writeSettings(next);
+ await saveSettings(next);
} catch (e) {
return { ok: false, error: (e as Error).message };
}
@@ -210,20 +210,20 @@ export async function saveBackfillLaneSettingsAction(
): Promise<SaveResult> {
const current = getSettings();
const dBack = current.backfill;
- const next: SiteSettings = withGateHeld(
- {
- ...current,
- backfill: {
- ...dBack,
- concurrency: num(formData, "backfillConcurrency", dBack.concurrency),
- allowRedownload: formData.get("backfillAllowRedownload") === "on",
- },
+ const next: Partial<SiteSettings> = {
+ backfill: {
+ ...dBack,
+ concurrency: num(formData, "backfillConcurrency", dBack.concurrency),
+ allowRedownload: formData.get("backfillAllowRedownload") === "on",
},
- "backfill",
- formData.get("backfillEnabled") !== "on",
- );
+ autoQueue: withGateHeld(
+ current,
+ "backfill",
+ formData.get("backfillEnabled") !== "on",
+ ).autoQueue,
+ };
try {
- await writeSettings(next);
+ await saveSettings(next);
} catch (e) {
return { ok: false, error: (e as Error).message };
}
@@ -245,8 +245,7 @@ export async function saveAttributionSettingsAction(
): Promise<SaveResult> {
const current = getSettings();
const dAttr = current.attribution;
- const next: SiteSettings = {
- ...current,
+ const next: Partial<SiteSettings> = {
attribution: {
...dAttr,
enabled: formData.get("attributionEnabled") === "on",
@@ -265,7 +264,7 @@ export async function saveAttributionSettingsAction(
},
};
try {
- await writeSettings(next);
+ await saveSettings(next);
} catch (e) {
return { ok: false, error: (e as Error).message };
}
diff --git a/editor/app/saved-videos/backupActions.ts b/editor/app/saved-videos/backupActions.ts
@@ -2,7 +2,8 @@
import { revalidatePath } from "next/cache";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { getSettings, writeSettings } from "yt-dlp-transcript-common/lib/settings";
+import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { saveSettings } from "../settings/saveSettings";
import { SYNC_INTERVAL_MAX_MINUTES } from "yt-dlp-transcript-common/lib/channelConfig";
import {
backupSavedVideos,
@@ -101,8 +102,7 @@ export async function saveSavedVideoBackupAction(
}
const settings = getSettings();
try {
- await writeSettings({
- ...settings,
+ await saveSettings({
savedVideoBackup: {
enabled,
dest,
diff --git a/editor/app/scheduler/actions.ts b/editor/app/scheduler/actions.ts
@@ -13,9 +13,9 @@ import {
} from "yt-dlp-transcript-common/lib/channelConfig";
import {
getSettings,
- writeSettings,
type SiteSettings,
} from "yt-dlp-transcript-common/lib/settings";
+import { saveSettings } from "../settings/saveSettings";
import { DURATION_KEEP } from "yt-dlp-transcript-common/lib/duration";
export type SaveResult = { ok: true } | { ok: false; error: string };
@@ -114,7 +114,7 @@ export async function setChannelCadencesAction(
// block is the sync operation's settings form now, on the sync operation's
// page, and this is what saves it.
//
-// Values are clamped/sanitized by sanitizeSyncScheduler inside writeSettings,
+// Values are clamped/sanitized by sanitizeSyncScheduler when saveSettings writes,
// so we only coerce here.
export async function saveSchedulerSettingsAction(
_prev: SaveResult | undefined,
@@ -143,8 +143,7 @@ export async function saveSchedulerSettingsAction(
const n = Number.parseInt(raw, 10);
return Number.isFinite(n) ? n : null;
};
- const next: SiteSettings = {
- ...current,
+ const next: Partial<SiteSettings> = {
syncScheduler: {
...current.syncScheduler,
enabled: formData.get("syncSchedulerEnabled") === "on",
@@ -189,7 +188,7 @@ export async function saveSchedulerSettingsAction(
},
};
try {
- await writeSettings(next);
+ await saveSettings(next);
} catch (e) {
return { ok: false, error: (e as Error).message };
}
diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts
@@ -8,7 +8,6 @@ import {
defaultBuildPipeline,
isBuildMode,
isReportDebouncePreset,
- getSettings,
MIN_FREE_DISK_GB_MAX,
RESUME_MARGIN_GB_DEFAULT,
RESUME_MARGIN_GB_MAX,
@@ -18,11 +17,10 @@ import {
SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS,
TRANSCRIPT_PAGE_HARD_CAP_BYTES,
TRANSCRIPT_PAGE_MIN_BYTES,
- writeSettings,
type SiteSettings,
type SocialLink,
} from "yt-dlp-transcript-common/lib/settings";
-import { DEFAULT_TRANSCRIPTION_APP_ID } from "yt-dlp-transcript-common/lib/transcriptionApps";
+import { saveSettings } from "./saveSettings";
import {
DEFAULT_COOKIE_MODE,
isCookieMode,
@@ -182,7 +180,7 @@ export async function saveSettingsAction(
}
// Build pipeline. Values are clamped/coerced by sanitizeBuildPipeline inside
- // writeSettings, so we only read the form here (NaN/blank → default). The
+ // the settings schema on save, so we only read the form here (NaN/blank → default). The
// deploy-page toggle also writes `mode`; whichever saves last wins.
const dB = defaultBuildPipeline();
const buildModeRaw = String(formData.get("buildMode") ?? "").trim();
@@ -198,22 +196,25 @@ export async function saveSettingsAction(
String(formData.get("dockerfile") ?? "").trim() || dB.dockerfile,
};
- const next: SiteSettings = {
+ // ONLY WHAT THIS FORM EDITS. saveSettings merges the patch over the stored
+ // file, so every block edited elsewhere — workers (/workers), syncScheduler
+ // (/operations/sync), autoQueue and channelPriority (the operation pages and
+ // /channels), savedVideoBackup (/saved-videos), storage (/storage) and the
+ // four operation blocks (/operations/<id>) — survives a save here without
+ // being read and re-listed. Before slice 4a this literal named all 31 fields
+ // and carried each foreign block through `getSettings()`, which is how an
+ // unrelated save once disarmed a sweep.
+ const next: Partial<SiteSettings> = {
adminTitle,
maxTranscriptPageBytes: parsed,
- // Workers are the source of truth; writeSettings derives the deprecated
- // transcriptionApp/transcriptionApps shadow from them. parallelTranscriptions
- // is vestigial (kept only for rollback) so pass the default.
- transcriptionApp: DEFAULT_TRANSCRIPTION_APP_ID,
- transcriptionApps: {},
- // workers: edited on /workers, beside the live list, by saveWorkersAction.
- workers: getSettings().workers,
cookiesFromBrowser,
cookieMode,
sleepBetweenDownloadsSeconds: sleepParsed,
downloadFormat,
minFreeDiskGB: minFreeDiskParsed,
resumeMarginGB: resumeMarginParsed,
+ // Vestigial (the worker list is the parallelism; kept only for rollback),
+ // and this form has always reset it to the default.
parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT,
inlineTranscribeOnFallback,
skipLiveDownloads,
@@ -222,43 +223,12 @@ export async function saveSettingsAction(
archiveStorage,
reportDebouncePreset,
autoRefreshIntervalSeconds: autoRefreshParsed,
- // syncScheduler: edited on /operations/sync, below the schedule, by
- // saveSchedulerSettingsAction.
- syncScheduler: getSettings().syncScheduler,
- // Preserve the existing auto-queue policy on an unrelated settings save
- // (this form doesn't edit it; the Auto-queue page does). writeSettings
- // re-sanitizes it regardless.
- autoQueue: getSettings().autoQueue,
- // Preserved for the same reason, and for one more: it is the SOURCE the
- // four roots above are compiled from, so rebuilding it here would silently
- // undo a focus. Edited on /channels by saveChannelPriorityAction, which is
- // its one writer.
- channelPriority: getSettings().channelPriority,
socialLinks,
homepageUrl,
- // Preserve the saved-video backup config on an unrelated settings save (the
- // Saved Videos page edits it). writeSettings re-sanitizes it regardless.
- savedVideoBackup: getSettings().savedVideoBackup,
- // NO LONGER THIS FORM'S. The storage block became a list of named
- // locations edited on /storage; preserved here the way autoQueue and
- // channelPriority are, so an unrelated settings save cannot erase it.
- // writeSettings re-sanitizes it regardless.
- storage: getSettings().storage,
buildPipeline,
- // Each edited on its own operation page; preserved here. Slice 3 moved
- // these four fieldsets to /operations/<id>, and with them the hidden
- // `*FormPresent` markers that used to gate them — a marker only existed
- // because one form saved everything, and reading an absent checkbox as
- // `false` could disarm a sweep, drop the diarization capture lane, flip
- // `allowRedownload` on, or arm ~194,000 model calls. Reading the CURRENT
- // block is now the whole protection, exactly as it is for autoQueue.
- digest: getSettings().digest,
- diarization: getSettings().diarization,
- backfill: getSettings().backfill,
- attribution: getSettings().attribution,
};
try {
- await writeSettings(next);
+ await saveSettings(next);
} catch (e) {
return { ok: false, error: (e as Error).message };
}
diff --git a/editor/app/settings/saveSettings.test.ts b/editor/app/settings/saveSettings.test.ts
@@ -0,0 +1,78 @@
+import { mkdtempSync, readFileSync, writeFileSync } from "node:fs";
+import { rm } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { test, after } from "node:test";
+import assert from "node:assert/strict";
+
+// Run with: pnpm -C editor exec tsx --test "app/**/*.test.ts"
+//
+// THE MERGE RULE of the editor's one settings writer: ONE LEVEL DEEP. An object
+// block merges over the stored block's keys; arrays and scalars replace; an
+// object nested INSIDE a block replaces. Same rule as editor/e2e/helpers.ts.
+//
+// SETTINGS SEAM as in common/lib/settingsWrite.test.ts: getPaths() memoizes, so
+// SETTINGS_FILE is set before the module under test is imported.
+
+const ROOT = mkdtempSync(path.join(os.tmpdir(), "save-settings-"));
+process.env.TRANSCRIPTS_DIR = ROOT;
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+
+const { mergeSettingsPatch, saveSettings } = await import("./saveSettings");
+const { defaultSiteSettings, getSettings } = await import(
+ "yt-dlp-transcript-common/lib/settings"
+);
+
+after(() => rm(ROOT, { recursive: true, force: true }));
+
+test("an object block merges over the stored block, one level", () => {
+ const base = defaultSiteSettings();
+ base.syncScheduler.maxConcurrentSyncs = 7;
+ const out = mergeSettingsPatch(base, {
+ syncScheduler: { enabled: true } as typeof base.syncScheduler,
+ });
+ assert.equal(out.syncScheduler.enabled, true);
+ assert.equal(out.syncScheduler.maxConcurrentSyncs, 7);
+ // The input is not mutated.
+ assert.equal(base.syncScheduler.enabled, false);
+});
+
+test("arrays and scalars replace", () => {
+ const base = defaultSiteSettings();
+ base.socialLinks = [{ label: "a", url: "/a", svg: "<svg></svg>" }];
+ const out = mergeSettingsPatch(base, { socialLinks: [], adminTitle: "X" });
+ assert.deepEqual(out.socialLinks, []);
+ assert.equal(out.adminTitle, "X");
+ assert.equal(out.minFreeDiskGB, base.minFreeDiskGB);
+});
+
+test("an object nested inside a block replaces, it is not merged", () => {
+ const base = defaultSiteSettings();
+ base.digest.apps = { a: { model: "m1", numCtx: 4096 } };
+ base.channelPriority.channels = { keep: { tier: "low" } };
+ const out = mergeSettingsPatch(base, {
+ digest: { apps: { b: { model: "m2" } } } as unknown as typeof base.digest,
+ channelPriority: {
+ focus: { kind: "none" },
+ channels: { other: { tier: "paused" } },
+ },
+ });
+ assert.deepEqual(out.digest.apps, { b: { model: "m2" } });
+ assert.equal(out.digest.localAppId, base.digest.localAppId);
+ assert.deepEqual(out.channelPriority.channels, { other: { tier: "paused" } });
+});
+
+test("saveSettings writes the merged result and touches nothing else", async () => {
+ writeFileSync(
+ process.env.SETTINGS_FILE!,
+ JSON.stringify({ adminTitle: "Kept", minFreeDiskGB: 9 }),
+ );
+ await saveSettings({ minFreeDiskGB: 3, backfill: { concurrency: 4 } as never });
+ const s = getSettings();
+ assert.equal(s.adminTitle, "Kept");
+ assert.equal(s.minFreeDiskGB, 3);
+ assert.equal(s.backfill.concurrency, 4);
+ assert.equal(s.backfill.allowRedownload, false);
+ const onDisk = JSON.parse(readFileSync(process.env.SETTINGS_FILE!, "utf8"));
+ assert.equal(onDisk.adminTitle, "Kept");
+});
diff --git a/editor/app/settings/saveSettings.ts b/editor/app/settings/saveSettings.ts
@@ -0,0 +1,58 @@
+import {
+ getSettings,
+ writeSettings,
+ type SiteSettings,
+} from "yt-dlp-transcript-common/lib/settings";
+
+// THE EDITOR'S ONE SETTINGS WRITER (one-core phase 3 slice 4a).
+//
+// Every server action that changes settings.json calls `saveSettings(patch)`
+// with only what it changed; this reads the current settings, merges the
+// patch, and hands the result to `writeSettings` — which is imported by this
+// file and by no other file under editor/app.
+//
+// THE MERGE IS ONE LEVEL DEEP, and no deeper — the same rule e2e's
+// `writeSettings` helper (editor/e2e/helpers.ts) applies to the fixture:
+//
+// - a patch key whose value is a PLAIN OBJECT merges over the current block's
+// keys: `{ syncScheduler: { enabled: true } }` changes one flag and keeps the
+// rest of the scheduler;
+// - ARRAYS and SCALARS replace: `workers: []` means no workers;
+// - an object NESTED INSIDE a block replaces: `autoQueue: { digest: {…} }`
+// replaces the digest lane whole (a half-merged policy tree would be a
+// worse surprise than a replaced one), and `channelPriority: { channels }`
+// replaces the whole channel map.
+//
+// WHY A PATCH OF THE WHOLE SETTINGS OBJECT, NOT `saveSettingsBlock(block,
+// patch)` as plans/one-core.md sketched: the channel-priority action writes
+// `channelPriority` AND the four `autoQueue` roots compiled from it in ONE
+// write. A one-block signature would split that into two, and a reader between
+// them would see a priority model and trees that disagree.
+//
+// NOT a "use server" module: it is a helper the action files call, never an
+// action a client can invoke with an arbitrary patch.
+
+type Patch = Partial<SiteSettings>;
+
+function isPlainObject(v: unknown): v is Record<string, unknown> {
+ return typeof v === "object" && v !== null && !Array.isArray(v);
+}
+
+// Pure, so the merge rule is testable without a settings file.
+export function mergeSettingsPatch(
+ current: SiteSettings,
+ patch: Patch,
+): SiteSettings {
+ const merged: Record<string, unknown> = { ...current };
+ for (const [key, value] of Object.entries(patch)) {
+ if (value === undefined) continue;
+ const base = (current as Record<string, unknown>)[key];
+ merged[key] =
+ isPlainObject(value) && isPlainObject(base) ? { ...base, ...value } : value;
+ }
+ return merged as SiteSettings;
+}
+
+export async function saveSettings(patch: Patch): Promise<void> {
+ await writeSettings(mergeSettingsPatch(getSettings(), patch));
+}
diff --git a/editor/app/sites/lib/buildModeAction.ts b/editor/app/sites/lib/buildModeAction.ts
@@ -4,9 +4,9 @@ import { revalidatePath } from "next/cache";
import {
getSettings,
isBuildMode,
- writeSettings,
type BuildMode,
} from "yt-dlp-transcript-common/lib/settings";
+import { saveSettings } from "../../settings/saveSettings";
// Persist the build-mode choice (Basic vs Docker) from the family page (/sites)
// so it becomes the default for every subsequent build. Only the mode is touched
@@ -17,10 +17,8 @@ export async function setBuildModeAction(
): Promise<{ ok: boolean }> {
if (!isBuildMode(mode)) return { ok: false };
const settings = getSettings();
- await writeSettings({
- ...settings,
- buildPipeline: { ...settings.buildPipeline, mode },
- });
+ // One key of one block: saveSettings merges it over the stored pipeline.
+ await saveSettings({ buildPipeline: { ...settings.buildPipeline, mode } });
revalidatePath("/sites");
return { ok: true };
}
diff --git a/editor/app/storage/actions.ts b/editor/app/storage/actions.ts
@@ -3,10 +3,8 @@
import path from "node:path";
import { revalidatePath } from "next/cache";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import {
- getSettings,
- writeSettings,
-} from "yt-dlp-transcript-common/lib/settings";
+import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { saveSettings } from "../settings/saveSettings";
import type { StorageLocation } from "yt-dlp-transcript-common/lib/storageLocations";
import {
mountByUuid,
@@ -105,8 +103,11 @@ export async function addStorageLocationAction(
autoRepoint: draft.autoRepoint,
};
const locations = [...settings.storage.locations, location];
- await writeSettings({
- ...settings,
+ // A PATCH of the storage block: saveSettings merges these keys over the
+ // stored block, so `savedVideosLocationId` — which this action does not
+ // edit — survives the save. (Before slice 4a it was rebuilt from these two
+ // keys alone, and adding or editing a location erased the store's record.)
+ await saveSettings({
storage: {
locations,
// The FIRST location is the default whether or not the box was ticked:
@@ -152,8 +153,8 @@ export async function editStorageLocationAction(
};
if (!keepVolume) delete (updated as { volume?: unknown }).volume;
- await writeSettings({
- ...settings,
+ // A patch of the storage block — `savedVideosLocationId` survives (see above).
+ await saveSettings({
storage: {
locations: settings.storage.locations.map((l) => (l.id === id ? updated : l)),
defaultLocationId: draft.makeDefault
@@ -214,8 +215,8 @@ export async function deleteStorageLocationAction(
};
}
const locations = settings.storage.locations.filter((l) => l.id !== id);
- await writeSettings({
- ...settings,
+ // A patch of the storage block — `savedVideosLocationId` survives (see above).
+ await saveSettings({
storage: {
locations,
defaultLocationId:
diff --git a/editor/app/workers/actions.ts b/editor/app/workers/actions.ts
@@ -3,7 +3,7 @@
import { revalidatePath } from "next/cache";
import { getWorkerPool } from "yt-dlp-transcript-common/jobs/workerPool";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { getSettings, writeSettings } from "yt-dlp-transcript-common/lib/settings";
+import { saveSettings } from "../settings/saveSettings";
import {
sanitizeWorkers,
validateWorkers,
@@ -38,7 +38,7 @@ export async function saveWorkersAction(
formData: FormData,
): Promise<SaveWorkersResult> {
// Workers are submitted as a JSON array by the WorkersField client component.
- // Sanitize + validate here for a friendly error; writeSettings re-validates.
+ // Sanitize + validate here for a friendly error; writeSettings re-validates on save.
let workersInput: unknown;
try {
workersInput = JSON.parse(String(formData.get("workersJson") ?? "[]"));
@@ -50,7 +50,7 @@ export async function saveWorkersAction(
if (workersErr) return { ok: false, error: workersErr };
try {
- await writeSettings({ ...getSettings(), workers });
+ await saveSettings({ workers });
} catch (e) {
return { ok: false, error: (e as Error).message };
}
diff --git a/plans/one-core-phase-3.md b/plans/one-core-phase-3.md
@@ -577,3 +577,110 @@ Deviations:
Not done here, by design: no batch route, no auth on views, `operations/status.ts`,
`requestCache.ts` and `liveInputs.ts` untouched. The full editor suite was not run; only
the 20-spec subset above.
+
+## Slice 4a, as shipped — one settings schema (2026-09-23)
+
+Branch `one-core/phase-3-s4a` off `main` `54cf1b31`, unmerged. Commits (shas after the
+trailer rewrite; this record is `plans:` commit 6 on top):
+
+| sha | what |
+|---|---|
+| `50215ab5` | zod `^4.3.6` joins `common` (lockfile +3 lines, no new resolution); the auto-queue defaults/clamps/tree normalisation/lane gate move from `jobs/autoQueuePolicy.ts` to `lib/autoQueueSchema.ts` (picker stays, re-exports every moved name); allow-list entry `lib/settings.ts -> jobs/autoQueuePolicy` burned, **10 → 9**; `autoQueuePolicy.test.ts` repointed (imports only), 59/59; `plans/tools/phase3-settings-numbers.ts` added |
+| `e120a2a5` | `workersSchema`, `channelPrioritySchema`, `autoQueueSchema` — zod seams over the existing sanitizers, in `lib/settingsFieldSchemas.ts` |
+| `f1abe024` | `lib/settingsSchema.ts`: `siteSettingsSchema` (31 fields, `.describe()` on each), `SiteSettings = z.infer`, `defaults()` = `defaultSiteSettings()` = `parse({})`; `lib/settings.ts` reduced to I/O (1,783 → 250 lines) and `export *`s the schema module; `settingsSchema.test.ts` |
+| `85478527` | `editor/app/settings/saveSettings.ts` + unit test; 19 call sites in 11 files converted to patches; `writeSettings` is imported by one editor file |
+| `0316e988` | `common/bin/settings-example.ts` (+`--check`), `lib/settingsDocs.ts`, generated `settings.json.example` + `SETTINGS.md`, `settingsDocs.test.ts`; SETUP.md points at SETTINGS.md |
+
+**What moved.** Every type, constant, clamp and block sanitizer that was in `lib/settings.ts`
+is in `lib/settingsSchema.ts` and re-exported, so no importer changed. `getSettings` =
+`finishRawMigrations(siteSettingsSchema.parse(premigrateRaw(raw)), raw)`: sweeps→lanes and
+mediaRoot→locations rewrite the raw input (keyed on the raw file's absence); legacy
+`transcribe*`→app registry and worker synthesis run after the parse, keyed on the raw
+object's `transcriptionApp` / `workers` absence. `writeSettings` =
+`parse({...deriveWorkerShadow(next), socialLinks: validatedSocialLinks(...)})` + tmp/rename;
+the two validators still throw. Every field is `z.unknown().catch(undefined).transform(coerce)`
+over the pre-existing clamp or sanitizer; no `.passthrough()`, no `.default()`.
+
+**Deviations from the spec, and why.**
+1. *The zod seams are not beside their sanitizers.* `workers.ts`, `channelPriority.ts` and
+ (through `jobs/autoQueuePolicy`) `autoQueueSchema.ts` are value-imported by `"use client"`
+ forms (WorkersConfigForm, ChannelTierSelect, LadderRung, …); a zod import there would ship
+ zod to the browser, contradicting "zod cannot reach a client bundle". The three schemas
+ live in `lib/settingsFieldSchemas.ts` (server-only importers); the sanitizers keep their
+ homes, names and signatures. Verified: no `ZodError`/`_zod` in `editor/.next/static` or
+ `export/.next/static` after both builds.
+2. *`archiveStorage` lost its `?`.* zod 4 cannot express an optional key that is always
+ emitted (`.optional()` omits it when absent; a transform returning `T | undefined` is a
+ required key). It was always emitted at runtime, so the shape test pins it as required;
+ tsc across the workspace needed no change.
+3. *`saveSettings(patch)` not `saveSettingsBlock(block, patch)`* — as instructed:
+ `channels/actions.ts` writes `channelPriority` and all four `autoQueue` roots in one write.
+4. *The example omits `workers`.* `defaultSiteSettings().workers` is `[]`; a copied template
+ spelling `workers: []` would mean zero transcription slots, where an absent key
+ synthesizes one. Stated in SETTINGS.md.
+5. *Only the 31 top-level comments moved into `.describe()`.* The nested block types
+ (`DigestSettings`, `DiarizationSettings`, …) keep their per-field comments on the
+ hand-written types, because each block is one sanitizer-backed field, not a zod object.
+
+**Behaviour changes (all intended, all small).**
+- A settings.json containing `null` threw in `getSettings` (`parsed[key]` on null); it now
+ reads as the empty file, like `[]`, `3` and `{`.
+- `adminTitle`, `cookiesFromBrowser`, `archiveStorage.*` are trimmed on read as they always
+ were on write (one schema). The live file has no untrimmed values.
+- `storage/actions.ts`: adding/editing a location used to rebuild the storage block from two
+ keys and so **erased `storage.savedVideosLocationId`**; the one-level merge keeps it.
+- `/settings` form (`editor/app/settings/actions.ts:207`): it used to pass
+ `transcriptionApp: DEFAULT` + `transcriptionApps: {}` with the stored workers. When the
+ stored list was `[]`, main's worker shadow therefore synthesized a default whisper-cpp
+ worker with NO config; the branch patches only the form's own fields, so the shadow
+ synthesizes from the STORED app and its stored config (better). And nothing now prunes
+ stale per-app entries from the `transcriptionApps` shadow — the old `{}` did — since the
+ shadow is rollback-only and still rewritten from workers on every save (accepted).
+
+**Dead example keys removed**: `transcribeBin`, `transcribeModel`, `transcribeArgs` (the
+pre-multi-app spelling, migrated on read).
+
+**Numbers** (`plans/tools/phase3-settings-numbers.ts`, `getSettings()` sorted-key JSON).
+The live settings.json was re-saved at 19:24 mid-slice — the operator's Gate B change, applied
+through the backfill lane form (lane held, `allowRedownload` off; that save also stripped the
+retired `backfill.enabled`, as every save does), not a stray write — so the
+comparison runs both builds over the SAME frozen inputs: the live file as of 19:24, the
+pre-slice example, the e2e fixture, and both `docker/entrypoint.sh` seeds (parakeet,
+whisper). Main `54cf1b31` vs branch tip: **empty diff, 3,846 lines**. After review the tool
+also prints what `writeSettings(getSettings())` puts on disk — written to a scratch copy
+under `os.tmpdir()`, never the measured file (the frozen inputs' md5s were re-checked
+after the run). Re-run over the same frozen inputs: **read AND write both diff-empty, 7,691
+lines**, no write threw. The expected
+`backfill.enabled` line never appeared: `sanitizeBackfill` already dropped it at main, so it
+was not in `getSettings()` output before or after. The regenerated example of course parses
+differently from the old one (no legacy `whisper-cli`/`firefox` keys) — by design.
+
+**Entrypoint seed.** Both seeds (`workers[0]` enabled local parakeet / whisper-cpp,
+`parallelTranscriptions: 1`) parse to byte-identical settings through main and through the
+schema (included in the numbers above). The entrypoint does not read the example.
+
+**Review fix (one commit after commit 6).** SETTINGS.md now documents every
+NESTED key, not only the 31 top-level ones: each block type carries a
+`<TYPE>_FIELD_DOCS: FieldDocs<Type>` record beside it (`lib/fieldDocs.ts`; the mapped type
+requires one entry per key, optional keys and every union member's keys included, so an
+undocumented new field is a tsc error). The per-field comments moved out of the types into
+those records — 24 records across `settingsSchema.ts` (the seven blocks + `SocialLink` +
+the newly named `ArchiveStorageSettings`), `storageLocations.ts` (settings, location,
+volume), `workers.ts` (worker, remote, llm), `transcriptionApps.ts` (`AppInstanceConfig`),
+`digest.ts` (`DigestAppConfig`), `autoQueueTypes.ts` (policy, tree node, match) and
+`channelPriority.ts` (document, focus, entry, and `autoPaused`, now the named type
+`ChannelAutoPause`). `settingsDocs.ts` renders each as a key · default · description table
+under its block (lane-policy defaults per lane, so `held`'s `[false,false,false,true]` is
+visible). Also: the `settingsField` comment no longer implies zod guards a throwing
+sanitizer (`z.unknown().catch` cannot fire — totality is each coercion's); the docs say
+`workers: []` means no transcription only until the next save; SETTINGS.md warns that a
+copied example pins every default, `held` included.
+
+**Gates.** tsc (`pnpm -r --workspace-concurrency=1 exec tsc --noEmit`; the parallel `-r` form
+was OOM-killed, exit 137) clean after every commit. common **1625 → 1648** (+20 schema, +3
+docs); `test:scripts` 156 pass + 1 skip of 157 (unchanged); mcp 219/219; editor unit **59 → 63**;
+`next build` editor and export clean.
+**e2e** (from the worktree root, detached, ports 3311/3310): auto-queue, backfill, digest,
+diarization, attribution, scheduler, cadence-ui, storage-locations, channel-storage, workers,
+worker-remote, parakeet, parakeet-partial, chough, transcription-app-migration, disk-space,
+channel-priority, settings — **157/157 passed, exit 0, 9.9 min**, first run, nothing re-run.
diff --git a/plans/tools/phase3-settings-numbers.ts b/plans/tools/phase3-settings-numbers.ts
@@ -0,0 +1,167 @@
+#!/usr/bin/env tsx
+// The one-core Phase 3 slice 4a measurement: what `getSettings()` ANSWERS, for
+// every settings.json this repo can point at, printed deterministically so two
+// runs can be diffed.
+//
+// WHY THIS IS A SCRIPT AND NOT A TEST, same as phase1-numbers.ts next door.
+// Slice 4a replaces ten hand-written sanitizers with one zod schema. The claim
+// it makes is "nothing an operator has configured reads differently", and the
+// only way to check that is to parse the REAL files — the live corpus's
+// settings.json, the shipped example, the e2e fixture — before and after, and
+// diff two files. A test would have to carry the operator's configuration.
+//
+// NEVER WRITES A MEASURED FILE. Each target is copied to a scratch directory
+// under os.tmpdir(); the read and the write-back both happen on the copy, which
+// is deleted afterwards. (It measures the WRITE side too since slice 4a's
+// review: what `writeSettings(getSettings())` puts on disk.)
+//
+// NEVER BOOT AN EDITOR FOR THIS. `getSettings` is called in-process, offline;
+// instrumentation.ts is not loaded, so no runner, sweep or scheduler is armed.
+//
+// ONE PROCESS PER FILE, because `getPaths()` memoises its answer at module
+// scope: `SETTINGS_FILE` has to be set before `lib/settings.ts` is imported, so
+// a second file needs a second process. The parent below spawns itself once per
+// target; the child prints one block.
+//
+// Usage, from the repo root:
+// node_modules/.bin/tsx plans/tools/phase3-settings-numbers.ts
+// node_modules/.bin/tsx plans/tools/phase3-settings-numbers.ts live=/abs/settings.json
+//
+// Each argv entry is `label=path`. Given any, they REPLACE the default list;
+// the label is what the output names, so a file copied elsewhere (an archived
+// "before" example, say) can still be diffed against its original line for line.
+
+import { spawnSync } from "node:child_process";
+import fs from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const REPO = path.resolve(HERE, "..", "..");
+
+type Target = { label: string; file: string };
+
+// THE LIVE CORPUS'S settings.json, not this checkout's. A worktree carries its
+// own copy, and the file that matters is the one the editor actually runs on.
+// Overridable so the script is not pinned to one machine's layout.
+function liveSettingsFile(): string {
+ return (
+ process.env.LIVE_SETTINGS_FILE ??
+ path.join(
+ path.dirname(REPO),
+ "yt-dlp-transcript-browser",
+ "settings.json",
+ )
+ );
+}
+
+function defaultTargets(): Target[] {
+ const out: Target[] = [
+ { label: "live", file: liveSettingsFile() },
+ { label: "example", file: path.join(REPO, "settings.json.example") },
+ ];
+ const fixtures = path.join(REPO, "editor", "e2e", "fixtures");
+ for (const name of fs.readdirSync(fixtures).sort()) {
+ if (!/settings.*\.json$/i.test(name)) continue;
+ out.push({ label: `fixture:${name}`, file: path.join(fixtures, name) });
+ }
+ return out;
+}
+
+// Sort every object's keys so the output is diffable regardless of the order a
+// sanitizer (or a schema) happens to build its result in. Arrays keep their
+// order — in settings.json order IS data (workers are priority-ordered, and an
+// auto-queue tree's children compete in the order they are listed).
+function sortedKeys(_key: string, value: unknown): unknown {
+ if (!value || typeof value !== "object" || Array.isArray(value)) return value;
+ const src = value as Record<string, unknown>;
+ const out: Record<string, unknown> = {};
+ for (const k of Object.keys(src).sort()) out[k] = src[k];
+ return out;
+}
+
+// READ, THEN WRITE — BOTH AGAINST A SCRATCH COPY. The target is copied into a
+// fresh directory under os.tmpdir() and SETTINGS_FILE points at the copy, so
+// `writeSettings` — which writes `getPaths().settingsFile` — can never touch
+// the file being measured (the live settings.json included). The read is of
+// identical bytes; the write side is what a save of that reading puts on disk.
+async function child(file: string): Promise<void> {
+ const dir = fs.mkdtempSync(path.join(os.tmpdir(), "phase3-settings-"));
+ const scratch = path.join(dir, "settings.json");
+ fs.copyFileSync(file, scratch);
+ process.env.SETTINGS_FILE = scratch;
+ process.env.TRANSCRIPTS_DIR = dir;
+ try {
+ const { getSettings, writeSettings } = await import(
+ "../../common/lib/settings"
+ );
+ const read = getSettings();
+ console.log(JSON.stringify(read, sortedKeys, 2));
+ console.log("### written by writeSettings(getSettings())");
+ try {
+ await writeSettings(read);
+ const written = JSON.parse(fs.readFileSync(scratch, "utf8"));
+ console.log(JSON.stringify(written, sortedKeys, 2));
+ } catch (e) {
+ console.log(`WRITE THREW: ${(e as Error).message}`);
+ }
+ } finally {
+ fs.rmSync(dir, { recursive: true, force: true });
+ }
+}
+
+function parent(targets: Target[]): void {
+ console.log("# one-core phase 3 slice 4a — getSettings() and writeSettings() over every settings file");
+ console.log("");
+ for (const { label, file } of targets) {
+ console.log(`## ${label}`);
+ if (!fs.existsSync(file)) {
+ console.log("MISSING");
+ console.log("");
+ continue;
+ }
+ const res = spawnSync(
+ process.execPath,
+ [
+ path.join(REPO, "node_modules", "tsx", "dist", "cli.mjs"),
+ fileURLToPath(import.meta.url),
+ ],
+ {
+ cwd: REPO,
+ encoding: "utf8",
+ env: {
+ ...process.env,
+ PHASE3_SETTINGS_TARGET: file,
+ // The snooze sanitizer compares against the clock, and a storage
+ // probe would shell out. Neither is settings data; neither is read
+ // here. (`sanitizeSnooze` self-clears a lapsed snooze, which is
+ // stable as long as nothing is snoozed — noted, not worked around.)
+ TZ: "UTC",
+ },
+ },
+ );
+ if (res.status !== 0) {
+ console.log(`FAILED status=${res.status}`);
+ console.log(res.stderr.trim());
+ } else {
+ console.log(res.stdout.trimEnd());
+ }
+ console.log("");
+ }
+}
+
+const target = process.env.PHASE3_SETTINGS_TARGET;
+if (target) {
+ await child(target);
+} else {
+ const args = process.argv.slice(2);
+ const targets: Target[] = args.length
+ ? args.map((a) => {
+ const eq = a.indexOf("=");
+ if (eq < 0) return { label: path.basename(a), file: path.resolve(a) };
+ return { label: a.slice(0, eq), file: path.resolve(a.slice(eq + 1)) };
+ })
+ : defaultTargets();
+ parent(targets);
+}
diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml
@@ -86,6 +86,9 @@ importers:
tw-animate-css:
specifier: ^1.4.0
version: 1.4.0
+ zod:
+ specifier: ^4.3.6
+ version: 4.3.6
devDependencies:
'@types/d3-scale':
specifier: ^4.0.9
diff --git a/settings.json.example b/settings.json.example
@@ -1,19 +1,182 @@
{
"adminTitle": "Transcript Browser Admin",
"maxTranscriptPageBytes": 8388608,
- "transcribeBin": "whisper-cli",
- "transcribeModel": "~/whispercpp/whisper.cpp/models/ggml-base.en.bin",
- "transcribeArgs": [
- "-ojf",
- "-l",
- "en",
- "-m",
- "{model}",
- "-of",
- "{outputBase}",
- "{audioFile}"
- ],
- "cookiesFromBrowser": "firefox",
+ "transcriptionApp": "whisper-cpp",
+ "transcriptionApps": {},
+ "cookiesFromBrowser": "",
+ "cookieMode": "when-required",
"sleepBetweenDownloadsSeconds": 10,
- "inlineTranscribeOnFallback": false
+ "downloadFormat": "auto",
+ "minFreeDiskGB": 5,
+ "resumeMarginGB": 2,
+ "parallelTranscriptions": 2,
+ "inlineTranscribeOnFallback": false,
+ "skipLiveDownloads": true,
+ "verifyAvailabilityBeforeClean": true,
+ "buildArchives": true,
+ "archiveStorage": {
+ "bucket": "",
+ "publicBaseUrl": ""
+ },
+ "reportDebouncePreset": "fast",
+ "autoRefreshIntervalSeconds": 5,
+ "syncScheduler": {
+ "enabled": false,
+ "defaultIntervalMinutes": 1440,
+ "maxConcurrentSyncs": 2,
+ "quietHoursStart": null,
+ "quietHoursEnd": null,
+ "backoffBaseMinutes": 30,
+ "backoffMaxMinutes": 1440,
+ "heartbeatSeconds": 0,
+ "keepLatestCheckIntervalMinutes": 1440,
+ "fullSweepIntervalMinutes": 1440,
+ "fullSweepConfirmMaxSuspects": 25,
+ "fullSweepShrinkGuardPercent": 10
+ },
+ "autoQueue": {
+ "transcription": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "listed",
+ "snoozeUntil": null,
+ "held": false,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": []
+ }
+ },
+ "download": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "listed",
+ "snoozeUntil": null,
+ "held": false,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": []
+ }
+ },
+ "digest": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "cheapest",
+ "snoozeUntil": null,
+ "held": false,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": [
+ {
+ "id": "all",
+ "match": {
+ "type": "all"
+ },
+ "weight": 1,
+ "maxWorkers": null
+ }
+ ]
+ }
+ },
+ "backfill": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "listed",
+ "snoozeUntil": null,
+ "held": true,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": [
+ {
+ "id": "all",
+ "match": {
+ "type": "all"
+ },
+ "weight": 1,
+ "maxWorkers": null
+ }
+ ]
+ }
+ }
+ },
+ "channelPriority": {
+ "focus": {
+ "kind": "none"
+ },
+ "channels": {}
+ },
+ "socialLinks": [],
+ "homepageUrl": "",
+ "savedVideoBackup": {
+ "enabled": false,
+ "dest": "",
+ "intervalMinutes": 1440
+ },
+ "storage": {
+ "locations": [],
+ "defaultLocationId": ""
+ },
+ "buildPipeline": {
+ "mode": "basic",
+ "maxParallelBuilds": 2,
+ "dockerImage": "yt-dlp-transcript-browser-build",
+ "dockerfile": "Dockerfile.build"
+ },
+ "digest": {
+ "remoteEnabled": false,
+ "longTailSeconds": 14400,
+ "localAppId": "ollama-direct",
+ "remoteAppId": "claude-code",
+ "apps": {},
+ "yieldToTranscription": true,
+ "yieldToCpuWorkers": false,
+ "spendCapUsd": 0,
+ "sections": [
+ "chapters"
+ ],
+ "timestampMode": "chunk-local",
+ "promptVariant": ""
+ },
+ "diarization": {
+ "enabled": false,
+ "inlineAfterTranscribe": false,
+ "threshold": 0.9,
+ "threads": 4,
+ "engine": "sherpa-onnx",
+ "backend": "vulkan",
+ "python": "python3",
+ "segModel": "",
+ "embModel": "",
+ "sortformerBin": "",
+ "sortformerModel": "",
+ "concurrency": 1,
+ "maxAudioHours": 0
+ },
+ "backfill": {
+ "concurrency": 1,
+ "allowRedownload": false
+ },
+ "attribution": {
+ "enabled": false,
+ "appId": "ollama-direct",
+ "model": "",
+ "diarizedEnabled": false,
+ "textOnlyEnabled": false,
+ "promptVersion": 1
+ }
}