commit b509430307d6bfafd2abd5a3771b99396336cc79
parent 0cb01fb60f54654150913c8014c3bc52b5403d6c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 23 Sep 2026 20:06:57 -0400
settings: every nested key documented from one source; review fixes
one-core phase 3 slice 4a, review should-fix.
1. SETTINGS.md documents every NESTED key, not only the 31 top-level ones.
Each block type carries a `<TYPE>_FIELD_DOCS: FieldDocs<Type>` record
beside it (lib/fieldDocs.ts: one entry required per key, optional keys
and every union member's keys included; an undocumented new field or a
stale entry is a tsc error). The per-field comments moved out of the
types into those 24 records (settingsSchema.ts, storageLocations.ts,
workers.ts, transcriptionApps.ts, digest.ts, autoQueueTypes.ts,
channelPriority.ts); `autoPaused` and `archiveStorage` gained names
(ChannelAutoPause, ArchiveStorageSettings), no shape changed.
settingsDocs.ts renders a key / default / description table under each
block — lane-policy defaults per lane — plus tree nodes, matches,
locations, volumes, worker configs, digest apps. Two wiring tests.
The strings are tree-shaken out of editor client chunks (checked).
2. settingsField's comment no longer implies zod guards a throwing
sanitizer: z.unknown().catch cannot fire; totality is each coercion's.
3. `workers: []` means no transcription only until the next save.
4. SETTINGS.md: a copied example pins every default, `held` included.
5. Record: the undeclared /settings-form worker-shadow change.
6. phase3-settings-numbers.ts also prints writeSettings output, written to
a scratch copy under os.tmpdir(); read AND write diff-empty on the
frozen inputs (7,691 lines).
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
14 files changed, 1364 insertions(+), 445 deletions(-)
diff --git a/SETTINGS.md b/SETTINGS.md
@@ -1,12 +1,14 @@
# settings.json keys
-<!-- GENERATED by common/bin/settings-example.ts from common/lib/settingsSchema.ts — do not edit by hand. -->
+<!-- GENERATED by common/bin/settings-example.ts from common/lib/settingsSchema.ts and the *_FIELD_DOCS records beside each block type — do not edit by hand. -->
Global operational settings shared by every site this editor powers, persisted to `settings.json` at the repo root (or `$SETTINGS_FILE`). Per-site presentation lives in `sites/<id>/site.json`. Every key is optional: a missing key reads as its default, an ill-typed one is coerced to its default or clamped, and an unknown one is dropped on the next save.
Regenerate this file and `settings.json.example` with `pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts`.
-`settings.json.example` is the default object with one key left out, `workers`: a file that does not name it gets a worker list synthesized from `transcriptionApp` on read, where `workers: []` would mean no transcription at all.
+`settings.json.example` is the default object with one key left out, `workers`: a file that does not name it gets a worker list synthesized from `transcriptionApp` on read. A file that spells `workers: []` READS as no transcription at all — until the next save, when the writer synthesizes a worker the same way.
+
+A copied example PINS every default it spells — including each lane's `autoQueue.<lane>.held` — so a default changed in a later release will not reach that file. Delete any key you would rather have track the defaults.
| Key | Default |
|---|---|
@@ -64,6 +66,19 @@ Default: `"whisper-cpp"`
Per-app configuration, keyed by app id. Each app reads only its own block; a missing block means "use the app's defaults". DEPRECATED in favor of `workers` (each local worker carries its own config); kept one release to drive migration and allow rollback. See common/lib/workers.ts.
+#### `transcriptionApps.<appId>`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `bin` | Binary path/name override. Empty/undefined falls back to app.defaultBin(). |
+| `model` | whisper.cpp model path (substituted for {model}); for chough this is the optional CHOUGH_MODEL env (chough auto-downloads a model when unset). |
+| `remoteUrl` | chough remote server URL (CHOUGH_URL). Empty/undefined = local transcription. |
+| `chunkSize` | chough chunk size in seconds (-c). Undefined = chough's own default. |
+| `customArgs` | whisper.cpp custom argv template using the {audioFile}/{outputBase}/{model} placeholders. Undefined = DEFAULT_TRANSCRIBE_ARGS. |
+| `device` | parakeet compute device passed to parakeet-cli (--device / PARAKEET_DEVICE), e.g. "cuda:0", "cpu". Undefined = parakeet-cli's default device. |
+
Default:
```json
@@ -74,6 +89,56 @@ Default:
Configured transcription workers (named processing slots). The scheduler distributes each video to the highest-priority free worker. A settings.json predating this field is migrated to a single enabled worker from the active app (see defaultWorkersFromApps). See common/lib/workers.ts.
+#### `workers[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `id` | Stable slug; used in settings, task ids, and logs. |
+| `name` | Human label shown in the UI. |
+| `kind` | "local" runs an app from TRANSCRIPTION_APPS on this machine (`appId` + `config`); "remote" delegates to another instance on the LAN (`remote`); "llm" is a bare ollama endpoint serving digest/attribution calls only (`llm`). |
+| `enabled` | Whether the scheduler may hand this slot work. Each worker is one slot, so parallelism is toggled per slot on the Workers page. Anything but an explicit `false` reads as enabled. |
+| `priority` | Lower = preferred. Ties broken by array order in the scheduler. |
+| `tags` | Capability routing. A tag is an OPERATION id from the backfill catalog ("diarization", "attribution-text", …) or a contended RESOURCE (WORKER_RESOURCE_TAGS). The scheduler consults them through workerMatches below: an untagged worker takes anything, a tagged worker takes only work whose requirement intersects its tags. Unknown tags are tolerated (they match nothing and warn in the settings UI), never fatal. |
+| `appId` | LOCAL: an instance of a TRANSCRIPTION_APPS entry + its per-worker config. |
+| `config` | LOCAL: the per-worker engine config (binary, model, device, …) — an AppInstanceConfig, see `transcriptionApps.<appId>`. |
+| `remote` | REMOTE: how to reach the delegate instance. |
+| `llm` | LLM: how to reach the bare model endpoint. |
+
+#### `workers[].config`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `bin` | Binary path/name override. Empty/undefined falls back to app.defaultBin(). |
+| `model` | whisper.cpp model path (substituted for {model}); for chough this is the optional CHOUGH_MODEL env (chough auto-downloads a model when unset). |
+| `remoteUrl` | chough remote server URL (CHOUGH_URL). Empty/undefined = local transcription. |
+| `chunkSize` | chough chunk size in seconds (-c). Undefined = chough's own default. |
+| `customArgs` | whisper.cpp custom argv template using the {audioFile}/{outputBase}/{model} placeholders. Undefined = DEFAULT_TRANSCRIBE_ARGS. |
+| `device` | parakeet compute device passed to parakeet-cli (--device / PARAKEET_DEVICE), e.g. "cuda:0", "cpu". Undefined = parakeet-cli's default device. |
+
+#### `workers[].remote`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `baseUrl` | e.g. http://gpu-box.lan:3011 |
+| `token` | Outbound bearer token sent with every /api/worker request to this remote. The accepting side validates against its own WORKER_TOKEN env, never this. |
+| `sharedFs` | When true the remote shares the transcripts mount, so we send {channelSlug, videoId} instead of uploading the audio bytes. |
+| `slots` | How many units this remote takes in parallel. The pool expands one remote config into this many independently-schedulable slot entries at reconfigure time (the defaultWorkersFromApps trick, applied live). Absent = probed from the remote's /api/worker/health (its enabled worker count) — see controller/remoteCapacity.ts; 1 until the probe answers. |
+
+#### `workers[].llm`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `baseUrl` | e.g. http://macbook.lan:11434 |
+| `slots` | Concurrent generations to allow this endpoint. Defaults to 1 — one model instance, one generation — unless the operator knows better. |
+
Default:
```json
@@ -150,6 +215,13 @@ Default: `true`
Overflow object storage (Cloudflare R2) for archive zips that exceed the Pages per-file size cap (see Site.archiveMaxBytes). When both fields are set, an oversize archive is uploaded here on deploy — via `wrangler r2 object put`, keyed `<siteId>/archives/<file>` — instead of being dropped, and the Downloads page links to `<publicBaseUrl>/<key>`. Blank/absent → no overflow, so oversize archives stay unavailable ("Too large to host").
+#### `archiveStorage`
+
+| Key | Default | Description |
+|---|---|---|
+| `bucket` | `""` | Cloudflare R2 bucket an oversize archive zip is uploaded to on deploy (`wrangler r2 object put`, keyed `<siteId>/archives/<file>`). Blank = no overflow. |
+| `publicBaseUrl` | `""` | Public base URL of that bucket; the Downloads page links `<publicBaseUrl>/<key>`. Both fields must be set for overflow to happen. |
+
Default:
```json
@@ -175,6 +247,23 @@ Default: `5`
Global configuration for the scheduled (cron-driven) channel sync system. The per-channel cadence lives on ChannelConfig.syncIntervalMinutes; this block holds the defaults and guard rails the scheduler applies across all channels. See common/jobs/syncScheduler.ts.
+#### `syncScheduler`
+
+| Key | Default | Description |
+|---|---|---|
+| `enabled` | `false` | Master switch. When false, a tick selects nothing (manual sync still works). |
+| `defaultIntervalMinutes` | `1440` | Fallback cadence (minutes) for channels with no per-channel override. |
+| `maxConcurrentSyncs` | `2` | Cap on sync jobs running/queued at once. A tick queues at most (cap - currently-active) channels; the rest roll to the next tick. This is also the stagger mechanism that keeps a big due-batch from hitting the source all at once. |
+| `quietHoursStart` | `null` | Optional local-clock quiet window during which auto-sync is suppressed. Both null = always allowed. The window may wrap past midnight (e.g. start=22, end=6). Hours are [0,23]; the window is [start, end). |
+| `quietHoursEnd` | `null` | End hour of the quiet window, [0,23], exclusive. See `quietHoursStart`: both must be valid hours or the window is cleared (null = always allowed). |
+| `backoffBaseMinutes` | `30` | Failure backoff bounds. After N consecutive failed scheduled syncs a channel waits min(base * 2^(N-1), max) minutes before it's eligible again. |
+| `backoffMaxMinutes` | `1440` | Ceiling on the failure backoff (see `backoffBaseMinutes`): a channel waits min(base * 2^(N-1), max) minutes after N consecutive failures. Never below the base. |
+| `heartbeatSeconds` | `0` | Cadence (seconds) for the editor's in-process heartbeat — the internal timer armed by the instrumentation hook (editor/instrumentation.ts) that calls the scheduler tick directly, so no external cron is needed. 0 = off: rely on the external `pnpm sync:tick` heartbeat instead. Any positive value is clamped to [SYNC_HEARTBEAT_MIN_SECONDS, SYNC_HEARTBEAT_MAX_SECONDS]. The env var SYNC_HEARTBEAT_SECONDS overrides this at runtime. See SCHEDULED_SYNC.md. |
+| `keepLatestCheckIntervalMinutes` | `1440` | Cadence (minutes) for the scheduled keep-latest deletion check. For each channel with ChannelConfig.keepLatest > 0, the tick re-probes the kept window for source deletion (checkKeptDeletedAction) at most this often and pins any gone videos as do-not-clean. Clamped into the sync-interval window; default daily. The check shares the same concurrency cap and quiet-hours window as scheduled syncs. See editor/app/scheduler/runTick.ts. |
+| `fullSweepIntervalMinutes` | `1440` | Default cadence (minutes) for the sync FULL SWEEP — the deep pass that re-enumerates a channel's whole listing in one yt-dlp spawn, refreshes the stored `playlist` file, and flags videos that have left the listing into maybe-missing.json. Ordinary syncs stay on the cheap newest-first paged walk; a sync only upgrades itself to a sweep when this interval has elapsed since the channel's lastFullSweepAt. Per-channel override: ChannelConfig.fullSweepIntervalMinutes. 0 = never sweep. Default daily. See common/jobs/deepSync.ts. |
+| `fullSweepConfirmMaxSuspects` | `25` | Upper bound on how many maybe-missing suspects a full sweep will resolve in-line with the per-video availability probe (deleted vs private vs unlisted). At or under the cap the sweep runs the targeted check itself, so "Sync all" surfaces upstream deletions with no extra clicks; over it, the suspects are flagged and left for a manual check rather than firing hundreds of probes inside a sync. 0 = never auto-confirm. |
+| `fullSweepShrinkGuardPercent` | `10` | Shrink guard: how far a fresh listing may fall below the stored one before it is treated as suspect rather than acted on. Expressed as a percentage of the previous count, floored at SHRINK_ABS_FLOOR entries so ordinary churn on a small channel doesn't trip it. A suspect listing does not rewrite `playlist` or maybe-missing.json and does not count as a sweep — but a SECOND enumeration reporting a similar count confirms it and is accepted, so a genuine mass deletion costs at most one cadence period. 0 = off (the empty-listing rejection still applies). See controller/acceptListing.ts. |
+
Default:
```json
@@ -198,6 +287,42 @@ Default:
Configuration for the automatic priority-queue runners (auto-transcribe / auto-download). Each holds a tree policy that decides which channel's video to process next, cross-channel, by priority/round-robin/weighted-fair rules. Independent of syncScheduler (which decides staleness, not work order). See common/jobs/autoQueuePolicy.ts.
+#### `autoQueue.<lane>`
+
+| Key | Default | Description |
+|---|---|---|
+| `enabled` | `false` | Master switch for this runner (transcription / download independently). |
+| `maxWorkers` | `null` | Overall ceiling on concurrent in-flight workers for this runner. null = no runner-level cap (the worker pool / platform queues are the real throttle). |
+| `replaceAutoSubs` | `false` | Opt in to the lowest-priority "replace YouTube auto-captions" lane: append this kind's opt-in buckets (autoSubsOnly / downloadedAutoSubsOnly) to the tail of the default union, so videos whose only transcript is YouTube ASR get re-done with our own engine whenever nothing more important is pending. Default false — the corpus-wide cost is large (an audio download plus a transcription per video). A leaf can also target the bucket by name for per-channel opt-in without flipping this switch. Optional: settings written before this field existed lack it; the sanitizer defaults it to false. |
+| `order` | transcription `"listed"`<br>download `"listed"`<br>digest `"cheapest"`<br>backfill `"listed"` | Ordering within each rule (see AutoQueueOrder). Optional exactly like replaceAutoSubs: settings files written before this field existed lack it, and the sanitizer defaults them to "listed" (today's behaviour). |
+| `snoozeUntil` | `null` | Epoch ms until which this runner idles WITHOUT stopping: next() returns null so the loop stays up, re-reads settings each iteration, and resumes by itself when the moment passes. null/absent/past = not snoozed. Survives a restart because it lives in settings.json, not in runner memory. |
+| `held` | transcription `false`<br>download `false`<br>digest `false`<br>backfill `true` | THE LANE'S PAUSE GATE. Shut means the lane holds: every dispatch path asks lib/pauseGates.ts, whose limit()/guard returns 0 so runPool idle-waits. A hold, never a stop — see that file's header.<br><br>OPTIONAL IN THE TYPE, FILLED BY THE SANITIZER. Until slice 1.4 four separate settings fields carried this — `transcriptionsPaused`, `downloadsPaused`, `digest.digestsPaused` and (inverted) `backfill.enabled` — so `undefined` meant "ask the legacy field" and `sanitizePolicy` deliberately refused to default it: a default would have read a paused corpus as running. S0-pause deleted those four, on the precondition that the live settings.json already carried every `held` key, and the default came in with them (`defaultHeldFor` — free everywhere except backfill, whose field was inverted and shipped held).<br><br>It stays optional because a reader may be handed a PARTIAL settings object (laneGuards.test.ts casts one), and `isGateHeld` answers `false` for a lane that carries no key at all rather than throwing. |
+| `root` | object — see below | The lane's rule tree: a group whose children are groups and leaves (see the node table). A missing root is the lane's default — empty for the runner lanes, one catch-all leaf for digest and backfill. While a channel-priority document exists, the four roots are compiled from it and not hand-edited. |
+
+#### `autoQueue.<lane>.root (tree nodes)`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `id` | Stable node id, unique within the lane's tree. Preserved on save when valid and not taken, so persisted fairness state survives an unrelated edit; a missing or duplicate id is replaced with a generated one. |
+| `match` | LEAF ONLY: which videos this leaf owns (see the match table). |
+| `weight` | Relative share under a weighted-fair parent. Default 1. Ignored otherwise. |
+| `maxWorkers` | Optional ceiling on concurrent in-flight workers drawn from this node (and, for a group, its whole subtree). A capped node reads as "no work" and the parent falls through to the next sibling, like an HTB class ceiling. null = no cap. |
+| `mode` | GROUP ONLY: how the children compete — "strict" (first child with work wins), "round-robin", or "weighted-fair" (by each child's `weight`). Unknown values read as "strict". |
+| `children` | GROUP ONLY: the child nodes, in priority order for a strict group. A node with a `children` array is a group; any other node is a leaf. |
+
+#### `autoQueue.<lane>.root … .match`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `type` | What the leaf matches: "channel" (one channel slug in `value`), "platform" (a platform name in `value`), or "all". |
+| `value` | Channel slug (type=channel) or platform name (type=platform). Ignored for type=all. A type=channel leaf with no value matches nothing. |
+| `bucket` | Optional snapshot bucket this leaf draws from, narrowing the default for the runner kind (transcription → downloadedNoTranscript, download → undownloadedIds). E.g. bucket="failedListed" prioritizes retries. |
+| `operation` | Optional OPERATION this leaf draws from — a registered backfill kind id, or "digest". Same meaning as `bucket` one level up: it narrows what the leaf claims, and it draws from ChannelWork.operations rather than ChannelWork.buckets.<br><br>It lives on the MATCH, beside `bucket`, and not on the node. A field on the node would need group inheritance — "this group is the digest subtree" — and inheritance is resolution logic buildPendingByLeaf does not have. Here it needs exactly one sanitizer and exactly one claiming path.<br><br>It is a SEPARATE id space from `bucket`, and the sanitizer enforces that a leaf names at most one of the two (operation wins): `defaultBuckets` is a priority-ordered union, so a name that meant a bucket to one leaf and an operation to another would silently mix two id spaces, and selectableBucketsForKind feeds the editor's bucket dropdown, where an operation must not appear as a bucket. |
+
Default:
```json
@@ -287,6 +412,44 @@ Default:
THE OPERATOR-FACING PRIORITY MODEL: one tier per channel plus one corpus-wide focus selector. It is the SOURCE the four `autoQueue[lane].root` trees are compiled from (common/lib/channelPriority.ts), not a second mechanism beside them — and its `paused` tier is the one part that is not a tree shape, filtering the runner's channel list instead. An empty document (the default) is today's behaviour exactly: no focus, every channel normal, the stored trees stand.
+#### `channelPriority`
+
+| Key | Default | Description |
+|---|---|---|
+| `focus` | object — see below | The corpus-wide focus selector: none, one site's channels, or a list of channels. A focus is compiled into a leading `prio-focus` group in every lane's tree. |
+| `channels` | `{}` | ONLY channels that differ from the default appear. An absent slug is `normal`, unranked — so the default document is empty and "absent document = today's behaviour" holds byte for byte. |
+
+#### `channelPriority.focus`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `kind` | "none" (no focus), "site" (the channels of one site, resolved at compile time so it tracks membership) or "channels" (an explicit list, from "Focus these"). |
+| `siteId` | kind "site" only: the site whose channels are focused. A blank id reads as no focus; an unknown one survives and focuses nothing. |
+| `slugs` | kind "channels" only: the focused channel slugs, trimmed and de-duplicated. An empty list reads as no focus. |
+
+#### `channelPriority.channels.<slug>`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `tier` | THE BASE TIER: what every operation gets unless an override says otherwise. |
+| `rank` | Order WITHIN the tier, ascending. Absent = unranked, which sorts after every ranked sibling and then by slug. ONE rank per channel, not one per lane — the two hand-made lane orders collapse into this on migration. |
+| `overrides` | PER-OPERATION OVERRIDES of the base tier. Only operations that DIFFER from the base appear: the sanitizer normalises an override equal to `tier` away, so the on-disk document stays a list of exceptions to a list of exceptions.<br><br>`{tier:"normal", overrides:{sync:"paused"}}` is "everything but sync" — the lossless reading of the retired `excludeFromSync`. Its inverse, `{tier:"paused", overrides:{sync:"normal"}}`, is "sync only": keep the playlist and metadata current, dispatch nothing. |
+| `autoPaused` | PAUSED BY THE MACHINE, NOT BY THE OPERATOR, and what to put back.<br><br>Set when the drive a channel's media is on stops being there: the watch pass records the tier the channel HAD and forces `paused`, so nothing in any lane dispatches against a `data/` nobody can read. Cleared — and the tier restored — when the drive comes back.<br><br>WHY IT IS A FIELD AND NOT A DERIVED STATE. The lanes read `tier`; making them all ask a second question would be four more places to forget. And the tier the channel is to be RESTORED to is not derivable from anything once it has been overwritten — that is the whole content of this field.<br><br>OPTIONAL, and an older binary that drops it leaves the channel Paused with nothing lost but the automatic restore. The operator's own word always wins: a MANUAL tier change clears it (see clearAutoPause), so a drive coming back can never un-pause a channel somebody paused on purpose. |
+
+#### `channelPriority.channels.<slug>.autoPaused`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `reason` | One reason today. A union so a second one has somewhere to go, and so a surface can say WHICH machine decided rather than "automatic". |
+| `since` | ISO, for "auto-paused — media unreachable since <date>". |
+| `previousTier` | The base tier the channel had before the machine paused it; what a restore puts back. Never `paused` (that would restore to paused — a no-op dressed as a restore). |
+
Default:
```json
@@ -302,6 +465,16 @@ Default:
Default social links applied to every site that doesn't define its own. A site inherits these unless its site.json carries an explicit `socialLinks` array — see Site.socialLinks / resolveSocialLinks in common/lib/site.ts. The one presentation field that lives globally so a shared footer doesn't have to be repeated per site.
+#### `socialLinks[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `label` | Visible name, also the accessible label of the icon. |
+| `url` | Link target: http(s), mailto: or a site-relative path. |
+| `svg` | Inline SVG markup. Normalized on save (width/height stripped, fill="currentColor", aria-hidden) and rejected when unsafe (script, foreignObject, event handlers, javascript: URLs) or when it has no viewBox. |
+
Default:
```json
@@ -318,6 +491,14 @@ Default: `""`
Backup configuration for the saved-video store (Phase 4 of the video-persistence feature). When enabled with a destination, the store is mirrored there (additively, no deletes) with a per-backup manifest, and the sync scheduler runs the backup on the configured cadence. See common/controller/backupSavedVideos.ts.
+#### `savedVideoBackup`
+
+| Key | Default | Description |
+|---|---|---|
+| `enabled` | `false` | Master switch for the scheduled backup. A backup can still be run manually when this is false, as long as a destination is set. |
+| `dest` | `""` | Destination root the store is mirrored into (a local path or any rsync target). Empty disables both scheduled and manual backups. |
+| `intervalMinutes` | `1440` | Cadence (minutes) for the scheduled backup when enabled. Clamped into the sync-interval window; default daily. |
+
Default:
```json
@@ -332,6 +513,38 @@ Default:
Where a channel's downloaded media goes when it is relocated off the corpus disk. A DEFAULT ONLY: the relocate controller never reads it and always takes an explicit root, so this is the value the per-channel Storage panel prefills and the /channels bulk move falls back to. Blank = no default. See StorageSettings.
+#### `storage`
+
+| Key | Default | Description |
+|---|---|---|
+| `locations` | `[]` | The named storage locations a channel's media may be relocated to — one entry per root, each with an id, label, root, `autoRepoint` and the learned volume identity. Order is display order. Managed on /storage. |
+| `defaultLocationId` | `""` | The location prefilled as the destination of a move. "" = no default. |
+| `savedVideosLocationId` | absent | WHERE THE SAVED-VIDEO STORE IS, by location id. "" = in place, under the corpus at `paths.savedVideosDir`.<br><br>A RECORD OF WHAT IS ON DISK, never an intention — the same contract as a channel's `config.dataDir`. It is written by the move, on success, after the copy has verified and the symlink is in place; nothing else writes it, and a reader that disagrees with the disk trusts the disk. Optional so an older settings.json parses (and an older binary that drops it leaves a store that still works, because the symlink is what every reader follows). |
+
+#### `storage.locations[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `id` | /^[a-z0-9][a-z0-9-]{0,63}$/, unique within the list. Stable: it is what `defaultLocationId` and every form and action refer to. |
+| `label` | Human name. Blank sanitizes to the id. |
+| `root` | Absolute directory, trailing "/" stripped. NEVER existence-checked on read — the whole point of a cold location is a drive that may not be mounted when settings are parsed. |
+| `autoRepoint` | Opt-in: when the volume is found mounted somewhere else, re-point without asking (if the preflight passes). Off by default — re-point rewrites every channel symlink on the location, and that is not something to do silently unless the operator asked for it. |
+| `volume` | Identity learned at the last successful probe. Optional because a location may never have been probed, and because in a container block devices are invisible and identity is permanently unknown. |
+
+#### `storage.locations[].volume`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `uuid` | Filesystem UUID, the one stable name a disk has across mountpoints. This is what makes "the platter came up somewhere else" a recoverable situation. |
+| `fstype` | Filesystem type reported by the probe (e.g. "ext4"). Informational; omitted when unknown. |
+| `label` | Filesystem label reported by the probe. Informational; omitted when unknown. |
+| `mountpoint` | Where the volume was mounted at the last successful probe, and the path of the location's root RELATIVE to that mountpoint. Invariant: `root === join(mountpoint, relPath)`. Keeping the two halves is what lets a probe compute a candidate root when the volume reappears elsewhere. |
+| `relPath` | The location root's path RELATIVE to `mountpoint` (see there). Invariant: `root === join(mountpoint, relPath)`. |
+
Default:
```json
@@ -345,6 +558,15 @@ Default:
How the static export is built: "basic" reuses the single export/ tree and serializes builds on one queue (the long-standing behavior); "docker" runs each site's build in an isolated container for safe parallelism. The Docker pipeline itself is a follow-up; this block persists the chosen mode plus the container/concurrency knobs the deploy page and the future orchestrator read.
+#### `buildPipeline`
+
+| Key | Default | Description |
+|---|---|---|
+| `mode` | `"basic"` | "basic" — `pnpm run build` in export/, serialized on the build queue (shared output tree, no safe parallelism). "docker" — isolated per-site container builds, parallel up to `maxParallelBuilds`. |
+| `maxParallelBuilds` | `2` | Cap on concurrent per-site container builds in docker mode. Ignored in basic mode (which is always serial). Clamped to [1, BUILD_MAX_PARALLEL_MAX]. |
+| `dockerImage` | `"yt-dlp-transcript-browser-build"` | Tag of the reusable build image (built once, reused for every site). |
+| `dockerfile` | `"Dockerfile.build"` | Dockerfile path relative to the monorepo root, used to (re)build the image. |
+
Default:
```json
@@ -360,6 +582,36 @@ Default:
AI digest generation (chapters + topic tags over the existing transcripts). Local-first: the metered lane is off by default. See DigestSettings.
+#### `digest`
+
+| Key | Default | Description |
+|---|---|---|
+| `remoteEnabled` | `false` | Master switch for the metered (remote-api) lane. OFF by default — an opt-in overflow for the long tail or a channel where local quality is poor, never the default path. |
+| `longTailSeconds` | `14400` | Videos longer than this are "long tail": 8.2% of the corpus by count, 46% of all transcript tokens. The batch's duration-aware ordering and the optional remote overflow both key off it. |
+| `localAppId` | `"ollama-direct"` | The engine each lane uses (ids from common/lib/digestApps.ts). |
+| `remoteAppId` | `"claude-code"` | The engine the metered (remote-api) lane uses — an id from common/lib/digestApps.ts. Unknown ids degrade to the default app rather than failing. |
+| `apps` | `{}` | Per-app config, keyed by app id — the same id-keyed sub-record shape as transcriptionApps. |
+| `yieldToTranscription` | `true` | Yield the GPU to the transcription lane: while transcription is working, the digest batch's limit() returns 0 and the pool idle-waits. ON by default, because `digest:local` is deliberately on a different queue from TRANSCRIPTION_QUEUE and so would otherwise run ollama and the transcription engine on the same 8 GB card. See controller/digestYield.ts. |
+| `yieldToCpuWorkers` | `false` | Whether a busy worker pinned to `device: "cpu"` counts as GPU contention.<br><br>OFF by default, which is the FIX for a real bug: the yield originally tested only `kind === "local"`, so on a box with one GPU worker and two CPU-pinned ones (this box, at parallelTranscriptions 2) the digest lane stopped dead for transcription that competes for zero GPU shaders.<br><br>Only an EXPLICIT "cpu" is treated as non-contending. A worker with no device set is using the engine binary's own default, which may be the GPU, so it still triggers the yield — the unknown case fails safe.<br><br>Composes with `yieldToTranscription`: that is the master switch, this only narrows which workers it reacts to. |
+| `spendCapUsd` | `0` | Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. Only ever consulted for a metered app. |
+| `sections` | list — see below | Which sections a sweep generates.<br><br>Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, and that is not a contradiction: a tag call sends the same transcript as the chapter call before it, so it hits the engine's cached prefix and pays essentially no prefill (+0.4 s across 4 extra calls, against 22.4 s for the first 4). All it pays is decode, and a tag list is ~30 output tokens where a chapter list is ~200–290.<br><br>The corollary matters more than the number: run them in the SAME pass. Tags generated later, on their own, pay full prefill again — measured at 44% of a whole chapters pass, i.e. 3–9× the marginal cost of just including them now. |
+| `timestampMode` | `"chunk-local"` | How each chunk's transcript markers are numbered — see DigestTimestampMode. Was a scored variable in the bake-off rather than a pre-applied fix; the measurement is in and "chunk-local" is now the shipped default. |
+| `promptVariant` | `""` | A free-text label for a non-default prompt shape, folded into the recorded provenance by digestPromptVariant(). Setting it invalidates every digest generated under a different label, which is exactly what makes a bake-off round re-run its sample instead of skipping it as fresh. Empty = default. |
+
+#### `digest.apps.<appId>`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `bin` | Binary path/name override (process-based apps only). |
+| `baseUrl` | Base URL override (HTTP apps only). |
+| `model` | Model id, e.g. "qwen2.5:7b" or "haiku". |
+| `numCtx` | Context window in tokens. MUST reach the engine explicitly for ollama: its 4096 default silently truncates the input and the model then summarizes whatever fragment survived — measured, and the single easiest way to get quietly-wrong output at scale. |
+| `temperature` | Sampling temperature. 0 for a structured extraction task. |
+| `think` | Reasoning-model toggle (ollama's top-level `think`). Only sent when set, so a model that does not support thinking is never handed a field it rejects.<br><br>It matters for throughput, not correctness: measured on this box, qwen3:8b with thinking on spends most of its output budget on a `thinking` block before the JSON body the schema constrains. For an extraction task with a pinned schema that reasoning buys little and costs a multiple of the tokens, and tokens are what a multi-week sweep is priced in. |
+| `timeoutMs` | Per-request wall-clock ceiling (ms). A wedged engine must not stall a sweep. |
+
Default:
```json
@@ -384,6 +636,24 @@ Default:
Speaker diarization captured right after transcription, while the audio is still on disk. OFF by default. See DiarizationSettings.
+#### `diarization`
+
+| Key | Default | Description |
+|---|---|---|
+| `enabled` | `false` | Master switch. OFF by default so a transcription batch can start before this lands, with diarization backfilled over the retained audio afterwards.<br><br>Turning it ON also arms the cleanup guard: the Clean-audio sweep stops deleting audio for a transcribed video that has no diarization.json yet. That is the point — it is what keeps the perishable input alive long enough to be captured — but it means enabling this holds disk. |
+| `inlineAfterTranscribe` | `false` | Run diarization inline in the post-transcribe hook.<br><br>OFF by default, and that default is a MEASURED decision, not caution. Measured on this box: GPU transcription runs at 221 s/audio-hour (16.3x realtime, over 3,602 real videos), CPU diarization at ~500-680 s/audio-hour. Diarization is therefore ~2-3x SLOWER than the transcription it follows, so running it inline drops whole-pipeline throughput by roughly 3-4x and leaves the GPU idle while the CPU catches up.<br><br>The intended sequence for a large batch is the opposite: leave this off, let the batch transcribe at full GPU speed with `enabled` holding the audio, and diarize afterwards with the backfill pass. Turn it on for steady state, once the arrival rate is a few videos a day rather than a corpus. |
+| `threshold` | `0.9` | Clustering threshold — the single most consequential knob, since it decides how many speakers come out. Larger merges more aggressively.<br><br>The default is 0.9, NOT sherpa-onnx's own 0.5, and that is measured on this corpus. On a 6-minute excerpt of a two-person interview (known ground truth: 2 speakers), sherpa's default produced 22 clusters; 0.9 produced 6, with the top two at 40%/40% of talk time — recognizably the two hosts. Sweep on the same clip: 0.4→23, 0.5→22, 0.6→17, 0.7→12, 0.8→10, 0.9→6.<br><br>It still over-splits, which is why this is a capture lane and not an answer: the turns are recorded with the threshold that produced them, so a later attribution pass can re-cluster or re-run without needing the audio back. |
+| `threads` | `4` | Engine threads per diarize run. |
+| `engine` | `"sherpa-onnx"` | Which engine runs. "sherpa-onnx" is the shipped default and what every sidecar on disk was produced by; "sortformer" is the ggml engine built by scripts/build-sortformer.sh.<br><br>CHANGING THIS RESTATES THE FRESHNESS IDENTITY (see diarizationTarget), so every sidecar written by the other engine becomes stale and the backfill lane offers to redo it. That is intended — the two disagree about how many speakers exist, and a corpus half-diarized by each is not one corpus — but on the retained audio it is weeks of work, not a toggle.<br><br>Why anyone would: on the same file, sherpa at its tuned threshold returns 13 speakers and sortformer returns 4, agreeing on the dominant speaker's share to within half a point (73.1% vs 73.5%). On the corpus's worst case sherpa returns 35 and sortformer 4. Over-splitting is the failure mode this lane has always had, and sortformer is end-to-end rather than clustered, so it does not have it. The cost is a hard ceiling of 4 speakers and ~1.8x the wall clock. |
+| `backend` | `"vulkan"` | Compute device for the sortformer engine; ignored by sherpa-onnx, which has no Vulkan compute path on Linux.<br><br>"vulkan" is 1.5x faster than a thread-tuned CPU run (894 vs 1305 s/audio-hour, measured on this box) and holds 558 MB resident instead of 4.84 GB by keeping weights and activations in VRAM. It also takes ~4.4 GB of an 8 GB card, which is why the lane YIELDS to transcription rather than sharing — see controller/digestYield.ts. |
+| `python` | `"python3"` | Python interpreter for the default sherpa-onnx engine. sherpa-onnx ships wheels only up to cp313, and this box's system python is 3.14 — so this usually points at a dedicated venv rather than `python3`. |
+| `segModel` | `""` | ONNX model paths for the default engine. Empty = the lane cannot run, which is reported as a skip rather than a failure. |
+| `embModel` | `""` | ONNX speaker-embedding model path for the sherpa-onnx engine. Empty = the lane cannot run, reported as a `not-configured` skip rather than a failure (same as `segModel`). |
+| `sortformerBin` | `""` | Binary and model for the sortformer engine, both produced by scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the same "not-configured" skip as an unset segModel/embModel. |
+| `sortformerModel` | `""` | Model for the sortformer engine, produced by scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the same `not-configured` skip as an unset `sortformerBin`. |
+| `concurrency` | `1` | How many diarize runs may execute at once in the backfill pass. Kept low by default: diarization is CPU-bound and competes with GPU feeding and the digest sweep for the same 8 threads. |
+| `maxAudioHours` | `0` | Videos longer than this are DEFERRED rather than diarized: reported as a third number that is never summed into reachable work, so a capped corpus can never read as finished.<br><br>THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering holds a pairwise distance matrix over speech-segment embeddings — O(n^2) in SEGMENT count — and speaker-turn density varies 40x across this corpus (33-1364 turns/hour), so duration does not actually predict the blowup: a sparse 7h42m video completed while a dense 6h12m one was OOM-killed. Duration is merely the only predictor available for free, from metadata already on disk, BEFORE spending 45 minutes to find out. n^2 at 30k segments is 6.7 GiB and at 40k is 11.9 GiB, which brackets the 10.6 GB and 9.6 GB peaks measured on this 16 GB box.<br><br>0 disables the cap. That is where this goes once windowed diarization lands: windowing divides per-window n by the window count, so the matrix falls by its square, and the cap stops being needed rather than being tuned. |
+
Default:
```json
@@ -408,6 +678,13 @@ Default:
The generic catch-up lane for derived data the existing corpus predates. OFF by default, and idle-only when on. See BackfillSettings.
+#### `backfill`
+
+| Key | Default | Description |
+|---|---|---|
+| `concurrency` | `1` | Slots the lane may use when it is not standing aside. Kept at 1 by default for the same reason diarization.concurrency is: this is CPU-bound work competing with GPU feeding and the digest sweep for the same 8 threads. |
+| `allowRedownload` | `false` | Re-acquire media for videos whose input is GONE (audio deleted after transcription). OFF by default and deliberately so: measured on this corpus, 836 videos still have media and ~76,270 would need a re-download — 91x the reachable work, against 45 GB free at 97% full. When on, each re-fetched file is removed in a `finally` as soon as the backfill has used it, unless the video is marked do-not-clean, or unless the auto-transcribe policy would replace its auto-captions (`replaceAutoSubs`, or a leaf on `downloadedAutoSubsOnly`), in which case the audio is kept for that runner.<br><br>WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: "youtube"` channel — which normally only fetches subtitles — the re-acquire applies a PER-VIDEO transcribe override so yt-dlp lands audio a diarizer can read; the channel's stored config is not changed. Without that override the fetch re-downloads the captions the video already has and lands nothing, which is what happened to ~16,000 videos on eight channels in 2026-08. |
+
Default:
```json
@@ -421,6 +698,17 @@ Default:
Naming the speakers diarization found (or reconstructing them from the transcript when it found none). OFF by default. See AttributionSettings.
+#### `attribution`
+
+| Key | Default | Description |
+|---|---|---|
+| `enabled` | `false` | Master switch. Off means the backfill registry reports no attribution work at all — the feature gate every Operation has. |
+| `appId` | `"ollama-direct"` | Which digest app runs the naming. Attribution IS a digest-app workload — constrained JSON decoding over transcript text — so it reuses that registry and that per-app config (settings.digest.apps[appId]) rather than growing a second copy of the ollama URL, context size and timeout. |
+| `model` | `""` | Model override. Empty = the app's configured model, then its default. It is separate from the digest's because the two workloads may want different sizes, and because it is part of the freshness identity: sharing the digest's model field would make a digest bake-off invalidate every attribution record on disk as a side effect. |
+| `diarizedEnabled` | `false` | The lanes, separately. Both default OFF even when `enabled` is on, so turning the feature on to look at it cannot start a corpus sweep.<br><br>They are not a fallback pair. `diarized` is one call per video and grounded in acoustic clustering; `textOnly` is ~30 calls and guesses at identity across chunk seams. An operator may reasonably want the first forever and the second never. |
+| `textOnlyEnabled` | `false` | The text-only attribution lane: names speakers from the transcript alone (~30 model calls per video, guessing identity across chunk seams). Default OFF even when `enabled` is on. See `diarizedEnabled` — the two are separate lanes, not a fallback pair. |
+| `promptVersion` | `1` | The prompt generation a record must match to count as fresh.<br><br>Defaults to (and is floored at) ATTRIBUTION_PROMPT_VERSION, the shipped constant. Raising it forces a corpus-wide regeneration without a code change, which is the honest way to redo everything after a prompt tweak. It cannot be set BELOW the shipped constant, and that floor is the lesson from digestPrompt.ts's version 1 -> 2 note: pinning freshness to an older generation freezes output from a superseded prompt into the corpus, looking identical to output from the current one. |
+
Default:
```json
diff --git a/common/lib/autoQueueTypes.ts b/common/lib/autoQueueTypes.ts
@@ -9,6 +9,8 @@
//
// Enforced by ../architecture.test.ts.
+import type { FieldDocs } from "./fieldDocs";
+
// --- Tree types -------------------------------------------------------------
export type AutoQueueMode = "strict" | "round-robin" | "weighted-fair";
@@ -33,40 +35,48 @@ export type AutoQueueMatchType = "channel" | "platform" | "all";
// which is what makes it safe to name here before anything computes it.
export type AutoQueueOrder = "listed" | "newest" | "oldest" | "cheapest";
+// Each field is documented in AUTO_QUEUE_MATCH_FIELD_DOCS below (rendered into SETTINGS.md).
export type AutoQueueMatch = {
type: AutoQueueMatchType;
- // Channel slug (type=channel) or platform name (type=platform). Ignored for
- // type=all. A type=channel leaf with no value matches nothing.
value?: string;
- // Optional snapshot bucket this leaf draws from, narrowing the default for the
- // runner kind (transcription → downloadedNoTranscript, download →
- // undownloadedIds). E.g. bucket="failedListed" prioritizes retries.
bucket?: string;
- // Optional OPERATION this leaf draws from — a registered backfill kind id, or
- // "digest". Same meaning as `bucket` one level up: it narrows what the leaf
- // claims, and it draws from ChannelWork.operations rather than
- // ChannelWork.buckets.
- //
- // It lives on the MATCH, beside `bucket`, and not on the node. A field on the
- // node would need group inheritance — "this group is the digest subtree" —
- // and inheritance is resolution logic buildPendingByLeaf does not have. Here
- // it needs exactly one sanitizer and exactly one claiming path.
- //
- // It is a SEPARATE id space from `bucket`, and the sanitizer enforces that a
- // leaf names at most one of the two (operation wins): `defaultBuckets` is a
- // priority-ordered union, so a name that meant a bucket to one leaf and an
- // operation to another would silently mix two id spaces, and
- // selectableBucketsForKind feeds the editor's bucket dropdown, where an
- // operation must not appear as a bucket.
operation?: string;
};
+export const AUTO_QUEUE_MATCH_FIELD_DOCS: FieldDocs<AutoQueueMatch> = {
+ type:
+ "What the leaf matches: \"channel\" (one channel slug in `value`), \"platform\" (a platform name in `value`), or \"all\".",
+ value:
+ "Channel slug (type=channel) or platform name (type=platform). Ignored " +
+ "for type=all. A type=channel leaf with no value matches nothing.",
+ bucket:
+ "Optional snapshot bucket this leaf draws from, narrowing the default " +
+ "for the runner kind (transcription → downloadedNoTranscript, download " +
+ "→ undownloadedIds). E.g. bucket=\"failedListed\" prioritizes retries.",
+ operation:
+ "Optional OPERATION this leaf draws from — a registered backfill kind " +
+ "id, or \"digest\". Same meaning as `bucket` one level up: it narrows " +
+ "what the leaf claims, and it draws from ChannelWork.operations rather " +
+ "than ChannelWork.buckets.\n\n" +
+ "It lives on the MATCH, beside `bucket`, and not on the node. A field " +
+ "on the node would need group inheritance — \"this group is the digest " +
+ "subtree\" — and inheritance is resolution logic buildPendingByLeaf does" +
+ " not have. Here it needs exactly one sanitizer and exactly one " +
+ "claiming path.\n\n" +
+ "It is a SEPARATE id space from `bucket`, and the sanitizer enforces " +
+ "that a leaf names at most one of the two (operation wins): " +
+ "`defaultBuckets` is a priority-ordered union, so a name that meant a " +
+ "bucket to one leaf and an operation to another would silently mix two " +
+ "id spaces, and selectableBucketsForKind feeds the editor's bucket " +
+ "dropdown, where an operation must not appear as a bucket.",
+};
+
+// A node of a lane's rule tree is a leaf or a group; both are documented in
+// AUTO_QUEUE_NODE_FIELD_DOCS below (rendered into SETTINGS.md).
export type AutoQueueLeaf = {
id: string;
match: AutoQueueMatch;
- // Relative share under a weighted-fair parent. Default 1. Ignored otherwise.
weight?: number;
- // Optional ceiling on concurrent in-flight workers drawn from this leaf.
maxWorkers?: number | null;
};
@@ -80,57 +90,96 @@ export type AutoQueueGroup = {
export type AutoQueueNode = AutoQueueLeaf | AutoQueueGroup;
+export const AUTO_QUEUE_NODE_FIELD_DOCS: FieldDocs<AutoQueueNode> = {
+ id:
+ "Stable node id, unique within the lane's tree. Preserved on save when " +
+ "valid and not taken, so persisted fairness state survives an unrelated " +
+ "edit; a missing or duplicate id is replaced with a generated one.",
+ match:
+ "LEAF ONLY: which videos this leaf owns (see the match table).",
+ weight:
+ "Relative share under a weighted-fair parent. Default 1. Ignored " +
+ "otherwise.",
+ maxWorkers:
+ "Optional ceiling on concurrent in-flight workers drawn from this node " +
+ "(and, for a group, its whole subtree). A capped node reads as \"no " +
+ "work\" and the parent falls through to the next sibling, like an HTB " +
+ "class ceiling. null = no cap.",
+ mode:
+ "GROUP ONLY: how the children compete — \"strict\" (first child with " +
+ "work wins), \"round-robin\", or \"weighted-fair\" (by each child's " +
+ "`weight`). Unknown values read as \"strict\".",
+ children:
+ "GROUP ONLY: the child nodes, in priority order for a strict group. A " +
+ "node with a `children` array is a group; any other node is a leaf.",
+};
+
export function isGroup(node: AutoQueueNode): node is AutoQueueGroup {
return Array.isArray((node as AutoQueueGroup).children);
}
// --- Settings (persisted in settings.json under `autoQueue`) ----------------
+// Each field is documented in AUTO_QUEUE_POLICY_FIELD_DOCS below (rendered into SETTINGS.md).
export type AutoQueuePolicy = {
- // Master switch for this runner (transcription / download independently).
enabled: boolean;
- // Overall ceiling on concurrent in-flight workers for this runner. null = no
- // runner-level cap (the worker pool / platform queues are the real throttle).
maxWorkers: number | null;
- // Opt in to the lowest-priority "replace YouTube auto-captions" lane: append
- // this kind's opt-in buckets (autoSubsOnly / downloadedAutoSubsOnly) to the
- // tail of the default union, so videos whose only transcript is YouTube ASR
- // get re-done with our own engine whenever nothing more important is pending.
- // Default false — the corpus-wide cost is large (an audio download plus a
- // transcription per video). A leaf can also target the bucket by name for
- // per-channel opt-in without flipping this switch. Optional: settings written
- // before this field existed lack it; the sanitizer defaults it to false.
replaceAutoSubs?: boolean;
- // Ordering within each rule (see AutoQueueOrder). Optional exactly like
- // replaceAutoSubs: settings files written before this field existed lack it,
- // and the sanitizer defaults them to "listed" (today's behaviour).
order?: AutoQueueOrder;
- // Epoch ms until which this runner idles WITHOUT stopping: next() returns null
- // so the loop stays up, re-reads settings each iteration, and resumes by
- // itself when the moment passes. null/absent/past = not snoozed. Survives a
- // restart because it lives in settings.json, not in runner memory.
snoozeUntil?: number | null;
- // THE LANE'S PAUSE GATE. Shut means the lane holds: every dispatch path asks
- // lib/pauseGates.ts, whose limit()/guard returns 0 so runPool idle-waits. A
- // hold, never a stop — see that file's header.
- //
- // OPTIONAL IN THE TYPE, FILLED BY THE SANITIZER. Until slice 1.4 four separate
- // settings fields carried this — `transcriptionsPaused`, `downloadsPaused`,
- // `digest.digestsPaused` and (inverted) `backfill.enabled` — so `undefined`
- // meant "ask the legacy field" and `sanitizePolicy` deliberately refused to
- // default it: a default would have read a paused corpus as running. S0-pause
- // deleted those four, on the precondition that the live settings.json already
- // carried every `held` key, and the default came in with them
- // (`defaultHeldFor` — free everywhere except backfill, whose field was
- // inverted and shipped held).
- //
- // It stays optional because a reader may be handed a PARTIAL settings object
- // (laneGuards.test.ts casts one), and `isGateHeld` answers `false` for a lane
- // that carries no key at all rather than throwing.
held?: boolean;
root: AutoQueueGroup;
};
+export const AUTO_QUEUE_POLICY_FIELD_DOCS: FieldDocs<AutoQueuePolicy> = {
+ enabled:
+ "Master switch for this runner (transcription / download " +
+ "independently).",
+ maxWorkers:
+ "Overall ceiling on concurrent in-flight workers for this runner. null " +
+ "= no runner-level cap (the worker pool / platform queues are the real " +
+ "throttle).",
+ replaceAutoSubs:
+ "Opt in to the lowest-priority \"replace YouTube auto-captions\" lane: " +
+ "append this kind's opt-in buckets (autoSubsOnly / " +
+ "downloadedAutoSubsOnly) to the tail of the default union, so videos " +
+ "whose only transcript is YouTube ASR get re-done with our own engine " +
+ "whenever nothing more important is pending. Default false — the " +
+ "corpus-wide cost is large (an audio download plus a transcription per " +
+ "video). A leaf can also target the bucket by name for per-channel opt-" +
+ "in without flipping this switch. Optional: settings written before " +
+ "this field existed lack it; the sanitizer defaults it to false.",
+ order:
+ "Ordering within each rule (see AutoQueueOrder). Optional exactly like " +
+ "replaceAutoSubs: settings files written before this field existed lack" +
+ " it, and the sanitizer defaults them to \"listed\" (today's behaviour).",
+ snoozeUntil:
+ "Epoch ms until which this runner idles WITHOUT stopping: next() " +
+ "returns null so the loop stays up, re-reads settings each iteration, " +
+ "and resumes by itself when the moment passes. null/absent/past = not " +
+ "snoozed. Survives a restart because it lives in settings.json, not in " +
+ "runner memory.",
+ held:
+ "THE LANE'S PAUSE GATE. Shut means the lane holds: every dispatch path " +
+ "asks lib/pauseGates.ts, whose limit()/guard returns 0 so runPool idle-" +
+ "waits. A hold, never a stop — see that file's header.\n\n" +
+ "OPTIONAL IN THE TYPE, FILLED BY THE SANITIZER. Until slice 1.4 four " +
+ "separate settings fields carried this — `transcriptionsPaused`, " +
+ "`downloadsPaused`, `digest.digestsPaused` and (inverted) " +
+ "`backfill.enabled` — so `undefined` meant \"ask the legacy field\" and " +
+ "`sanitizePolicy` deliberately refused to default it: a default would " +
+ "have read a paused corpus as running. S0-pause deleted those four, on " +
+ "the precondition that the live settings.json already carried every " +
+ "`held` key, and the default came in with them (`defaultHeldFor` — free" +
+ " everywhere except backfill, whose field was inverted and shipped " +
+ "held).\n\n" +
+ "It stays optional because a reader may be handed a PARTIAL settings " +
+ "object (laneGuards.test.ts casts one), and `isGateHeld` answers " +
+ "`false` for a lane that carries no key at all rather than throwing.",
+ root:
+ "The lane's rule tree: a group whose children are groups and leaves (see the node table). A missing root is the lane's default — empty for the runner lanes, one catch-all leaf for digest and backfill. While a channel-priority document exists, the four roots are compiled from it and not hand-edited.",
+};
+
export type AutoQueueSettings = Record<AutoQueueKind, AutoQueuePolicy>;
// --- Lanes ------------------------------------------------------------------
diff --git a/common/lib/channelPriority.ts b/common/lib/channelPriority.ts
@@ -58,6 +58,7 @@ import {
type AutoQueueNode,
isGroup,
} from "./autoQueueTypes";
+import type { FieldDocs } from "./fieldDocs";
// --- The vocabulary ---------------------------------------------------------
@@ -119,56 +120,98 @@ export type ChannelFocus =
| { kind: "site"; siteId: string }
| { kind: "channels"; slugs: string[] };
+export const CHANNEL_FOCUS_FIELD_DOCS: FieldDocs<ChannelFocus> = {
+ kind:
+ "\"none\" (no focus), \"site\" (the channels of one site, resolved at " +
+ "compile time so it tracks membership) or \"channels\" (an explicit " +
+ "list, from \"Focus these\").",
+ siteId:
+ "kind \"site\" only: the site whose channels are focused. A blank id " +
+ "reads as no focus; an unknown one survives and focuses nothing.",
+ slugs:
+ "kind \"channels\" only: the focused channel slugs, trimmed and " +
+ "de-duplicated. An empty list reads as no focus.",
+};
+
+// Each field is documented in CHANNEL_PRIORITY_ENTRY_FIELD_DOCS below (rendered into SETTINGS.md).
export type ChannelPriorityEntry = {
- // THE BASE TIER: what every operation gets unless an override says otherwise.
tier: StoredChannelTier;
- // Order WITHIN the tier, ascending. Absent = unranked, which sorts after
- // every ranked sibling and then by slug. ONE rank per channel, not one per
- // lane — the two hand-made lane orders collapse into this on migration.
rank?: number;
- // PER-OPERATION OVERRIDES of the base tier. Only operations that DIFFER from
- // the base appear: the sanitizer normalises an override equal to `tier` away,
- // so the on-disk document stays a list of exceptions to a list of exceptions.
- //
- // `{tier:"normal", overrides:{sync:"paused"}}` is "everything but sync" — the
- // lossless reading of the retired `excludeFromSync`. Its inverse,
- // `{tier:"paused", overrides:{sync:"normal"}}`, is "sync only": keep the
- // playlist and metadata current, dispatch nothing.
overrides?: Partial<Record<PriorityOperation, StoredChannelTier>>;
- // PAUSED BY THE MACHINE, NOT BY THE OPERATOR, and what to put back.
- //
- // Set when the drive a channel's media is on stops being there: the watch
- // pass records the tier the channel HAD and forces `paused`, so nothing in
- // any lane dispatches against a `data/` nobody can read. Cleared — and the
- // tier restored — when the drive comes back.
- //
- // WHY IT IS A FIELD AND NOT A DERIVED STATE. The lanes read `tier`; making
- // them all ask a second question would be four more places to forget. And
- // the tier the channel is to be RESTORED to is not derivable from anything
- // once it has been overwritten — that is the whole content of this field.
- //
- // OPTIONAL, and an older binary that drops it leaves the channel Paused with
- // nothing lost but the automatic restore. The operator's own word always
- // wins: a MANUAL tier change clears it (see clearAutoPause), so a drive
- // coming back can never un-pause a channel somebody paused on purpose.
- autoPaused?: {
- // One reason today. A union so a second one has somewhere to go, and so a
- // surface can say WHICH machine decided rather than "automatic".
- reason: "storage";
- // ISO, for "auto-paused — media unreachable since <date>".
- since: string;
- previousTier: StoredChannelTier;
- };
+ autoPaused?: ChannelAutoPause;
};
+export const CHANNEL_PRIORITY_ENTRY_FIELD_DOCS: FieldDocs<ChannelPriorityEntry> = {
+ tier:
+ "THE BASE TIER: what every operation gets unless an override says " +
+ "otherwise.",
+ rank:
+ "Order WITHIN the tier, ascending. Absent = unranked, which sorts after" +
+ " every ranked sibling and then by slug. ONE rank per channel, not one " +
+ "per lane — the two hand-made lane orders collapse into this on " +
+ "migration.",
+ overrides:
+ "PER-OPERATION OVERRIDES of the base tier. Only operations that DIFFER " +
+ "from the base appear: the sanitizer normalises an override equal to " +
+ "`tier` away, so the on-disk document stays a list of exceptions to a " +
+ "list of exceptions.\n\n" +
+ "`{tier:\"normal\", overrides:{sync:\"paused\"}}` is \"everything but sync\" " +
+ "— the lossless reading of the retired `excludeFromSync`. Its inverse, " +
+ "`{tier:\"paused\", overrides:{sync:\"normal\"}}`, is \"sync only\": keep the" +
+ " playlist and metadata current, dispatch nothing.",
+ autoPaused:
+ "PAUSED BY THE MACHINE, NOT BY THE OPERATOR, and what to put back.\n\n" +
+ "Set when the drive a channel's media is on stops being there: the " +
+ "watch pass records the tier the channel HAD and forces `paused`, so " +
+ "nothing in any lane dispatches against a `data/` nobody can read. " +
+ "Cleared — and the tier restored — when the drive comes back.\n\n" +
+ "WHY IT IS A FIELD AND NOT A DERIVED STATE. The lanes read `tier`; " +
+ "making them all ask a second question would be four more places to " +
+ "forget. And the tier the channel is to be RESTORED to is not derivable" +
+ " from anything once it has been overwritten — that is the whole " +
+ "content of this field.\n\n" +
+ "OPTIONAL, and an older binary that drops it leaves the channel Paused " +
+ "with nothing lost but the automatic restore. The operator's own word " +
+ "always wins: a MANUAL tier change clears it (see clearAutoPause), so a" +
+ " drive coming back can never un-pause a channel somebody paused on " +
+ "purpose.",
+};
+
+// The machine's pause record on a channel entry (see `autoPaused` above).
+// Each field is documented in CHANNEL_AUTO_PAUSE_FIELD_DOCS below (rendered into SETTINGS.md).
+export type ChannelAutoPause = {
+ reason: "storage";
+ since: string;
+ previousTier: StoredChannelTier;
+};
+
+export const CHANNEL_AUTO_PAUSE_FIELD_DOCS: FieldDocs<ChannelAutoPause> = {
+ reason:
+ "One reason today. A union so a second one has somewhere to go, and so " +
+ "a surface can say WHICH machine decided rather than \"automatic\".",
+ since:
+ "ISO, for \"auto-paused — media unreachable since <date>\".",
+ previousTier:
+ "The base tier the channel had before the machine paused it; what a " +
+ "restore puts back. Never `paused` (that would restore to paused — a " +
+ "no-op dressed as a restore).",
+};
+
+// Each field is documented in CHANNEL_PRIORITY_FIELD_DOCS below (rendered into SETTINGS.md).
export type ChannelPriority = {
focus: ChannelFocus;
- // ONLY channels that differ from the default appear. An absent slug is
- // `normal`, unranked — so the default document is empty and "absent document
- // = today's behaviour" holds byte for byte.
channels: Record<string, ChannelPriorityEntry>;
};
+export const CHANNEL_PRIORITY_FIELD_DOCS: FieldDocs<ChannelPriority> = {
+ focus:
+ "The corpus-wide focus selector: none, one site's channels, or a list of channels. A focus is compiled into a leading `prio-focus` group in every lane's tree.",
+ channels:
+ "ONLY channels that differ from the default appear. An absent slug is " +
+ "`normal`, unranked — so the default document is empty and \"absent " +
+ "document = today's behaviour\" holds byte for byte.",
+};
+
export function defaultChannelPriority(): ChannelPriority {
return { focus: { kind: "none" }, channels: {} };
}
diff --git a/common/lib/digest.ts b/common/lib/digest.ts
@@ -17,6 +17,8 @@
// NEVER rename these to `transcript.<x>.<y>` — SUB_FILE_RE in videoStatus.ts
// would claim such a file as a subtitle track.
+import type { FieldDocs } from "./fieldDocs";
+
export const DIGEST_FILENAME = "ai-digest.json";
export const DIGEST_OVERRIDES_FILENAME = "ai-digest.overrides.json";
@@ -144,33 +146,46 @@ export const DEFAULT_DIGEST_APP_ID = OLLAMA_DIGEST_APP_ID;
// Per-app configuration persisted under settings.digest.apps[id]. Every field is
// optional; an app falls back to its own defaults.
+// Each field is documented in DIGEST_APP_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md).
export type DigestAppConfig = {
- // Binary path/name override (process-based apps only).
bin?: string;
- // Base URL override (HTTP apps only).
baseUrl?: string;
- // Model id, e.g. "qwen2.5:7b" or "haiku".
model?: string;
- // Context window in tokens. MUST reach the engine explicitly for ollama: its
- // 4096 default silently truncates the input and the model then summarizes
- // whatever fragment survived — measured, and the single easiest way to get
- // quietly-wrong output at scale.
numCtx?: number;
- // Sampling temperature. 0 for a structured extraction task.
temperature?: number;
- // Reasoning-model toggle (ollama's top-level `think`). Only sent when set, so
- // a model that does not support thinking is never handed a field it rejects.
- //
- // It matters for throughput, not correctness: measured on this box, qwen3:8b
- // with thinking on spends most of its output budget on a `thinking` block
- // before the JSON body the schema constrains. For an extraction task with a
- // pinned schema that reasoning buys little and costs a multiple of the tokens,
- // and tokens are what a multi-week sweep is priced in.
think?: boolean;
- // Per-request wall-clock ceiling (ms). A wedged engine must not stall a sweep.
timeoutMs?: number;
};
+export const DIGEST_APP_CONFIG_FIELD_DOCS: FieldDocs<DigestAppConfig> = {
+ bin:
+ "Binary path/name override (process-based apps only).",
+ baseUrl:
+ "Base URL override (HTTP apps only).",
+ model:
+ "Model id, e.g. \"qwen2.5:7b\" or \"haiku\".",
+ numCtx:
+ "Context window in tokens. MUST reach the engine explicitly for ollama:" +
+ " its 4096 default silently truncates the input and the model then " +
+ "summarizes whatever fragment survived — measured, and the single " +
+ "easiest way to get quietly-wrong output at scale.",
+ temperature:
+ "Sampling temperature. 0 for a structured extraction task.",
+ think:
+ "Reasoning-model toggle (ollama's top-level `think`). Only sent when " +
+ "set, so a model that does not support thinking is never handed a field" +
+ " it rejects.\n\n" +
+ "It matters for throughput, not correctness: measured on this box, " +
+ "qwen3:8b with thinking on spends most of its output budget on a " +
+ "`thinking` block before the JSON body the schema constrains. For an " +
+ "extraction task with a pinned schema that reasoning buys little and " +
+ "costs a multiple of the tokens, and tokens are what a multi-week sweep" +
+ " is priced in.",
+ timeoutMs:
+ "Per-request wall-clock ceiling (ms). A wedged engine must not stall a " +
+ "sweep.",
+};
+
// Whether an engine-reported model resolution looks like a DIFFERENT model
// rather than a benign tag completion. "qwen2.5" resolving to "qwen2.5:7b" or
// "qwen2.5:latest" is ollama filling in a tag; "qwen2.5:7b" coming back as
diff --git a/common/lib/fieldDocs.ts b/common/lib/fieldDocs.ts
@@ -0,0 +1,14 @@
+// A DESCRIPTION FOR EVERY KEY OF A SETTINGS BLOCK, checked by the compiler.
+//
+// Each settings.json block type carries a `<TYPE>_FIELD_DOCS: FieldDocs<Type>`
+// record beside it. The mapped type requires one entry per key — optional keys
+// included, and every member's keys when the type is a union — so adding a
+// field without documenting it is a tsc error, and a stale entry for a removed
+// field is an excess-property error. The records are rendered into SETTINGS.md
+// by lib/settingsDocs.ts; they are the one home of each field's documentation.
+//
+// Pure, no imports: a `"use client"` module may carry a record.
+
+type AllKeys<T> = T extends unknown ? keyof T : never;
+
+export type FieldDocs<T> = { readonly [K in AllKeys<T> & string]: string };
diff --git a/common/lib/settingsDocs.test.ts b/common/lib/settingsDocs.test.ts
@@ -32,3 +32,32 @@ test("the example parses back to the defaults", async () => {
const parsed = siteSettingsSchema.parse(JSON.parse(renderSettingsExample()));
assert.deepEqual(parsed, defaultSiteSettings());
});
+
+// EVERY FIELD, NOT ONLY THE 31 TOP-LEVEL ONES. The *_FIELD_DOCS records are
+// complete by type (FieldDocs<T> requires one entry per key); these two check
+// the wiring — that every object-valued block has a key table, and that a
+// block's table names every key its default actually carries.
+test("every object-valued block has a nested key table", async () => {
+ const { defaultSiteSettings } = await import("./settingsSchema");
+ const { blockTables } = await import("./settingsDocs");
+ const d = defaultSiteSettings() as Record<string, unknown>;
+ const tables = blockTables(defaultSiteSettings()) as Record<string, unknown[]>;
+ for (const [key, value] of Object.entries(d)) {
+ if (value === null || typeof value !== "object") continue;
+ assert.ok((tables[key]?.length ?? 0) > 0, `${key} has no key table`);
+ }
+});
+
+test("a block table documents every key its default carries", async () => {
+ const { defaultSiteSettings } = await import("./settingsSchema");
+ const { blockTables } = await import("./settingsDocs");
+ const d = defaultSiteSettings() as Record<string, unknown>;
+ for (const [key, list] of Object.entries(blockTables(defaultSiteSettings()))) {
+ const first = list?.[0];
+ if (!first?.defaults || key === "autoQueue") continue;
+ const value = d[key] as Record<string, unknown>;
+ for (const field of Object.keys(value)) {
+ assert.ok(field in first.docs, `${key}.${field} is undocumented`);
+ }
+ }
+});
diff --git a/common/lib/settingsDocs.ts b/common/lib/settingsDocs.ts
@@ -10,14 +10,52 @@
// prose from each field's `.describe()`. Changing a default or a description is
// a schema edit followed by regenerating, never an edit here.
-import { defaultSiteSettings, siteSettingsSchema } from "./settingsSchema";
+import {
+ ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS,
+ ATTRIBUTION_SETTINGS_FIELD_DOCS,
+ BACKFILL_SETTINGS_FIELD_DOCS,
+ BUILD_PIPELINE_SETTINGS_FIELD_DOCS,
+ DIARIZATION_SETTINGS_FIELD_DOCS,
+ DIGEST_SETTINGS_FIELD_DOCS,
+ SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS,
+ SOCIAL_LINK_FIELD_DOCS,
+ SYNC_SCHEDULER_SETTINGS_FIELD_DOCS,
+ defaultSiteSettings,
+ siteSettingsSchema,
+ type SiteSettings,
+} from "./settingsSchema";
+import {
+ AUTO_QUEUE_MATCH_FIELD_DOCS,
+ AUTO_QUEUE_NODE_FIELD_DOCS,
+ AUTO_QUEUE_POLICY_FIELD_DOCS,
+ LANES,
+} from "./autoQueueTypes";
+import {
+ CHANNEL_AUTO_PAUSE_FIELD_DOCS,
+ CHANNEL_FOCUS_FIELD_DOCS,
+ CHANNEL_PRIORITY_ENTRY_FIELD_DOCS,
+ CHANNEL_PRIORITY_FIELD_DOCS,
+} from "./channelPriority";
+import {
+ LLM_WORKER_CONFIG_FIELD_DOCS,
+ REMOTE_WORKER_CONFIG_FIELD_DOCS,
+ WORKER_FIELD_DOCS,
+} from "./workers";
+import { APP_INSTANCE_CONFIG_FIELD_DOCS } from "./transcriptionApps";
+import { DIGEST_APP_CONFIG_FIELD_DOCS } from "./digest";
+import {
+ STORAGE_LOCATION_FIELD_DOCS,
+ STORAGE_SETTINGS_FIELD_DOCS,
+ STORAGE_VOLUME_FIELD_DOCS,
+} from "./storageLocations";
// `workers` IS LEFT OUT OF THE EXAMPLE, and that is the one place the example
// is not the literal default object. Its default is `[]`, and a settings.json
-// that SPELLS `workers: []` means "no transcription workers" — auto-transcribe
-// then does nothing, silently. A file that does not name the key gets a worker
-// list synthesized from `transcriptionApp` on read (see getSettings), which is
-// what a template copied to settings.json should give.
+// that SPELLS `workers: []` READS as "no transcription workers" — until the
+// next save, when writeSettings' worker shadow synthesizes one from
+// `transcriptionApp`; in between, auto-transcribe does nothing, silently. A file
+// that does not name the key gets that worker list synthesized on read (see
+// getSettings), which is what a template copied to settings.json should give.
export const EXAMPLE_OMITTED_KEYS = ["workers"] as const;
export function renderSettingsExample(): string {
@@ -39,18 +77,172 @@ function defaultCell(v: unknown): string {
return Array.isArray(v) ? "list — see below" : "object — see below";
}
+// A description inside a table cell: one line, pipes escaped, paragraphs kept.
+function cell(text: string): string {
+ return text.replace(/\|/g, "\\|").replace(/\n\n/g, "<br><br>").replace(/\n/g, " ");
+}
+
+// One nested key table. `defaults(key)` answers the Default column; a table of
+// per-entry fields (list items, map values, tree nodes) has no defaults — each
+// entry spells its own — and says so.
+type KeyTable = {
+ path: string;
+ docs: Readonly<Record<string, string>>;
+ defaults?: (key: string) => string;
+};
+
+function fromObject(obj: unknown): (key: string) => string {
+ const r = (obj ?? {}) as Record<string, unknown>;
+ return (key) => (key in r ? defaultCell(r[key]) : "absent");
+}
+
+// A lane-policy field's default can differ per lane (`held`, `order`, `root`),
+// and that difference is exactly what a reader needs to see.
+function perLane(d: SiteSettings): (key: string) => string {
+ return (key) => {
+ const cells = LANES.map((lane) => {
+ const policy = d.autoQueue[lane] as Record<string, unknown>;
+ return key in policy ? defaultCell(policy[key]) : "absent";
+ });
+ if (cells.every((c) => c === cells[0])) return cells[0];
+ return LANES.map((lane, i) => `${lane} ${cells[i]}`).join("<br>");
+ };
+}
+
+export function blockTables(d: SiteSettings): Partial<Record<keyof SiteSettings, KeyTable[]>> {
+ return {
+ transcriptionApps: [
+ { path: "transcriptionApps.<appId>", docs: APP_INSTANCE_CONFIG_FIELD_DOCS },
+ ],
+ workers: [
+ { path: "workers[]", docs: WORKER_FIELD_DOCS },
+ { path: "workers[].config", docs: APP_INSTANCE_CONFIG_FIELD_DOCS },
+ { path: "workers[].remote", docs: REMOTE_WORKER_CONFIG_FIELD_DOCS },
+ { path: "workers[].llm", docs: LLM_WORKER_CONFIG_FIELD_DOCS },
+ ],
+ archiveStorage: [
+ {
+ path: "archiveStorage",
+ docs: ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.archiveStorage),
+ },
+ ],
+ syncScheduler: [
+ {
+ path: "syncScheduler",
+ docs: SYNC_SCHEDULER_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.syncScheduler),
+ },
+ ],
+ autoQueue: [
+ {
+ path: "autoQueue.<lane>",
+ docs: AUTO_QUEUE_POLICY_FIELD_DOCS,
+ defaults: perLane(d),
+ },
+ { path: "autoQueue.<lane>.root (tree nodes)", docs: AUTO_QUEUE_NODE_FIELD_DOCS },
+ { path: "autoQueue.<lane>.root … .match", docs: AUTO_QUEUE_MATCH_FIELD_DOCS },
+ ],
+ channelPriority: [
+ {
+ path: "channelPriority",
+ docs: CHANNEL_PRIORITY_FIELD_DOCS,
+ defaults: fromObject(d.channelPriority),
+ },
+ { path: "channelPriority.focus", docs: CHANNEL_FOCUS_FIELD_DOCS },
+ { path: "channelPriority.channels.<slug>", docs: CHANNEL_PRIORITY_ENTRY_FIELD_DOCS },
+ {
+ path: "channelPriority.channels.<slug>.autoPaused",
+ docs: CHANNEL_AUTO_PAUSE_FIELD_DOCS,
+ },
+ ],
+ socialLinks: [{ path: "socialLinks[]", docs: SOCIAL_LINK_FIELD_DOCS }],
+ savedVideoBackup: [
+ {
+ path: "savedVideoBackup",
+ docs: SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.savedVideoBackup),
+ },
+ ],
+ storage: [
+ {
+ path: "storage",
+ docs: STORAGE_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.storage),
+ },
+ { path: "storage.locations[]", docs: STORAGE_LOCATION_FIELD_DOCS },
+ { path: "storage.locations[].volume", docs: STORAGE_VOLUME_FIELD_DOCS },
+ ],
+ buildPipeline: [
+ {
+ path: "buildPipeline",
+ docs: BUILD_PIPELINE_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.buildPipeline),
+ },
+ ],
+ digest: [
+ { path: "digest", docs: DIGEST_SETTINGS_FIELD_DOCS, defaults: fromObject(d.digest) },
+ { path: "digest.apps.<appId>", docs: DIGEST_APP_CONFIG_FIELD_DOCS },
+ ],
+ diarization: [
+ {
+ path: "diarization",
+ docs: DIARIZATION_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.diarization),
+ },
+ ],
+ backfill: [
+ {
+ path: "backfill",
+ docs: BACKFILL_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.backfill),
+ },
+ ],
+ attribution: [
+ {
+ path: "attribution",
+ docs: ATTRIBUTION_SETTINGS_FIELD_DOCS,
+ defaults: fromObject(d.attribution),
+ },
+ ],
+ };
+}
+
+function renderTable(out: string[], table: KeyTable): void {
+ out.push(`#### \`${table.path}\``);
+ out.push("");
+ if (table.defaults) {
+ out.push("| Key | Default | Description |");
+ out.push("|---|---|---|");
+ for (const [key, text] of Object.entries(table.docs)) {
+ out.push(`| \`${key}\` | ${table.defaults(key)} | ${cell(text)} |`);
+ }
+ } else {
+ out.push("Per entry — each entry spells its own values.");
+ out.push("");
+ out.push("| Key | Description |");
+ out.push("|---|---|");
+ for (const [key, text] of Object.entries(table.docs)) {
+ out.push(`| \`${key}\` | ${cell(text)} |`);
+ }
+ }
+ out.push("");
+}
+
export function renderSettingsMarkdown(): string {
- const d = defaultSiteSettings() as Record<string, unknown>;
+ const d = defaultSiteSettings();
+ const values = d as Record<string, unknown>;
const shape = siteSettingsSchema.shape as Record<
string,
{ description?: string }
>;
+ const tables = blockTables(d);
const keys = Object.keys(shape);
const out: string[] = [];
out.push("# settings.json keys");
out.push("");
out.push(
- "<!-- GENERATED by common/bin/settings-example.ts from common/lib/settingsSchema.ts — do not edit by hand. -->",
+ "<!-- GENERATED by common/bin/settings-example.ts from common/lib/settingsSchema.ts and the *_FIELD_DOCS records beside each block type — do not edit by hand. -->",
);
out.push("");
out.push(
@@ -70,14 +262,22 @@ export function renderSettingsMarkdown(): string {
out.push(
"`settings.json.example` is the default object with one key left out, " +
"`workers`: a file that does not name it gets a worker list synthesized " +
- "from `transcriptionApp` on read, where `workers: []` would mean no " +
- "transcription at all.",
+ "from `transcriptionApp` on read. A file that spells `workers: []` READS " +
+ "as no transcription at all — until the next save, when the writer " +
+ "synthesizes a worker the same way.",
+ );
+ out.push("");
+ out.push(
+ "A copied example PINS every default it spells — including each lane's " +
+ "`autoQueue.<lane>.held` — so a default changed in a later release will " +
+ "not reach that file. Delete any key you would rather have track the " +
+ "defaults.",
);
out.push("");
out.push("| Key | Default |");
out.push("|---|---|");
for (const key of keys) {
- out.push(`| [\`${key}\`](#${key.toLowerCase()}) | ${defaultCell(d[key])} |`);
+ out.push(`| [\`${key}\`](#${key.toLowerCase()}) | ${defaultCell(values[key])} |`);
}
out.push("");
for (const key of keys) {
@@ -85,17 +285,21 @@ export function renderSettingsMarkdown(): string {
out.push("");
out.push(shape[key].description ?? "");
out.push("");
- const v = d[key];
+ const v = values[key];
if (isScalar(v)) {
out.push(`Default: \`${JSON.stringify(v)}\``);
+ out.push("");
} else {
+ for (const table of tables[key as keyof SiteSettings] ?? []) {
+ renderTable(out, table);
+ }
out.push("Default:");
out.push("");
out.push("```json");
out.push(JSON.stringify(v, null, 2));
out.push("```");
+ out.push("");
}
- out.push("");
}
return out.join("\n");
}
diff --git a/common/lib/settingsFieldSchemas.ts b/common/lib/settingsFieldSchemas.ts
@@ -8,9 +8,8 @@
// its home and its signature. What this file adds is ONE schema per block, so
// `lib/settingsSchema.ts` can compose them with the other twenty-eight fields
// the same way it composes everything: zod supplies the plumbing (the key, the
-// strip of unknown siblings, a `.catch` that makes a field total), the existing
-// sanitizer supplies the arithmetic. Nothing was re-implemented, so nothing
-// could drift.
+// strip of unknown siblings), the existing sanitizer supplies the arithmetic —
+// and the totality. Nothing was re-implemented, so nothing could drift.
//
// WHY A SEPARATE FILE, and not a `workersSchema` export beside each sanitizer:
// all three homes are imported AS VALUES by `"use client"` forms
@@ -28,13 +27,20 @@ import {
import { sanitizeAutoQueue } from "./autoQueueSchema";
import type { AutoQueueSettings } from "./autoQueueTypes";
-// A settings FIELD: any JSON value in, a legal value out, never a throw.
+// A settings FIELD: any JSON value in, a legal value out.
//
-// `.catch(undefined)` is today's try-and-continue: whatever zod could object
-// to becomes `undefined`, which every coercion below treats as "absent → the
-// default". `z.unknown()` accepts a missing key, and zod 4 still runs the
-// transform for it and emits the key — so an absent field is DEFAULTED, not
-// dropped, exactly as `defaults()` used to fill it.
+// TOTALITY COMES FROM THE COERCION, NOT FROM ZOD. `z.unknown()` accepts every
+// input, so its `.catch(undefined)` never fires, and zod does not guard the
+// transform: a coercion that threw would throw out of `parse`. What makes a
+// settings read never throw is that every coercion passed here — each clamp and
+// sanitizer — is itself total over `unknown`. The `.catch` is kept only so the
+// field keeps its shape if `z.unknown()` is ever swapped for a validating
+// schema.
+//
+// What zod does contribute: `z.unknown()` accepts a missing key, and zod 4
+// still runs the transform for it and emits the key — so an absent field is
+// DEFAULTED, not dropped, exactly as `defaults()` used to fill it — and unknown
+// sibling keys are stripped by the enclosing object.
//
// There is deliberately no `.default()` anywhere: a default applies only to
// `undefined`, and every coercion here already decides that case itself — the
diff --git a/common/lib/settingsSchema.ts b/common/lib/settingsSchema.ts
@@ -14,7 +14,12 @@
// ZOD SUPPLIES THE PLUMBING, NOT THE ARITHMETIC. Every field is
// `settingsField(coerce)` — `z.unknown().catch(undefined).transform(coerce)` —
// and every `coerce` is the clamp or sanitizer that already existed, reused, so
-// no boundary moved. Unknown keys are dropped by zod's default strip (never
+// no boundary moved. Each is total over `unknown`, and that — not zod, whose
+// `.catch` cannot fire on `z.unknown()` — is what makes a read never throw.
+// The nested blocks' own fields are documented in the `*_FIELD_DOCS` record
+// beside each block type (here and in workers.ts, channelPriority.ts,
+// autoQueueTypes.ts, storageLocations.ts, transcriptionApps.ts, digest.ts),
+// type-checked complete. Unknown keys are dropped by zod's default strip (never
// `.passthrough()`), which is what retires a field: a key the schema does not
// name cannot survive a read or a save.
//
@@ -82,6 +87,7 @@ import {
type DigestSectionKind,
type DigestTimestampMode,
} from "./digest";
+import type { FieldDocs } from "./fieldDocs";
export type { Worker } from "./workers";
export type { AutoQueueSettings } from "./autoQueueTypes";
@@ -109,42 +115,53 @@ export {
// between them (the backfill lane's yield deliberately watches only the
// transcription lane). Nothing here arms anything; a pilot decides whether the
// corpus-wide text-only pass is worth 25-55 GPU-days at all.
+// Each field is documented in ATTRIBUTION_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
export type AttributionSettings = {
- // Master switch. Off means the backfill registry reports no attribution work
- // at all — the feature gate every Operation has.
enabled: boolean;
- // Which digest app runs the naming. Attribution IS a digest-app workload —
- // constrained JSON decoding over transcript text — so it reuses that registry
- // and that per-app config (settings.digest.apps[appId]) rather than growing a
- // second copy of the ollama URL, context size and timeout.
appId: string;
- // Model override. Empty = the app's configured model, then its default. It is
- // separate from the digest's because the two workloads may want different
- // sizes, and because it is part of the freshness identity: sharing the digest's
- // model field would make a digest bake-off invalidate every attribution record
- // on disk as a side effect.
model: string;
- // The lanes, separately. Both default OFF even when `enabled` is on, so
- // turning the feature on to look at it cannot start a corpus sweep.
- //
- // They are not a fallback pair. `diarized` is one call per video and grounded
- // in acoustic clustering; `textOnly` is ~30 calls and guesses at identity
- // across chunk seams. An operator may reasonably want the first forever and
- // the second never.
diarizedEnabled: boolean;
textOnlyEnabled: boolean;
- // The prompt generation a record must match to count as fresh.
- //
- // Defaults to (and is floored at) ATTRIBUTION_PROMPT_VERSION, the shipped
- // constant. Raising it forces a corpus-wide regeneration without a code
- // change, which is the honest way to redo everything after a prompt tweak.
- // It cannot be set BELOW the shipped constant, and that floor is the lesson
- // from digestPrompt.ts's version 1 -> 2 note: pinning freshness to an older
- // generation freezes output from a superseded prompt into the corpus, looking
- // identical to output from the current one.
promptVersion: number;
};
+export const ATTRIBUTION_SETTINGS_FIELD_DOCS: FieldDocs<AttributionSettings> = {
+ enabled:
+ "Master switch. Off means the backfill registry reports no attribution " +
+ "work at all — the feature gate every Operation has.",
+ appId:
+ "Which digest app runs the naming. Attribution IS a digest-app workload" +
+ " — constrained JSON decoding over transcript text — so it reuses that " +
+ "registry and that per-app config (settings.digest.apps[appId]) rather " +
+ "than growing a second copy of the ollama URL, context size and " +
+ "timeout.",
+ model:
+ "Model override. Empty = the app's configured model, then its default. " +
+ "It is separate from the digest's because the two workloads may want " +
+ "different sizes, and because it is part of the freshness identity: " +
+ "sharing the digest's model field would make a digest bake-off " +
+ "invalidate every attribution record on disk as a side effect.",
+ diarizedEnabled:
+ "The lanes, separately. Both default OFF even when `enabled` is on, so " +
+ "turning the feature on to look at it cannot start a corpus sweep.\n\n" +
+ "They are not a fallback pair. `diarized` is one call per video and " +
+ "grounded in acoustic clustering; `textOnly` is ~30 calls and guesses " +
+ "at identity across chunk seams. An operator may reasonably want the " +
+ "first forever and the second never.",
+ textOnlyEnabled:
+ "The text-only attribution lane: names speakers from the transcript alone (~30 model calls per video, guessing identity across chunk seams). Default OFF even when `enabled` is on. See `diarizedEnabled` — the two are separate lanes, not a fallback pair.",
+ promptVersion:
+ "The prompt generation a record must match to count as fresh.\n\n" +
+ "Defaults to (and is floored at) ATTRIBUTION_PROMPT_VERSION, the " +
+ "shipped constant. Raising it forces a corpus-wide regeneration without" +
+ " a code change, which is the honest way to redo everything after a " +
+ "prompt tweak. It cannot be set BELOW the shipped constant, and that " +
+ "floor is the lesson from digestPrompt.ts's version 1 -> 2 note: " +
+ "pinning freshness to an older generation freezes output from a " +
+ "superseded prompt into the corpus, looking identical to output from " +
+ "the current one.",
+};
+
// Configuration for the backfill lane — the generic answer to "a derived-data
// feature landed and 77,000 existing videos do not have it".
//
@@ -157,29 +174,37 @@ export type AttributionSettings = {
// THE SHARE IS `concurrency` AND THE LANE'S `autoQueue.backfill.maxWorkers`; the
// yield is the operation's declared `contendsFor`. Slice 1.3 retired the
// `weight` scalar that used to mean both — see backfillLimit().
+// Each field is documented in BACKFILL_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
export type BackfillSettings = {
- // Slots the lane may use when it is not standing aside. Kept at 1 by default
- // for the same reason diarization.concurrency is: this is CPU-bound work
- // competing with GPU feeding and the digest sweep for the same 8 threads.
concurrency: number;
- // Re-acquire media for videos whose input is GONE (audio deleted after
- // transcription). OFF by default and deliberately so: measured on this corpus,
- // 836 videos still have media and ~76,270 would need a re-download — 91x the
- // reachable work, against 45 GB free at 97% full. When on, each re-fetched
- // file is removed in a `finally` as soon as the backfill has used it, unless
- // the video is marked do-not-clean, or unless the auto-transcribe policy would
- // replace its auto-captions (`replaceAutoSubs`, or a leaf on
- // `downloadedAutoSubsOnly`), in which case the audio is kept for that runner.
- //
- // WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: "youtube"`
- // channel — which normally only fetches subtitles — the re-acquire applies a
- // PER-VIDEO transcribe override so yt-dlp lands audio a diarizer can read;
- // the channel's stored config is not changed. Without that override the fetch
- // re-downloads the captions the video already has and lands nothing, which is
- // what happened to ~16,000 videos on eight channels in 2026-08.
allowRedownload: boolean;
};
+export const BACKFILL_SETTINGS_FIELD_DOCS: FieldDocs<BackfillSettings> = {
+ concurrency:
+ "Slots the lane may use when it is not standing aside. Kept at 1 by " +
+ "default for the same reason diarization.concurrency is: this is CPU-" +
+ "bound work competing with GPU feeding and the digest sweep for the " +
+ "same 8 threads.",
+ allowRedownload:
+ "Re-acquire media for videos whose input is GONE (audio deleted after " +
+ "transcription). OFF by default and deliberately so: measured on this " +
+ "corpus, 836 videos still have media and ~76,270 would need a re-" +
+ "download — 91x the reachable work, against 45 GB free at 97% full. " +
+ "When on, each re-fetched file is removed in a `finally` as soon as the" +
+ " backfill has used it, unless the video is marked do-not-clean, or " +
+ "unless the auto-transcribe policy would replace its auto-captions " +
+ "(`replaceAutoSubs`, or a leaf on `downloadedAutoSubsOnly`), in which " +
+ "case the audio is kept for that runner.\n\n" +
+ "WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: " +
+ "\"youtube\"` channel — which normally only fetches subtitles — the re-" +
+ "acquire applies a PER-VIDEO transcribe override so yt-dlp lands audio " +
+ "a diarizer can read; the channel's stored config is not changed. " +
+ "Without that override the fetch re-downloads the captions the video " +
+ "already has and lands nothing, which is what happened to ~16,000 " +
+ "videos on eight channels in 2026-08.",
+};
+
// Configuration for the speaker-diarization capture lane.
//
// This exists because AUDIO IS THE ONE PERISHABLE INPUT in the pipeline:
@@ -188,268 +213,375 @@ export type BackfillSettings = {
// capture half is deliberately all that ships here — attribution, LLM speaker
// naming, viewer badges and quote filtering can all be redone later from the
// saved JSON, whereas the audio cannot.
+// Each field is documented in DIARIZATION_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
export type DiarizationSettings = {
- // Master switch. OFF by default so a transcription batch can start before this
- // lands, with diarization backfilled over the retained audio afterwards.
- //
- // Turning it ON also arms the cleanup guard: the Clean-audio sweep stops
- // deleting audio for a transcribed video that has no diarization.json yet.
- // That is the point — it is what keeps the perishable input alive long enough
- // to be captured — but it means enabling this holds disk.
enabled: boolean;
- // Run diarization inline in the post-transcribe hook.
- //
- // OFF by default, and that default is a MEASURED decision, not caution.
- // Measured on this box: GPU transcription runs at 221 s/audio-hour (16.3x
- // realtime, over 3,602 real videos), CPU diarization at ~500-680 s/audio-hour.
- // Diarization is therefore ~2-3x SLOWER than the transcription it follows, so
- // running it inline drops whole-pipeline throughput by roughly 3-4x and leaves
- // the GPU idle while the CPU catches up.
- //
- // The intended sequence for a large batch is the opposite: leave this off, let
- // the batch transcribe at full GPU speed with `enabled` holding the audio, and
- // diarize afterwards with the backfill pass. Turn it on for steady state, once
- // the arrival rate is a few videos a day rather than a corpus.
inlineAfterTranscribe: boolean;
- // Clustering threshold — the single most consequential knob, since it decides
- // how many speakers come out. Larger merges more aggressively.
- //
- // The default is 0.9, NOT sherpa-onnx's own 0.5, and that is measured on this
- // corpus. On a 6-minute excerpt of a two-person interview (known ground truth:
- // 2 speakers), sherpa's default produced 22 clusters; 0.9 produced 6, with the
- // top two at 40%/40% of talk time — recognizably the two hosts. Sweep on the
- // same clip: 0.4→23, 0.5→22, 0.6→17, 0.7→12, 0.8→10, 0.9→6.
- //
- // It still over-splits, which is why this is a capture lane and not an answer:
- // the turns are recorded with the threshold that produced them, so a later
- // attribution pass can re-cluster or re-run without needing the audio back.
threshold: number;
- // Engine threads per diarize run.
threads: number;
- // Which engine runs. "sherpa-onnx" is the shipped default and what every
- // sidecar on disk was produced by; "sortformer" is the ggml engine built by
- // scripts/build-sortformer.sh.
- //
- // CHANGING THIS RESTATES THE FRESHNESS IDENTITY (see diarizationTarget), so
- // every sidecar written by the other engine becomes stale and the backfill lane
- // offers to redo it. That is intended — the two disagree about how many
- // speakers exist, and a corpus half-diarized by each is not one corpus — but on
- // the retained audio it is weeks of work, not a toggle.
- //
- // Why anyone would: on the same file, sherpa at its tuned threshold returns 13
- // speakers and sortformer returns 4, agreeing on the dominant speaker's share
- // to within half a point (73.1% vs 73.5%). On the corpus's worst case sherpa
- // returns 35 and sortformer 4. Over-splitting is the failure mode this lane has
- // always had, and sortformer is end-to-end rather than clustered, so it does
- // not have it. The cost is a hard ceiling of 4 speakers and ~1.8x the wall
- // clock.
engine: DiarizationEngineId;
- // Compute device for the sortformer engine; ignored by sherpa-onnx, which has
- // no Vulkan compute path on Linux.
- //
- // "vulkan" is 1.5x faster than a thread-tuned CPU run (894 vs 1305
- // s/audio-hour, measured on this box) and holds 558 MB resident instead of
- // 4.84 GB by keeping weights and activations in VRAM. It also takes ~4.4 GB of
- // an 8 GB card, which is why the lane YIELDS to transcription rather than
- // sharing — see controller/digestYield.ts.
backend: DiarizationBackend;
- // Python interpreter for the default sherpa-onnx engine. sherpa-onnx ships
- // wheels only up to cp313, and this box's system python is 3.14 — so this
- // usually points at a dedicated venv rather than `python3`.
python: string;
- // ONNX model paths for the default engine. Empty = the lane cannot run, which
- // is reported as a skip rather than a failure.
segModel: string;
embModel: string;
- // Binary and model for the sortformer engine, both produced by
- // scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the
- // same "not-configured" skip as an unset segModel/embModel.
sortformerBin: string;
sortformerModel: string;
- // How many diarize runs may execute at once in the backfill pass. Kept low by
- // default: diarization is CPU-bound and competes with GPU feeding and the
- // digest sweep for the same 8 threads.
concurrency: number;
- // Videos longer than this are DEFERRED rather than diarized: reported as a
- // third number that is never summed into reachable work, so a capped corpus
- // can never read as finished.
- //
- // THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering holds a
- // pairwise distance matrix over speech-segment embeddings — O(n^2) in SEGMENT
- // count — and speaker-turn density varies 40x across this corpus (33-1364
- // turns/hour), so duration does not actually predict the blowup: a sparse
- // 7h42m video completed while a dense 6h12m one was OOM-killed. Duration is
- // merely the only predictor available for free, from metadata already on disk,
- // BEFORE spending 45 minutes to find out. n^2 at 30k segments is 6.7 GiB and
- // at 40k is 11.9 GiB, which brackets the 10.6 GB and 9.6 GB peaks measured on
- // this 16 GB box.
- //
- // 0 disables the cap. That is where this goes once windowed diarization lands:
- // windowing divides per-window n by the window count, so the matrix falls by
- // its square, and the cap stops being needed rather than being tuned.
maxAudioHours: number;
};
+export const DIARIZATION_SETTINGS_FIELD_DOCS: FieldDocs<DiarizationSettings> = {
+ enabled:
+ "Master switch. OFF by default so a transcription batch can start " +
+ "before this lands, with diarization backfilled over the retained audio" +
+ " afterwards.\n\n" +
+ "Turning it ON also arms the cleanup guard: the Clean-audio sweep stops" +
+ " deleting audio for a transcribed video that has no diarization.json " +
+ "yet. That is the point — it is what keeps the perishable input alive " +
+ "long enough to be captured — but it means enabling this holds disk.",
+ inlineAfterTranscribe:
+ "Run diarization inline in the post-transcribe hook.\n\n" +
+ "OFF by default, and that default is a MEASURED decision, not caution. " +
+ "Measured on this box: GPU transcription runs at 221 s/audio-hour " +
+ "(16.3x realtime, over 3,602 real videos), CPU diarization at ~500-680 " +
+ "s/audio-hour. Diarization is therefore ~2-3x SLOWER than the " +
+ "transcription it follows, so running it inline drops whole-pipeline " +
+ "throughput by roughly 3-4x and leaves the GPU idle while the CPU " +
+ "catches up.\n\n" +
+ "The intended sequence for a large batch is the opposite: leave this " +
+ "off, let the batch transcribe at full GPU speed with `enabled` holding" +
+ " the audio, and diarize afterwards with the backfill pass. Turn it on " +
+ "for steady state, once the arrival rate is a few videos a day rather " +
+ "than a corpus.",
+ threshold:
+ "Clustering threshold — the single most consequential knob, since it " +
+ "decides how many speakers come out. Larger merges more aggressively.\n\n" +
+ "The default is 0.9, NOT sherpa-onnx's own 0.5, and that is measured on" +
+ " this corpus. On a 6-minute excerpt of a two-person interview (known " +
+ "ground truth: 2 speakers), sherpa's default produced 22 clusters; 0.9 " +
+ "produced 6, with the top two at 40%/40% of talk time — recognizably " +
+ "the two hosts. Sweep on the same clip: 0.4→23, 0.5→22, 0.6→17, 0.7→12," +
+ " 0.8→10, 0.9→6.\n\n" +
+ "It still over-splits, which is why this is a capture lane and not an " +
+ "answer: the turns are recorded with the threshold that produced them, " +
+ "so a later attribution pass can re-cluster or re-run without needing " +
+ "the audio back.",
+ threads:
+ "Engine threads per diarize run.",
+ engine:
+ "Which engine runs. \"sherpa-onnx\" is the shipped default and what every" +
+ " sidecar on disk was produced by; \"sortformer\" is the ggml engine " +
+ "built by scripts/build-sortformer.sh.\n\n" +
+ "CHANGING THIS RESTATES THE FRESHNESS IDENTITY (see diarizationTarget)," +
+ " so every sidecar written by the other engine becomes stale and the " +
+ "backfill lane offers to redo it. That is intended — the two disagree " +
+ "about how many speakers exist, and a corpus half-diarized by each is " +
+ "not one corpus — but on the retained audio it is weeks of work, not a " +
+ "toggle.\n\n" +
+ "Why anyone would: on the same file, sherpa at its tuned threshold " +
+ "returns 13 speakers and sortformer returns 4, agreeing on the dominant" +
+ " speaker's share to within half a point (73.1% vs 73.5%). On the " +
+ "corpus's worst case sherpa returns 35 and sortformer 4. Over-splitting" +
+ " is the failure mode this lane has always had, and sortformer is end-" +
+ "to-end rather than clustered, so it does not have it. The cost is a " +
+ "hard ceiling of 4 speakers and ~1.8x the wall clock.",
+ backend:
+ "Compute device for the sortformer engine; ignored by sherpa-onnx, " +
+ "which has no Vulkan compute path on Linux.\n\n" +
+ "\"vulkan\" is 1.5x faster than a thread-tuned CPU run (894 vs 1305 " +
+ "s/audio-hour, measured on this box) and holds 558 MB resident instead " +
+ "of 4.84 GB by keeping weights and activations in VRAM. It also takes " +
+ "~4.4 GB of an 8 GB card, which is why the lane YIELDS to transcription" +
+ " rather than sharing — see controller/digestYield.ts.",
+ python:
+ "Python interpreter for the default sherpa-onnx engine. sherpa-onnx " +
+ "ships wheels only up to cp313, and this box's system python is 3.14 — " +
+ "so this usually points at a dedicated venv rather than `python3`.",
+ segModel:
+ "ONNX model paths for the default engine. Empty = the lane cannot run, " +
+ "which is reported as a skip rather than a failure.",
+ embModel:
+ "ONNX speaker-embedding model path for the sherpa-onnx engine. Empty = the lane cannot run, reported as a `not-configured` skip rather than a failure (same as `segModel`).",
+ sortformerBin:
+ "Binary and model for the sortformer engine, both produced by " +
+ "scripts/build-sortformer.sh. Empty = that engine cannot run, reported " +
+ "as the same \"not-configured\" skip as an unset segModel/embModel.",
+ sortformerModel:
+ "Model for the sortformer engine, produced by scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the same `not-configured` skip as an unset `sortformerBin`.",
+ concurrency:
+ "How many diarize runs may execute at once in the backfill pass. Kept " +
+ "low by default: diarization is CPU-bound and competes with GPU feeding" +
+ " and the digest sweep for the same 8 threads.",
+ maxAudioHours:
+ "Videos longer than this are DEFERRED rather than diarized: reported as" +
+ " a third number that is never summed into reachable work, so a capped " +
+ "corpus can never read as finished.\n\n" +
+ "THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering " +
+ "holds a pairwise distance matrix over speech-segment embeddings — " +
+ "O(n^2) in SEGMENT count — and speaker-turn density varies 40x across " +
+ "this corpus (33-1364 turns/hour), so duration does not actually " +
+ "predict the blowup: a sparse 7h42m video completed while a dense 6h12m" +
+ " one was OOM-killed. Duration is merely the only predictor available " +
+ "for free, from metadata already on disk, BEFORE spending 45 minutes to" +
+ " find out. n^2 at 30k segments is 6.7 GiB and at 40k is 11.9 GiB, " +
+ "which brackets the 10.6 GB and 9.6 GB peaks measured on this 16 GB " +
+ "box.\n\n" +
+ "0 disables the cap. That is where this goes once windowed diarization " +
+ "lands: windowing divides per-window n by the window count, so the " +
+ "matrix falls by its square, and the cap stops being needed rather than" +
+ " being tuned.",
+};
+
// Configuration for the derived-corpus digest layer. Local-first by decision:
// `remoteEnabled` gates the metered lane and defaults to false, so nothing here
// can spend money until it is explicitly turned on.
+// Each field is documented in DIGEST_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
export type DigestSettings = {
- // Master switch for the metered (remote-api) lane. OFF by default — an opt-in
- // overflow for the long tail or a channel where local quality is poor, never
- // the default path.
remoteEnabled: boolean;
- // Videos longer than this are "long tail": 8.2% of the corpus by count, 46% of
- // all transcript tokens. The batch's duration-aware ordering and the optional
- // remote overflow both key off it.
longTailSeconds: number;
- // The engine each lane uses (ids from common/lib/digestApps.ts).
localAppId: string;
remoteAppId: string;
- // Per-app config, keyed by app id — the same id-keyed sub-record shape as
- // transcriptionApps.
apps: Record<string, DigestAppConfig>;
- // Yield the GPU to the transcription lane: while transcription is working, the
- // digest batch's limit() returns 0 and the pool idle-waits. ON by default,
- // because `digest:local` is deliberately on a different queue from
- // TRANSCRIPTION_QUEUE and so would otherwise run ollama and the transcription
- // engine on the same 8 GB card. See controller/digestYield.ts.
yieldToTranscription: boolean;
- // Whether a busy worker pinned to `device: "cpu"` counts as GPU contention.
- //
- // OFF by default, which is the FIX for a real bug: the yield originally tested
- // only `kind === "local"`, so on a box with one GPU worker and two CPU-pinned
- // ones (this box, at parallelTranscriptions 2) the digest lane stopped dead for
- // transcription that competes for zero GPU shaders.
- //
- // Only an EXPLICIT "cpu" is treated as non-contending. A worker with no device
- // set is using the engine binary's own default, which may be the GPU, so it
- // still triggers the yield — the unknown case fails safe.
- //
- // Composes with `yieldToTranscription`: that is the master switch, this only
- // narrows which workers it reacts to.
yieldToCpuWorkers: boolean;
- // Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. Only ever
- // consulted for a metered app.
spendCapUsd: number;
- // Which sections a sweep generates.
- //
- // Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, and that
- // is not a contradiction: a tag call sends the same transcript as the chapter
- // call before it, so it hits the engine's cached prefix and pays essentially no
- // prefill (+0.4 s across 4 extra calls, against 22.4 s for the first 4). All it
- // pays is decode, and a tag list is ~30 output tokens where a chapter list is
- // ~200–290.
- //
- // The corollary matters more than the number: run them in the SAME pass. Tags
- // generated later, on their own, pay full prefill again — measured at 44% of a
- // whole chapters pass, i.e. 3–9× the marginal cost of just including them now.
sections: DigestSectionKind[];
- // How each chunk's transcript markers are numbered — see DigestTimestampMode.
- // Was a scored variable in the bake-off rather than a pre-applied fix; the
- // measurement is in and "chunk-local" is now the shipped default.
timestampMode: DigestTimestampMode;
- // A free-text label for a non-default prompt shape, folded into the recorded
- // provenance by digestPromptVariant(). Setting it invalidates every digest
- // generated under a different label, which is exactly what makes a bake-off
- // round re-run its sample instead of skipping it as fresh. Empty = default.
promptVariant: string;
};
+export const DIGEST_SETTINGS_FIELD_DOCS: FieldDocs<DigestSettings> = {
+ remoteEnabled:
+ "Master switch for the metered (remote-api) lane. OFF by default — an " +
+ "opt-in overflow for the long tail or a channel where local quality is " +
+ "poor, never the default path.",
+ longTailSeconds:
+ "Videos longer than this are \"long tail\": 8.2% of the corpus by count, " +
+ "46% of all transcript tokens. The batch's duration-aware ordering and " +
+ "the optional remote overflow both key off it.",
+ localAppId:
+ "The engine each lane uses (ids from common/lib/digestApps.ts).",
+ remoteAppId:
+ "The engine the metered (remote-api) lane uses — an id from common/lib/digestApps.ts. Unknown ids degrade to the default app rather than failing.",
+ apps:
+ "Per-app config, keyed by app id — the same id-keyed sub-record shape " +
+ "as transcriptionApps.",
+ yieldToTranscription:
+ "Yield the GPU to the transcription lane: while transcription is " +
+ "working, the digest batch's limit() returns 0 and the pool idle-waits." +
+ " ON by default, because `digest:local` is deliberately on a different " +
+ "queue from TRANSCRIPTION_QUEUE and so would otherwise run ollama and " +
+ "the transcription engine on the same 8 GB card. See " +
+ "controller/digestYield.ts.",
+ yieldToCpuWorkers:
+ "Whether a busy worker pinned to `device: \"cpu\"` counts as GPU " +
+ "contention.\n\n" +
+ "OFF by default, which is the FIX for a real bug: the yield originally " +
+ "tested only `kind === \"local\"`, so on a box with one GPU worker and " +
+ "two CPU-pinned ones (this box, at parallelTranscriptions 2) the digest" +
+ " lane stopped dead for transcription that competes for zero GPU " +
+ "shaders.\n\n" +
+ "Only an EXPLICIT \"cpu\" is treated as non-contending. A worker with no " +
+ "device set is using the engine binary's own default, which may be the " +
+ "GPU, so it still triggers the yield — the unknown case fails safe.\n\n" +
+ "Composes with `yieldToTranscription`: that is the master switch, this " +
+ "only narrows which workers it reacts to.",
+ spendCapUsd:
+ "Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. " +
+ "Only ever consulted for a metered app.",
+ sections:
+ "Which sections a sweep generates.\n\n" +
+ "Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, " +
+ "and that is not a contradiction: a tag call sends the same transcript " +
+ "as the chapter call before it, so it hits the engine's cached prefix " +
+ "and pays essentially no prefill (+0.4 s across 4 extra calls, against " +
+ "22.4 s for the first 4). All it pays is decode, and a tag list is ~30 " +
+ "output tokens where a chapter list is ~200–290.\n\n" +
+ "The corollary matters more than the number: run them in the SAME pass." +
+ " Tags generated later, on their own, pay full prefill again — measured" +
+ " at 44% of a whole chapters pass, i.e. 3–9× the marginal cost of just " +
+ "including them now.",
+ timestampMode:
+ "How each chunk's transcript markers are numbered — see " +
+ "DigestTimestampMode. Was a scored variable in the bake-off rather than" +
+ " a pre-applied fix; the measurement is in and \"chunk-local\" is now the" +
+ " shipped default.",
+ promptVariant:
+ "A free-text label for a non-default prompt shape, folded into the " +
+ "recorded provenance by digestPromptVariant(). Setting it invalidates " +
+ "every digest generated under a different label, which is exactly what " +
+ "makes a bake-off round re-run its sample instead of skipping it as " +
+ "fresh. Empty = default.",
+};
+
// "basic" — `pnpm run build` in export/, serialized on the build queue (shared
// output tree → no safe parallelism).
// "docker" — isolated per-site container builds (follow-up); enables real
// parallel multi-site builds capped by maxParallelBuilds.
export type BuildMode = "basic" | "docker";
+// Each field is documented in BUILD_PIPELINE_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
export type BuildPipelineSettings = {
mode: BuildMode;
- // Cap on concurrent per-site container builds in docker mode. Ignored in basic
- // mode (which is always serial). Clamped to [1, BUILD_MAX_PARALLEL_MAX].
maxParallelBuilds: number;
- // Tag of the reusable build image (built once, reused for every site).
dockerImage: string;
- // Dockerfile path relative to the monorepo root, used to (re)build the image.
dockerfile: string;
};
+export const BUILD_PIPELINE_SETTINGS_FIELD_DOCS: FieldDocs<BuildPipelineSettings> = {
+ mode:
+ "\"basic\" — `pnpm run build` in export/, serialized on the build queue (shared output tree, no safe parallelism). \"docker\" — isolated per-site container builds, parallel up to `maxParallelBuilds`.",
+ maxParallelBuilds:
+ "Cap on concurrent per-site container builds in docker mode. Ignored in" +
+ " basic mode (which is always serial). Clamped to [1, " +
+ "BUILD_MAX_PARALLEL_MAX].",
+ dockerImage:
+ "Tag of the reusable build image (built once, reused for every site).",
+ dockerfile:
+ "Dockerfile path relative to the monorepo root, used to (re)build the " +
+ "image.",
+};
+
+// Each field is documented in SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
export type SavedVideoBackupSettings = {
- // Master switch for the scheduled backup. A backup can still be run manually
- // when this is false, as long as a destination is set.
enabled: boolean;
- // Destination root the store is mirrored into (a local path or any rsync
- // target). Empty disables both scheduled and manual backups.
dest: string;
- // Cadence (minutes) for the scheduled backup when enabled. Clamped into the
- // sync-interval window; default daily.
intervalMinutes: number;
};
+export const SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS: FieldDocs<SavedVideoBackupSettings> = {
+ enabled:
+ "Master switch for the scheduled backup. A backup can still be run " +
+ "manually when this is false, as long as a destination is set.",
+ dest:
+ "Destination root the store is mirrored into (a local path or any rsync" +
+ " target). Empty disables both scheduled and manual backups.",
+ intervalMinutes:
+ "Cadence (minutes) for the scheduled backup when enabled. Clamped into " +
+ "the sync-interval window; default daily.",
+};
+
+// Each field is documented in SYNC_SCHEDULER_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
export type SyncSchedulerSettings = {
- // Master switch. When false, a tick selects nothing (manual sync still works).
enabled: boolean;
- // Fallback cadence (minutes) for channels with no per-channel override.
defaultIntervalMinutes: number;
- // Cap on sync jobs running/queued at once. A tick queues at most
- // (cap - currently-active) channels; the rest roll to the next tick. This is
- // also the stagger mechanism that keeps a big due-batch from hitting the
- // source all at once.
maxConcurrentSyncs: number;
- // Optional local-clock quiet window during which auto-sync is suppressed.
- // Both null = always allowed. The window may wrap past midnight
- // (e.g. start=22, end=6). Hours are [0,23]; the window is [start, end).
quietHoursStart: number | null;
quietHoursEnd: number | null;
- // Failure backoff bounds. After N consecutive failed scheduled syncs a
- // channel waits min(base * 2^(N-1), max) minutes before it's eligible again.
backoffBaseMinutes: number;
backoffMaxMinutes: number;
- // Cadence (seconds) for the editor's in-process heartbeat — the internal timer
- // armed by the instrumentation hook (editor/instrumentation.ts) that calls the
- // scheduler tick directly, so no external cron is needed. 0 = off: rely on the
- // external `pnpm sync:tick` heartbeat instead. Any positive value is clamped to
- // [SYNC_HEARTBEAT_MIN_SECONDS, SYNC_HEARTBEAT_MAX_SECONDS]. The env var
- // SYNC_HEARTBEAT_SECONDS overrides this at runtime. See SCHEDULED_SYNC.md.
heartbeatSeconds: number;
- // Cadence (minutes) for the scheduled keep-latest deletion check. For each
- // channel with ChannelConfig.keepLatest > 0, the tick re-probes the kept
- // window for source deletion (checkKeptDeletedAction) at most this often and
- // pins any gone videos as do-not-clean. Clamped into the sync-interval window;
- // default daily. The check shares the same concurrency cap and quiet-hours
- // window as scheduled syncs. See editor/app/scheduler/runTick.ts.
keepLatestCheckIntervalMinutes: number;
- // Default cadence (minutes) for the sync FULL SWEEP — the deep pass that
- // re-enumerates a channel's whole listing in one yt-dlp spawn, refreshes the
- // stored `playlist` file, and flags videos that have left the listing into
- // maybe-missing.json. Ordinary syncs stay on the cheap newest-first paged
- // walk; a sync only upgrades itself to a sweep when this interval has elapsed
- // since the channel's lastFullSweepAt. Per-channel override:
- // ChannelConfig.fullSweepIntervalMinutes. 0 = never sweep. Default daily.
- // See common/jobs/deepSync.ts.
fullSweepIntervalMinutes: number;
- // Upper bound on how many maybe-missing suspects a full sweep will resolve
- // in-line with the per-video availability probe (deleted vs private vs
- // unlisted). At or under the cap the sweep runs the targeted check itself, so
- // "Sync all" surfaces upstream deletions with no extra clicks; over it, the
- // suspects are flagged and left for a manual check rather than firing hundreds
- // of probes inside a sync. 0 = never auto-confirm.
fullSweepConfirmMaxSuspects: number;
- // Shrink guard: how far a fresh listing may fall below the stored one before
- // it is treated as suspect rather than acted on. Expressed as a percentage of
- // the previous count, floored at SHRINK_ABS_FLOOR entries so ordinary churn on
- // a small channel doesn't trip it. A suspect listing does not rewrite
- // `playlist` or maybe-missing.json and does not count as a sweep — but a
- // SECOND enumeration reporting a similar count confirms it and is accepted, so
- // a genuine mass deletion costs at most one cadence period. 0 = off (the
- // empty-listing rejection still applies). See controller/acceptListing.ts.
fullSweepShrinkGuardPercent: number;
};
+export const SYNC_SCHEDULER_SETTINGS_FIELD_DOCS: FieldDocs<SyncSchedulerSettings> = {
+ enabled:
+ "Master switch. When false, a tick selects nothing (manual sync still " +
+ "works).",
+ defaultIntervalMinutes:
+ "Fallback cadence (minutes) for channels with no per-channel override.",
+ maxConcurrentSyncs:
+ "Cap on sync jobs running/queued at once. A tick queues at most (cap - " +
+ "currently-active) channels; the rest roll to the next tick. This is " +
+ "also the stagger mechanism that keeps a big due-batch from hitting the" +
+ " source all at once.",
+ quietHoursStart:
+ "Optional local-clock quiet window during which auto-sync is " +
+ "suppressed. Both null = always allowed. The window may wrap past " +
+ "midnight (e.g. start=22, end=6). Hours are [0,23]; the window is " +
+ "[start, end).",
+ quietHoursEnd:
+ "End hour of the quiet window, [0,23], exclusive. See `quietHoursStart`: both must be valid hours or the window is cleared (null = always allowed).",
+ backoffBaseMinutes:
+ "Failure backoff bounds. After N consecutive failed scheduled syncs a " +
+ "channel waits min(base * 2^(N-1), max) minutes before it's eligible " +
+ "again.",
+ backoffMaxMinutes:
+ "Ceiling on the failure backoff (see `backoffBaseMinutes`): a channel waits min(base * 2^(N-1), max) minutes after N consecutive failures. Never below the base.",
+ heartbeatSeconds:
+ "Cadence (seconds) for the editor's in-process heartbeat — the internal" +
+ " timer armed by the instrumentation hook (editor/instrumentation.ts) " +
+ "that calls the scheduler tick directly, so no external cron is needed." +
+ " 0 = off: rely on the external `pnpm sync:tick` heartbeat instead. Any" +
+ " positive value is clamped to [SYNC_HEARTBEAT_MIN_SECONDS, " +
+ "SYNC_HEARTBEAT_MAX_SECONDS]. The env var SYNC_HEARTBEAT_SECONDS " +
+ "overrides this at runtime. See SCHEDULED_SYNC.md.",
+ keepLatestCheckIntervalMinutes:
+ "Cadence (minutes) for the scheduled keep-latest deletion check. For " +
+ "each channel with ChannelConfig.keepLatest > 0, the tick re-probes the" +
+ " kept window for source deletion (checkKeptDeletedAction) at most this" +
+ " often and pins any gone videos as do-not-clean. Clamped into the " +
+ "sync-interval window; default daily. The check shares the same " +
+ "concurrency cap and quiet-hours window as scheduled syncs. See " +
+ "editor/app/scheduler/runTick.ts.",
+ fullSweepIntervalMinutes:
+ "Default cadence (minutes) for the sync FULL SWEEP — the deep pass that" +
+ " re-enumerates a channel's whole listing in one yt-dlp spawn, " +
+ "refreshes the stored `playlist` file, and flags videos that have left " +
+ "the listing into maybe-missing.json. Ordinary syncs stay on the cheap " +
+ "newest-first paged walk; a sync only upgrades itself to a sweep when " +
+ "this interval has elapsed since the channel's lastFullSweepAt. Per-" +
+ "channel override: ChannelConfig.fullSweepIntervalMinutes. 0 = never " +
+ "sweep. Default daily. See common/jobs/deepSync.ts.",
+ fullSweepConfirmMaxSuspects:
+ "Upper bound on how many maybe-missing suspects a full sweep will " +
+ "resolve in-line with the per-video availability probe (deleted vs " +
+ "private vs unlisted). At or under the cap the sweep runs the targeted " +
+ "check itself, so \"Sync all\" surfaces upstream deletions with no extra " +
+ "clicks; over it, the suspects are flagged and left for a manual check " +
+ "rather than firing hundreds of probes inside a sync. 0 = never auto-" +
+ "confirm.",
+ fullSweepShrinkGuardPercent:
+ "Shrink guard: how far a fresh listing may fall below the stored one " +
+ "before it is treated as suspect rather than acted on. Expressed as a " +
+ "percentage of the previous count, floored at SHRINK_ABS_FLOOR entries " +
+ "so ordinary churn on a small channel doesn't trip it. A suspect " +
+ "listing does not rewrite `playlist` or maybe-missing.json and does not" +
+ " count as a sweep — but a SECOND enumeration reporting a similar count" +
+ " confirms it and is accepted, so a genuine mass deletion costs at most" +
+ " one cadence period. 0 = off (the empty-listing rejection still " +
+ "applies). See controller/acceptListing.ts.",
+};
+
+// Each field is documented in SOCIAL_LINK_FIELD_DOCS below (rendered into SETTINGS.md).
export type SocialLink = {
label: string;
url: string;
svg: string;
};
+export const SOCIAL_LINK_FIELD_DOCS: FieldDocs<SocialLink> = {
+ label:
+ "Visible name, also the accessible label of the icon.",
+ url:
+ "Link target: http(s), mailto: or a site-relative path.",
+ svg:
+ "Inline SVG markup. Normalized on save (width/height stripped, " +
+ "fill=\"currentColor\", aria-hidden) and rejected when unsafe (script, " +
+ "foreignObject, event handlers, javascript: URLs) or when it has no " +
+ "viewBox.",
+};
+
+// Each field is documented in ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
+export type ArchiveStorageSettings = {
+ bucket: string;
+ publicBaseUrl: string;
+};
+
+export const ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS: FieldDocs<ArchiveStorageSettings> = {
+ bucket:
+ "Cloudflare R2 bucket an oversize archive zip is uploaded to on deploy " +
+ "(`wrangler r2 object put`, keyed `<siteId>/archives/<file>`). Blank = " +
+ "no overflow.",
+ publicBaseUrl:
+ "Public base URL of that bucket; the Downloads page links " +
+ "`<publicBaseUrl>/<key>`. Both fields must be set for overflow to " +
+ "happen.",
+};
+
export const SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS = 600;
export const SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS = 10;
@@ -1353,7 +1485,7 @@ export const siteSettingsSchema = z.object({
buildArchives: settingsField((v): boolean => v !== false).describe(
"Whether site builds generate downloadable transcript/live-chat archive zips (into public/archives, linked on the Downloads page). Global default; a site can opt out via site.json `archives: false`, and a single build can skip via the \"Skip archive zips\" build control. Opt-out: default true.",
),
- archiveStorage: settingsField((v): { bucket: string; publicBaseUrl: string } => {
+ archiveStorage: settingsField((v): ArchiveStorageSettings => {
const r = (v && typeof v === "object" ? v : {}) as Record<string, unknown>;
return {
bucket: typeof r.bucket === "string" ? r.bucket.trim() : "",
diff --git a/common/lib/storageLocations.ts b/common/lib/storageLocations.ts
@@ -1,4 +1,5 @@
import path from "node:path";
+import type { FieldDocs } from "./fieldDocs";
// STORAGE LOCATIONS — the named places a channel's media may live.
//
@@ -31,57 +32,88 @@ import path from "node:path";
export const INTERNAL_LOCATION_ID = "internal";
export const INTERNAL_LOCATION_LABEL = "Internal (in place)";
+// Each field is documented in STORAGE_VOLUME_FIELD_DOCS below (rendered into SETTINGS.md).
export type StorageVolume = {
- // Filesystem UUID, the one stable name a disk has across mountpoints. This is
- // what makes "the platter came up somewhere else" a recoverable situation.
uuid: string;
fstype?: string;
label?: string;
- // Where the volume was mounted at the last successful probe, and the path of
- // the location's root RELATIVE to that mountpoint. Invariant:
- // `root === join(mountpoint, relPath)`. Keeping the two halves is what lets a
- // probe compute a candidate root when the volume reappears elsewhere.
mountpoint: string;
relPath: string;
};
+export const STORAGE_VOLUME_FIELD_DOCS: FieldDocs<StorageVolume> = {
+ uuid:
+ "Filesystem UUID, the one stable name a disk has across mountpoints. " +
+ "This is what makes \"the platter came up somewhere else\" a recoverable " +
+ "situation.",
+ fstype:
+ "Filesystem type reported by the probe (e.g. \"ext4\"). Informational; omitted when unknown.",
+ label:
+ "Filesystem label reported by the probe. Informational; omitted when unknown.",
+ mountpoint:
+ "Where the volume was mounted at the last successful probe, and the " +
+ "path of the location's root RELATIVE to that mountpoint. Invariant: " +
+ "`root === join(mountpoint, relPath)`. Keeping the two halves is what " +
+ "lets a probe compute a candidate root when the volume reappears " +
+ "elsewhere.",
+ relPath:
+ "The location root's path RELATIVE to `mountpoint` (see there). Invariant: `root === join(mountpoint, relPath)`.",
+};
+
+// Each field is documented in STORAGE_LOCATION_FIELD_DOCS below (rendered into SETTINGS.md).
export type StorageLocation = {
- // /^[a-z0-9][a-z0-9-]{0,63}$/, unique within the list. Stable: it is what
- // `defaultLocationId` and every form and action refer to.
id: string;
- // Human name. Blank sanitizes to the id.
label: string;
- // Absolute directory, trailing "/" stripped. NEVER existence-checked on read
- // — the whole point of a cold location is a drive that may not be mounted
- // when settings are parsed.
root: string;
- // Opt-in: when the volume is found mounted somewhere else, re-point without
- // asking (if the preflight passes). Off by default — re-point rewrites every
- // channel symlink on the location, and that is not something to do silently
- // unless the operator asked for it.
autoRepoint: boolean;
- // Identity learned at the last successful probe. Optional because a location
- // may never have been probed, and because in a container block devices are
- // invisible and identity is permanently unknown.
volume?: StorageVolume;
};
+export const STORAGE_LOCATION_FIELD_DOCS: FieldDocs<StorageLocation> = {
+ id:
+ "/^[a-z0-9][a-z0-9-]{0,63}$/, unique within the list. Stable: it is " +
+ "what `defaultLocationId` and every form and action refer to.",
+ label:
+ "Human name. Blank sanitizes to the id.",
+ root:
+ "Absolute directory, trailing \"/\" stripped. NEVER existence-checked on " +
+ "read — the whole point of a cold location is a drive that may not be " +
+ "mounted when settings are parsed.",
+ autoRepoint:
+ "Opt-in: when the volume is found mounted somewhere else, re-point " +
+ "without asking (if the preflight passes). Off by default — re-point " +
+ "rewrites every channel symlink on the location, and that is not " +
+ "something to do silently unless the operator asked for it.",
+ volume:
+ "Identity learned at the last successful probe. Optional because a " +
+ "location may never have been probed, and because in a container block " +
+ "devices are invisible and identity is permanently unknown.",
+};
+
+// Each field is documented in STORAGE_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
export type StorageSettings = {
locations: StorageLocation[];
- // The location prefilled as the destination of a move. "" = no default.
defaultLocationId: string;
- // WHERE THE SAVED-VIDEO STORE IS, by location id. "" = in place, under the
- // corpus at `paths.savedVideosDir`.
- //
- // A RECORD OF WHAT IS ON DISK, never an intention — the same contract as a
- // channel's `config.dataDir`. It is written by the move, on success, after
- // the copy has verified and the symlink is in place; nothing else writes it,
- // and a reader that disagrees with the disk trusts the disk. Optional so an
- // older settings.json parses (and an older binary that drops it leaves a
- // store that still works, because the symlink is what every reader follows).
savedVideosLocationId?: string;
};
+export const STORAGE_SETTINGS_FIELD_DOCS: FieldDocs<StorageSettings> = {
+ locations:
+ "The named storage locations a channel's media may be relocated to — one entry per root, each with an id, label, root, `autoRepoint` and the learned volume identity. Order is display order. Managed on /storage.",
+ defaultLocationId:
+ "The location prefilled as the destination of a move. \"\" = no default.",
+ savedVideosLocationId:
+ "WHERE THE SAVED-VIDEO STORE IS, by location id. \"\" = in place, under " +
+ "the corpus at `paths.savedVideosDir`.\n\n" +
+ "A RECORD OF WHAT IS ON DISK, never an intention — the same contract as" +
+ " a channel's `config.dataDir`. It is written by the move, on success, " +
+ "after the copy has verified and the symlink is in place; nothing else " +
+ "writes it, and a reader that disagrees with the disk trusts the disk. " +
+ "Optional so an older settings.json parses (and an older binary that " +
+ "drops it leaves a store that still works, because the symlink is what " +
+ "every reader follows).",
+};
+
// Strip trailing slashes so "/mnt/platter/" and "/mnt/platter" are one root.
// The sanitizer does this on write too; this is here so a hand-edited
// settings.json still compares correctly.
diff --git a/common/lib/transcriptionApps.ts b/common/lib/transcriptionApps.ts
@@ -17,29 +17,45 @@ import {
createChoughProgressParser,
createParakeetProgressParser,
} from "../jobs/progressParsers";
+import type { FieldDocs } from "./fieldDocs";
export type TranscriptOutputFormat = "whisper-json" | "chough-json" | "vtt";
// Per-app configuration persisted under settings.transcriptionApps[id]. Every
// field is optional; an app falls back to its own defaults (defaultBin, env).
+// Each field is documented in APP_INSTANCE_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md).
export type AppInstanceConfig = {
- // Binary path/name override. Empty/undefined falls back to app.defaultBin().
bin?: string;
- // whisper.cpp model path (substituted for {model}); for chough this is the
- // optional CHOUGH_MODEL env (chough auto-downloads a model when unset).
model?: string;
- // chough remote server URL (CHOUGH_URL). Empty/undefined = local transcription.
remoteUrl?: string;
- // chough chunk size in seconds (-c). Undefined = chough's own default.
chunkSize?: number;
- // whisper.cpp custom argv template using the {audioFile}/{outputBase}/{model}
- // placeholders. Undefined = DEFAULT_TRANSCRIBE_ARGS.
customArgs?: string[];
- // parakeet compute device passed to parakeet-cli (--device / PARAKEET_DEVICE),
- // e.g. "cuda:0", "cpu". Undefined = parakeet-cli's default device.
device?: string;
};
+export const APP_INSTANCE_CONFIG_FIELD_DOCS: FieldDocs<AppInstanceConfig> = {
+ bin:
+ "Binary path/name override. Empty/undefined falls back to " +
+ "app.defaultBin().",
+ model:
+ "whisper.cpp model path (substituted for {model}); for chough this is " +
+ "the optional CHOUGH_MODEL env (chough auto-downloads a model when " +
+ "unset).",
+ remoteUrl:
+ "chough remote server URL (CHOUGH_URL). Empty/undefined = local " +
+ "transcription.",
+ chunkSize:
+ "chough chunk size in seconds (-c). Undefined = chough's own default.",
+ customArgs:
+ "whisper.cpp custom argv template using the " +
+ "{audioFile}/{outputBase}/{model} placeholders. Undefined = " +
+ "DEFAULT_TRANSCRIBE_ARGS.",
+ device:
+ "parakeet compute device passed to parakeet-cli (--device / " +
+ "PARAKEET_DEVICE), e.g. \"cuda:0\", \"cpu\". Undefined = parakeet-cli's " +
+ "default device.",
+};
+
export type TranscribeBuild = {
// Args passed after the binary.
argv: string[];
diff --git a/common/lib/workers.ts b/common/lib/workers.ts
@@ -19,6 +19,7 @@ import {
getTranscriptionApp,
validateTranscribeArgs,
} from "./transcriptionApps";
+import type { FieldDocs } from "./fieldDocs";
export type WorkerKind = "local" | "remote" | "llm";
@@ -36,32 +37,50 @@ export type WorkerKind = "local" | "remote" | "llm";
// "close enough": its output would be permanently-stale. The dispatcher
// verifies the tag against /api/tags before first use and degrades the worker
// when it is missing (controller/llmWorkers.ts).
+// Each field is documented in LLM_WORKER_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md).
export type LlmWorkerConfig = {
- baseUrl: string; // e.g. http://macbook.lan:11434
- // Concurrent generations to allow this endpoint. Defaults to 1 — one model
- // instance, one generation — unless the operator knows better.
+ baseUrl: string;
slots?: number;
};
+export const LLM_WORKER_CONFIG_FIELD_DOCS: FieldDocs<LlmWorkerConfig> = {
+ baseUrl:
+ "e.g. http://macbook.lan:11434",
+ slots:
+ "Concurrent generations to allow this endpoint. Defaults to 1 — one " +
+ "model instance, one generation — unless the operator knows better.",
+};
+
// Where a remote worker delegates. The remote runs its OWN worker pool and picks
// among ITS local workers — so the primary stores only how to reach it, not which
// engine to use.
+// Each field is documented in REMOTE_WORKER_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md).
export type RemoteWorkerConfig = {
- baseUrl: string; // e.g. http://gpu-box.lan:3011
- // Outbound bearer token sent with every /api/worker request to this remote.
- // The accepting side validates against its own WORKER_TOKEN env, never this.
+ baseUrl: string;
token?: string;
- // When true the remote shares the transcripts mount, so we send
- // {channelSlug, videoId} instead of uploading the audio bytes.
sharedFs?: boolean;
- // How many units this remote takes in parallel. The pool expands one remote
- // config into this many independently-schedulable slot entries at
- // reconfigure time (the defaultWorkersFromApps trick, applied live). Absent =
- // probed from the remote's /api/worker/health (its enabled worker count) —
- // see controller/remoteCapacity.ts; 1 until the probe answers.
slots?: number;
};
+export const REMOTE_WORKER_CONFIG_FIELD_DOCS: FieldDocs<RemoteWorkerConfig> = {
+ baseUrl:
+ "e.g. http://gpu-box.lan:3011",
+ token:
+ "Outbound bearer token sent with every /api/worker request to this " +
+ "remote. The accepting side validates against its own WORKER_TOKEN env," +
+ " never this.",
+ sharedFs:
+ "When true the remote shares the transcripts mount, so we send " +
+ "{channelSlug, videoId} instead of uploading the audio bytes.",
+ slots:
+ "How many units this remote takes in parallel. The pool expands one " +
+ "remote config into this many independently-schedulable slot entries at" +
+ " reconfigure time (the defaultWorkersFromApps trick, applied live). " +
+ "Absent = probed from the remote's /api/worker/health (its enabled " +
+ "worker count) — see controller/remoteCapacity.ts; 1 until the probe " +
+ "answers.",
+};
+
// A worker is ONE processing slot — one transcription at a time. To run N in
// parallel on the same engine, define N workers (the settings editor's "Copy"
// button duplicates one). This makes each slot independently togglable on the
@@ -71,34 +90,53 @@ export type RemoteWorkerConfig = {
// probed capacity): the pool expands it into N slot entries itself, because
// asking the operator to hand-copy a remote once per slot of a machine whose
// slot count the machine already reports would be busywork.
+// Each field is documented in WORKER_FIELD_DOCS below (rendered into SETTINGS.md).
export type Worker = {
- // Stable slug; used in settings, task ids, and logs.
id: string;
- // Human label shown in the UI.
name: string;
kind: WorkerKind;
enabled: boolean;
- // Lower = preferred. Ties broken by array order in the scheduler.
priority: number;
- // Capability routing. A tag is an OPERATION id from the backfill catalog
- // ("diarization", "attribution-text", …) or a contended RESOURCE
- // (WORKER_RESOURCE_TAGS). The scheduler consults them through workerMatches
- // below: an untagged worker takes anything, a tagged worker takes only work
- // whose requirement intersects its tags. Unknown tags are tolerated (they
- // match nothing and warn in the settings UI), never fatal.
tags?: string[];
- // LOCAL: an instance of a TRANSCRIPTION_APPS entry + its per-worker config.
appId?: string;
config?: AppInstanceConfig;
- // REMOTE: how to reach the delegate instance.
remote?: RemoteWorkerConfig;
- // LLM: how to reach the bare model endpoint.
llm?: LlmWorkerConfig;
};
+export const WORKER_FIELD_DOCS: FieldDocs<Worker> = {
+ id:
+ "Stable slug; used in settings, task ids, and logs.",
+ name:
+ "Human label shown in the UI.",
+ kind:
+ "\"local\" runs an app from TRANSCRIPTION_APPS on this machine (`appId` + `config`); \"remote\" delegates to another instance on the LAN (`remote`); \"llm\" is a bare ollama endpoint serving digest/attribution calls only (`llm`).",
+ enabled:
+ "Whether the scheduler may hand this slot work. Each worker is one slot, so parallelism is toggled per slot on the Workers page. Anything but an explicit `false` reads as enabled.",
+ priority:
+ "Lower = preferred. Ties broken by array order in the scheduler.",
+ tags:
+ "Capability routing. A tag is an OPERATION id from the backfill catalog" +
+ " (\"diarization\", \"attribution-text\", …) or a contended RESOURCE " +
+ "(WORKER_RESOURCE_TAGS). The scheduler consults them through " +
+ "workerMatches below: an untagged worker takes anything, a tagged " +
+ "worker takes only work whose requirement intersects its tags. Unknown " +
+ "tags are tolerated (they match nothing and warn in the settings UI), " +
+ "never fatal.",
+ appId:
+ "LOCAL: an instance of a TRANSCRIPTION_APPS entry + its per-worker " +
+ "config.",
+ config:
+ "LOCAL: the per-worker engine config (binary, model, device, …) — an AppInstanceConfig, see `transcriptionApps.<appId>`.",
+ remote:
+ "REMOTE: how to reach the delegate instance.",
+ llm:
+ "LLM: how to reach the bare model endpoint.",
+};
+
// The contended-resource half of the tag vocabulary — Lane.contendsFor's
// three values, restated here because this module must stay client-safe and the
// lane type lives in operations.ts, whose import graph reaches controllers.
diff --git a/plans/one-core-phase-3.md b/plans/one-core-phase-3.md
@@ -529,8 +529,13 @@ over the pre-existing clamp or sanitizer; no `.passthrough()`, no `.default()`.
were on write (one schema). The live file has no untrimmed values.
- `storage/actions.ts`: adding/editing a location used to rebuild the storage block from two
keys and so **erased `storage.savedVideosLocationId`**; the one-level merge keeps it.
-- `/settings` form: patches only its own 19 fields; it no longer resets the rollback-only
- `transcriptionApps` shadow to `{}` (the shadow is still rewritten from workers every save).
+- `/settings` form (`editor/app/settings/actions.ts:207`): it used to pass
+ `transcriptionApp: DEFAULT` + `transcriptionApps: {}` with the stored workers. When the
+ stored list was `[]`, main's worker shadow therefore synthesized a default whisper-cpp
+ worker with NO config; the branch patches only the form's own fields, so the shadow
+ synthesizes from the STORED app and its stored config (better). And nothing now prunes
+ stale per-app entries from the `transcriptionApps` shadow — the old `{}` did — since the
+ shadow is rollback-only and still rewritten from workers on every save (accepted).
**Dead example keys removed**: `transcribeBin`, `transcribeModel`, `transcribeArgs` (the
pre-multi-app spelling, migrated on read).
@@ -541,7 +546,11 @@ through the backfill lane form (lane held, `allowRedownload` off; that save also
retired `backfill.enabled`, as every save does), not a stray write — so the
comparison runs both builds over the SAME frozen inputs: the live file as of 19:24, the
pre-slice example, the e2e fixture, and both `docker/entrypoint.sh` seeds (parakeet,
-whisper). Main `54cf1b31` vs branch tip: **empty diff, 3,846 lines**. The expected
+whisper). Main `54cf1b31` vs branch tip: **empty diff, 3,846 lines**. After review the tool
+also prints what `writeSettings(getSettings())` puts on disk — written to a scratch copy
+under `os.tmpdir()`, never the measured file (the frozen inputs' md5s were re-checked
+after the run). Re-run over the same frozen inputs: **read AND write both diff-empty, 7,691
+lines**, no write threw. The expected
`backfill.enabled` line never appeared: `sanitizeBackfill` already dropped it at main, so it
was not in `getSettings()` output before or after. The regenerated example of course parses
differently from the old one (no legacy `whisper-cli`/`firefox` keys) — by design.
@@ -550,6 +559,23 @@ differently from the old one (no legacy `whisper-cli`/`firefox` keys) — by des
`parallelTranscriptions: 1`) parse to byte-identical settings through main and through the
schema (included in the numbers above). The entrypoint does not read the example.
+**Review fix (one commit after commit 6).** SETTINGS.md now documents every
+NESTED key, not only the 31 top-level ones: each block type carries a
+`<TYPE>_FIELD_DOCS: FieldDocs<Type>` record beside it (`lib/fieldDocs.ts`; the mapped type
+requires one entry per key, optional keys and every union member's keys included, so an
+undocumented new field is a tsc error). The per-field comments moved out of the types into
+those records — 24 records across `settingsSchema.ts` (the seven blocks + `SocialLink` +
+the newly named `ArchiveStorageSettings`), `storageLocations.ts` (settings, location,
+volume), `workers.ts` (worker, remote, llm), `transcriptionApps.ts` (`AppInstanceConfig`),
+`digest.ts` (`DigestAppConfig`), `autoQueueTypes.ts` (policy, tree node, match) and
+`channelPriority.ts` (document, focus, entry, and `autoPaused`, now the named type
+`ChannelAutoPause`). `settingsDocs.ts` renders each as a key · default · description table
+under its block (lane-policy defaults per lane, so `held`'s `[false,false,false,true]` is
+visible). Also: the `settingsField` comment no longer implies zod guards a throwing
+sanitizer (`z.unknown().catch` cannot fire — totality is each coercion's); the docs say
+`workers: []` means no transcription only until the next save; SETTINGS.md warns that a
+copied example pins every default, `held` included.
+
**Gates.** tsc (`pnpm -r --workspace-concurrency=1 exec tsc --noEmit`; the parallel `-r` form
was OOM-killed, exit 137) clean after every commit. common **1625 → 1648** (+20 schema, +3
docs); `test:scripts` 156 pass + 1 skip of 157 (unchanged); mcp 219/219; editor unit **59 → 63**;
diff --git a/plans/tools/phase3-settings-numbers.ts b/plans/tools/phase3-settings-numbers.ts
@@ -10,8 +10,10 @@
// settings.json, the shipped example, the e2e fixture — before and after, and
// diff two files. A test would have to carry the operator's configuration.
//
-// STRICTLY READ-ONLY. It opens each file and writes nothing anywhere. It must
-// never be pointed at a settings.json through a writer.
+// NEVER WRITES A MEASURED FILE. Each target is copied to a scratch directory
+// under os.tmpdir(); the read and the write-back both happen on the copy, which
+// is deleted afterwards. (It measures the WRITE side too since slice 4a's
+// review: what `writeSettings(getSettings())` puts on disk.)
//
// NEVER BOOT AN EDITOR FOR THIS. `getSettings` is called in-process, offline;
// instrumentation.ts is not loaded, so no runner, sweep or scheduler is armed.
@@ -31,6 +33,7 @@
import { spawnSync } from "node:child_process";
import fs from "node:fs";
+import os from "node:os";
import path from "node:path";
import { fileURLToPath } from "node:url";
@@ -78,14 +81,38 @@ function sortedKeys(_key: string, value: unknown): unknown {
return out;
}
+// READ, THEN WRITE — BOTH AGAINST A SCRATCH COPY. The target is copied into a
+// fresh directory under os.tmpdir() and SETTINGS_FILE points at the copy, so
+// `writeSettings` — which writes `getPaths().settingsFile` — can never touch
+// the file being measured (the live settings.json included). The read is of
+// identical bytes; the write side is what a save of that reading puts on disk.
async function child(file: string): Promise<void> {
- process.env.SETTINGS_FILE = file;
- const { getSettings } = await import("../../common/lib/settings");
- console.log(JSON.stringify(getSettings(), sortedKeys, 2));
+ const dir = fs.mkdtempSync(path.join(os.tmpdir(), "phase3-settings-"));
+ const scratch = path.join(dir, "settings.json");
+ fs.copyFileSync(file, scratch);
+ process.env.SETTINGS_FILE = scratch;
+ process.env.TRANSCRIPTS_DIR = dir;
+ try {
+ const { getSettings, writeSettings } = await import(
+ "../../common/lib/settings"
+ );
+ const read = getSettings();
+ console.log(JSON.stringify(read, sortedKeys, 2));
+ console.log("### written by writeSettings(getSettings())");
+ try {
+ await writeSettings(read);
+ const written = JSON.parse(fs.readFileSync(scratch, "utf8"));
+ console.log(JSON.stringify(written, sortedKeys, 2));
+ } catch (e) {
+ console.log(`WRITE THREW: ${(e as Error).message}`);
+ }
+ } finally {
+ fs.rmSync(dir, { recursive: true, force: true });
+ }
}
function parent(targets: Target[]): void {
- console.log("# one-core phase 3 slice 4a — getSettings() over every settings file");
+ console.log("# one-core phase 3 slice 4a — getSettings() and writeSettings() over every settings file");
console.log("");
for (const { label, file } of targets) {
console.log(`## ${label}`);