Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a7501cb3f4b84831663a44b93cc68c68252d05d7
parent 8d0e46617961beba4f1303d90eec7cc021844ad0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Wed, 23 Sep 2026 20:13:09 -0400

merge: one-core/phase-3-s4a — one settings schema (zod), one editor writer, SETTINGS.md generated

Phase 3 slice 4a. common/lib/settingsSchema.ts is the one zod schema for
settings.json (read lenient, write strict, unknown keys stripped, clamps
reused not rewritten); the autoQueue sanitizer lives in lib/autoQueueSchema.ts
(allow-list 10 → 9); editor writes go through saveSettings(patch) from one
file; settings.json.example and SETTINGS.md are generated from the schema and
its per-block field docs, with a test pinning the committed bytes. Gates: tsc,
common 1663, editor unit 67, scripts 156, mcp 219, both builds green with no
zod in client chunks, e2e 157/157; settings numbers diff empty on read and
write over frozen inputs.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
ASETTINGS.md | 723+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
MSETUP.md | 6++++--
Mcommon/architecture.test.ts | 7-------
Acommon/bin/settings-example.ts | 58++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/jobs/autoQueuePolicy.test.ts | 12+++++++++---
Mcommon/jobs/autoQueuePolicy.ts | 261+++++--------------------------------------------------------------------------
Acommon/lib/autoQueueSchema.ts | 284+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/autoQueueTypes.ts | 165+++++++++++++++++++++++++++++++++++++++++++++++++++----------------------------
Mcommon/lib/channelPriority.ts | 121+++++++++++++++++++++++++++++++++++++++++++++++++++++--------------------------
Mcommon/lib/digest.ts | 49++++++++++++++++++++++++++++++++-----------------
Acommon/lib/fieldDocs.ts | 14++++++++++++++
Mcommon/lib/settings.ts | 1791++++++-------------------------------------------------------------------------
Acommon/lib/settingsDocs.test.ts | 63+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/settingsDocs.ts | 305++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/settingsFieldSchemas.ts | 63+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/settingsSchema.test.ts | 425+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/settingsSchema.ts | 1557+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/storageLocations.ts | 90+++++++++++++++++++++++++++++++++++++++++++++++++++++--------------------------
Mcommon/lib/transcriptionApps.ts | 34+++++++++++++++++++++++++---------
Mcommon/lib/workers.ts | 88++++++++++++++++++++++++++++++++++++++++++++++++++++++++-----------------------
Mcommon/package.json | 3++-
Meditor/CHANGELOG.md | 1+
Meditor/app/channels/[slug]/incompleteTranscriptActions.ts | 6+++---
Meditor/app/channels/actions.ts | 11++++++-----
Meditor/app/jobs/actions.ts | 12++++--------
Meditor/app/operations/actions.ts | 14++++++--------
Meditor/app/operations/settingsActions.ts | 47+++++++++++++++++++++++------------------------
Meditor/app/saved-videos/backupActions.ts | 6+++---
Meditor/app/scheduler/actions.ts | 9++++-----
Meditor/app/settings/actions.ts | 58++++++++++++++--------------------------------------------
Aeditor/app/settings/saveSettings.test.ts | 78++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/app/settings/saveSettings.ts | 58++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/sites/lib/buildModeAction.ts | 8+++-----
Meditor/app/storage/actions.ts | 21+++++++++++----------
Meditor/app/workers/actions.ts | 6+++---
Mplans/one-core-phase-3.md | 107+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aplans/tools/phase3-settings-numbers.ts | 167+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mpnpm-lock.yaml | 3+++
Msettings.json.example | 191+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
39 files changed, 4693 insertions(+), 2229 deletions(-)

diff --git a/SETTINGS.md b/SETTINGS.md @@ -0,0 +1,723 @@ +# settings.json keys + +<!-- GENERATED by common/bin/settings-example.ts from common/lib/settingsSchema.ts and the *_FIELD_DOCS records beside each block type — do not edit by hand. --> + +Global operational settings shared by every site this editor powers, persisted to `settings.json` at the repo root (or `$SETTINGS_FILE`). Per-site presentation lives in `sites/<id>/site.json`. Every key is optional: a missing key reads as its default, an ill-typed one is coerced to its default or clamped, and an unknown one is dropped on the next save. + +Regenerate this file and `settings.json.example` with `pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts`. + +`settings.json.example` is the default object with one key left out, `workers`: a file that does not name it gets a worker list synthesized from `transcriptionApp` on read. A file that spells `workers: []` READS as no transcription at all — until the next save, when the writer synthesizes a worker the same way. + +A copied example PINS every default it spells — including each lane's `autoQueue.<lane>.held` — so a default changed in a later release will not reach that file. Delete any key you would rather have track the defaults. + +| Key | Default | +|---|---| +| [`adminTitle`](#admintitle) | `"Transcript Browser Admin"` | +| [`maxTranscriptPageBytes`](#maxtranscriptpagebytes) | `8388608` | +| [`transcriptionApp`](#transcriptionapp) | `"whisper-cpp"` | +| [`transcriptionApps`](#transcriptionapps) | `{}` | +| [`workers`](#workers) | `[]` | +| [`cookiesFromBrowser`](#cookiesfrombrowser) | `""` | +| [`cookieMode`](#cookiemode) | `"when-required"` | +| [`sleepBetweenDownloadsSeconds`](#sleepbetweendownloadsseconds) | `10` | +| [`downloadFormat`](#downloadformat) | `"auto"` | +| [`minFreeDiskGB`](#minfreediskgb) | `5` | +| [`resumeMarginGB`](#resumemargingb) | `2` | +| [`parallelTranscriptions`](#paralleltranscriptions) | `2` | +| [`inlineTranscribeOnFallback`](#inlinetranscribeonfallback) | `false` | +| [`skipLiveDownloads`](#skiplivedownloads) | `true` | +| [`verifyAvailabilityBeforeClean`](#verifyavailabilitybeforeclean) | `true` | +| [`buildArchives`](#buildarchives) | `true` | +| [`archiveStorage`](#archivestorage) | object — see below | +| [`reportDebouncePreset`](#reportdebouncepreset) | `"fast"` | +| [`autoRefreshIntervalSeconds`](#autorefreshintervalseconds) | `5` | +| [`syncScheduler`](#syncscheduler) | object — see below | +| [`autoQueue`](#autoqueue) | object — see below | +| [`channelPriority`](#channelpriority) | object — see below | +| [`socialLinks`](#sociallinks) | `[]` | +| [`homepageUrl`](#homepageurl) | `""` | +| [`savedVideoBackup`](#savedvideobackup) | object — see below | +| [`storage`](#storage) | object — see below | +| [`buildPipeline`](#buildpipeline) | object — see below | +| [`digest`](#digest) | object — see below | +| [`diarization`](#diarization) | object — see below | +| [`backfill`](#backfill) | object — see below | +| [`attribution`](#attribution) | object — see below | + +## `adminTitle` + +Title for the EDITOR admin shell only (the editor manages all sites and so is not tied to any one site's branding). Public sites get their own titles from site.json. + +Default: `"Transcript Browser Admin"` + +## `maxTranscriptPageBytes` + +Target size (bytes) of one exported transcript page shard — the unit the export site fetches. Clamped into [TRANSCRIPT_PAGE_MIN_BYTES, TRANSCRIPT_PAGE_HARD_CAP_BYTES] (256 KiB – 20 MiB); default 8 MiB. + +Default: `8388608` + +## `transcriptionApp` + +Active transcription app id (key into TRANSCRIPTION_APPS, e.g. "whisper-cpp" or "chough"). Selected globally; see common/lib/transcriptionApps.ts. + +Default: `"whisper-cpp"` + +## `transcriptionApps` + +Per-app configuration, keyed by app id. Each app reads only its own block; a missing block means "use the app's defaults". DEPRECATED in favor of `workers` (each local worker carries its own config); kept one release to drive migration and allow rollback. See common/lib/workers.ts. + +#### `transcriptionApps.<appId>` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `bin` | Binary path/name override. Empty/undefined falls back to app.defaultBin(). | +| `model` | whisper.cpp model path (substituted for {model}); for chough this is the optional CHOUGH_MODEL env (chough auto-downloads a model when unset). | +| `remoteUrl` | chough remote server URL (CHOUGH_URL). Empty/undefined = local transcription. | +| `chunkSize` | chough chunk size in seconds (-c). Undefined = chough's own default. | +| `customArgs` | whisper.cpp custom argv template using the {audioFile}/{outputBase}/{model} placeholders. Undefined = DEFAULT_TRANSCRIBE_ARGS. | +| `device` | parakeet compute device passed to parakeet-cli (--device / PARAKEET_DEVICE), e.g. "cuda:0", "cpu". Undefined = parakeet-cli's default device. | + +Default: + +```json +{} +``` + +## `workers` + +Configured transcription workers (named processing slots). The scheduler distributes each video to the highest-priority free worker. A settings.json predating this field is migrated to a single enabled worker from the active app (see defaultWorkersFromApps). See common/lib/workers.ts. + +#### `workers[]` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `id` | Stable slug; used in settings, task ids, and logs. | +| `name` | Human label shown in the UI. | +| `kind` | "local" runs an app from TRANSCRIPTION_APPS on this machine (`appId` + `config`); "remote" delegates to another instance on the LAN (`remote`); "llm" is a bare ollama endpoint serving digest/attribution calls only (`llm`). | +| `enabled` | Whether the scheduler may hand this slot work. Each worker is one slot, so parallelism is toggled per slot on the Workers page. Anything but an explicit `false` reads as enabled. | +| `priority` | Lower = preferred. Ties broken by array order in the scheduler. | +| `tags` | Capability routing. A tag is an OPERATION id from the backfill catalog ("diarization", "attribution-text", …) or a contended RESOURCE (WORKER_RESOURCE_TAGS). The scheduler consults them through workerMatches below: an untagged worker takes anything, a tagged worker takes only work whose requirement intersects its tags. Unknown tags are tolerated (they match nothing and warn in the settings UI), never fatal. | +| `appId` | LOCAL: an instance of a TRANSCRIPTION_APPS entry + its per-worker config. | +| `config` | LOCAL: the per-worker engine config (binary, model, device, …) — an AppInstanceConfig, see `transcriptionApps.<appId>`. | +| `remote` | REMOTE: how to reach the delegate instance. | +| `llm` | LLM: how to reach the bare model endpoint. | + +#### `workers[].config` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `bin` | Binary path/name override. Empty/undefined falls back to app.defaultBin(). | +| `model` | whisper.cpp model path (substituted for {model}); for chough this is the optional CHOUGH_MODEL env (chough auto-downloads a model when unset). | +| `remoteUrl` | chough remote server URL (CHOUGH_URL). Empty/undefined = local transcription. | +| `chunkSize` | chough chunk size in seconds (-c). Undefined = chough's own default. | +| `customArgs` | whisper.cpp custom argv template using the {audioFile}/{outputBase}/{model} placeholders. Undefined = DEFAULT_TRANSCRIBE_ARGS. | +| `device` | parakeet compute device passed to parakeet-cli (--device / PARAKEET_DEVICE), e.g. "cuda:0", "cpu". Undefined = parakeet-cli's default device. | + +#### `workers[].remote` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `baseUrl` | e.g. http://gpu-box.lan:3011 | +| `token` | Outbound bearer token sent with every /api/worker request to this remote. The accepting side validates against its own WORKER_TOKEN env, never this. | +| `sharedFs` | When true the remote shares the transcripts mount, so we send {channelSlug, videoId} instead of uploading the audio bytes. | +| `slots` | How many units this remote takes in parallel. The pool expands one remote config into this many independently-schedulable slot entries at reconfigure time (the defaultWorkersFromApps trick, applied live). Absent = probed from the remote's /api/worker/health (its enabled worker count) — see controller/remoteCapacity.ts; 1 until the probe answers. | + +#### `workers[].llm` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `baseUrl` | e.g. http://macbook.lan:11434 | +| `slots` | Concurrent generations to allow this endpoint. Defaults to 1 — one model instance, one generation — unless the operator knows better. | + +Default: + +```json +[] +``` + +## `cookiesFromBrowser` + +Browser spec (e.g. "firefox", "chrome:Default") passed to `yt-dlp --cookies-from-browser`. WHEN it is passed is governed by `cookieMode` below. Empty string = no cookies configured. Per-channel override available (ChannelConfig.cookiesFromBrowser). + +Default: `""` + +## `cookieMode` + +How yt-dlp invocations use the configured cookies (see common/lib/cookiePolicy.ts): "always" passes them on every invocation, "when-required" (default; the historical behavior) only to retry an auth/age failure, "defer" never in normal runs — auth-gated videos are excluded from batches and collected into the per-channel "Needs cookies" bucket for a manual cookie run. Per-channel override available (ChannelConfig.cookieMode). + +Default: `"when-required"` + +## `sleepBetweenDownloadsSeconds` + +Pause (seconds) inserted between per-video yt-dlp invocations in managed batch downloads. yt-dlp's own `-t sleep` only paces requests within one invocation, so without this the managed loop hammers the source IP back-to-back. 0 disables. Per-channel override available. + +Default: `10` + +## `downloadFormat` + +Default yt-dlp `-f` download format for every channel that doesn't set its own (ChannelConfig.downloadFormat). "auto" picks per-source: `original` for Odysee (whose HLS rungs are CDN-truncated), `bestaudio/worst` elsewhere. See common/ytdlp/downloadFormat.ts. + +Default: `"auto"` + +## `minFreeDiskGB` + +Minimum free disk space (GB) required on the transcripts data directory for downloads to run. When free space is below this floor, a download job is prevented from starting and a running batch stops launching new videos (the in-flight one finishes). 0 disables the gate. See common/lib/diskSpace.ts. + +Default: `5` + +## `resumeMarginGB` + +Extra headroom (GB) above minFreeDiskGB that a stopped pipeline must see before it resumes. Resuming at the same number we stopped at flaps — the first restarted download pushes free space back under the floor. This is the hysteresis margin, so "resumed" means the operator actually freed something rather than a scratch file being cleaned up. 0 disables the hysteresis (resume at the floor). See diskGate() in common/lib/diskSpace.ts. + +Default: `2` + +## `parallelTranscriptions` + +Default number of videos transcribed in parallel when a "Transcribe missing" / bucket run doesn't specify its own concurrency. The per-run Concurrency input in the channel UI overrides this for a single run. + +Default: `2` + +## `inlineTranscribeOnFallback` + +When true, the no-subs fallback in the managed downloader runs whisper inline immediately after the audio download succeeds. When false (default), audio is left for the next "Transcribe missing" pass so a batch download finishes faster and whisper can parallelize. + +Default: `false` + +## `skipLiveDownloads` + +When true (default), managed downloads skip videos that are currently live or scheduled/upcoming, decided from a metadata-only prefetch pass. Finished livestream VODs (was_live) are NOT skipped and download normally. A skip is recorded but not archived, so the next sync/download-missing retries the video once the stream ends. Per-channel override available (ChannelConfig.skipLiveDownloads). + +Default: `true` + +## `verifyAvailabilityBeforeClean` + +Whether the transcribed-audio cleanup sweep checks each candidate is still available upstream before deleting its audio, pinning (do-not-clean) any video found permanently gone. The delete is irreversible and a gone video's audio is irreplaceable, so this defaults to true. Turn it off for an offline or URL-less setup, where the check can never resolve and cleanup would otherwise never delete anything. See verifyBeforeClean.ts. + +Default: `true` + +## `buildArchives` + +Whether site builds generate downloadable transcript/live-chat archive zips (into public/archives, linked on the Downloads page). Global default; a site can opt out via site.json `archives: false`, and a single build can skip via the "Skip archive zips" build control. Opt-out: default true. + +Default: `true` + +## `archiveStorage` + +Overflow object storage (Cloudflare R2) for archive zips that exceed the Pages per-file size cap (see Site.archiveMaxBytes). When both fields are set, an oversize archive is uploaded here on deploy — via `wrangler r2 object put`, keyed `<siteId>/archives/<file>` — instead of being dropped, and the Downloads page links to `<publicBaseUrl>/<key>`. Blank/absent → no overflow, so oversize archives stay unavailable ("Too large to host"). + +#### `archiveStorage` + +| Key | Default | Description | +|---|---|---| +| `bucket` | `""` | Cloudflare R2 bucket an oversize archive zip is uploaded to on deploy (`wrangler r2 object put`, keyed `<siteId>/archives/<file>`). Blank = no overflow. | +| `publicBaseUrl` | `""` | Public base URL of that bucket; the Downloads page links `<publicBaseUrl>/<key>`. Both fields must be set for overflow to happen. | + +Default: + +```json +{ + "bucket": "", + "publicBaseUrl": "" +} +``` + +## `reportDebouncePreset` + +Debounce preset for the global snapshot scheduler: how long it waits after the last report-changing action before regenerating affected channel reports. See REPORT_DEBOUNCE_PRESETS. Default "fast" (~1s, no cap). + +Default: `"fast"` + +## `autoRefreshIntervalSeconds` + +How often (seconds) the editor UI passively re-fetches the current page's server-rendered data via router.refresh(), so sidebar badges and reports stay live without a manual reload. Mounted globally; pauses while the tab is hidden. 0 disables passive refresh entirely. See AUTO_REFRESH_INTERVAL_*. + +Default: `5` + +## `syncScheduler` + +Global configuration for the scheduled (cron-driven) channel sync system. The per-channel cadence lives on ChannelConfig.syncIntervalMinutes; this block holds the defaults and guard rails the scheduler applies across all channels. See common/jobs/syncScheduler.ts. + +#### `syncScheduler` + +| Key | Default | Description | +|---|---|---| +| `enabled` | `false` | Master switch. When false, a tick selects nothing (manual sync still works). | +| `defaultIntervalMinutes` | `1440` | Fallback cadence (minutes) for channels with no per-channel override. | +| `maxConcurrentSyncs` | `2` | Cap on sync jobs running/queued at once. A tick queues at most (cap - currently-active) channels; the rest roll to the next tick. This is also the stagger mechanism that keeps a big due-batch from hitting the source all at once. | +| `quietHoursStart` | `null` | Optional local-clock quiet window during which auto-sync is suppressed. Both null = always allowed. The window may wrap past midnight (e.g. start=22, end=6). Hours are [0,23]; the window is [start, end). | +| `quietHoursEnd` | `null` | End hour of the quiet window, [0,23], exclusive. See `quietHoursStart`: both must be valid hours or the window is cleared (null = always allowed). | +| `backoffBaseMinutes` | `30` | Failure backoff bounds. After N consecutive failed scheduled syncs a channel waits min(base * 2^(N-1), max) minutes before it's eligible again. | +| `backoffMaxMinutes` | `1440` | Ceiling on the failure backoff (see `backoffBaseMinutes`): a channel waits min(base * 2^(N-1), max) minutes after N consecutive failures. Never below the base. | +| `heartbeatSeconds` | `0` | Cadence (seconds) for the editor's in-process heartbeat — the internal timer armed by the instrumentation hook (editor/instrumentation.ts) that calls the scheduler tick directly, so no external cron is needed. 0 = off: rely on the external `pnpm sync:tick` heartbeat instead. Any positive value is clamped to [SYNC_HEARTBEAT_MIN_SECONDS, SYNC_HEARTBEAT_MAX_SECONDS]. The env var SYNC_HEARTBEAT_SECONDS overrides this at runtime. See SCHEDULED_SYNC.md. | +| `keepLatestCheckIntervalMinutes` | `1440` | Cadence (minutes) for the scheduled keep-latest deletion check. For each channel with ChannelConfig.keepLatest > 0, the tick re-probes the kept window for source deletion (checkKeptDeletedAction) at most this often and pins any gone videos as do-not-clean. Clamped into the sync-interval window; default daily. The check shares the same concurrency cap and quiet-hours window as scheduled syncs. See editor/app/scheduler/runTick.ts. | +| `fullSweepIntervalMinutes` | `1440` | Default cadence (minutes) for the sync FULL SWEEP — the deep pass that re-enumerates a channel's whole listing in one yt-dlp spawn, refreshes the stored `playlist` file, and flags videos that have left the listing into maybe-missing.json. Ordinary syncs stay on the cheap newest-first paged walk; a sync only upgrades itself to a sweep when this interval has elapsed since the channel's lastFullSweepAt. Per-channel override: ChannelConfig.fullSweepIntervalMinutes. 0 = never sweep. Default daily. See common/jobs/deepSync.ts. | +| `fullSweepConfirmMaxSuspects` | `25` | Upper bound on how many maybe-missing suspects a full sweep will resolve in-line with the per-video availability probe (deleted vs private vs unlisted). At or under the cap the sweep runs the targeted check itself, so "Sync all" surfaces upstream deletions with no extra clicks; over it, the suspects are flagged and left for a manual check rather than firing hundreds of probes inside a sync. 0 = never auto-confirm. | +| `fullSweepShrinkGuardPercent` | `10` | Shrink guard: how far a fresh listing may fall below the stored one before it is treated as suspect rather than acted on. Expressed as a percentage of the previous count, floored at SHRINK_ABS_FLOOR entries so ordinary churn on a small channel doesn't trip it. A suspect listing does not rewrite `playlist` or maybe-missing.json and does not count as a sweep — but a SECOND enumeration reporting a similar count confirms it and is accepted, so a genuine mass deletion costs at most one cadence period. 0 = off (the empty-listing rejection still applies). See controller/acceptListing.ts. | + +Default: + +```json +{ + "enabled": false, + "defaultIntervalMinutes": 1440, + "maxConcurrentSyncs": 2, + "quietHoursStart": null, + "quietHoursEnd": null, + "backoffBaseMinutes": 30, + "backoffMaxMinutes": 1440, + "heartbeatSeconds": 0, + "keepLatestCheckIntervalMinutes": 1440, + "fullSweepIntervalMinutes": 1440, + "fullSweepConfirmMaxSuspects": 25, + "fullSweepShrinkGuardPercent": 10 +} +``` + +## `autoQueue` + +Configuration for the automatic priority-queue runners (auto-transcribe / auto-download). Each holds a tree policy that decides which channel's video to process next, cross-channel, by priority/round-robin/weighted-fair rules. Independent of syncScheduler (which decides staleness, not work order). See common/jobs/autoQueuePolicy.ts. + +#### `autoQueue.<lane>` + +| Key | Default | Description | +|---|---|---| +| `enabled` | `false` | Master switch for this runner (transcription / download independently). | +| `maxWorkers` | `null` | Overall ceiling on concurrent in-flight workers for this runner. null = no runner-level cap (the worker pool / platform queues are the real throttle). | +| `replaceAutoSubs` | `false` | Opt in to the lowest-priority "replace YouTube auto-captions" lane: append this kind's opt-in buckets (autoSubsOnly / downloadedAutoSubsOnly) to the tail of the default union, so videos whose only transcript is YouTube ASR get re-done with our own engine whenever nothing more important is pending. Default false — the corpus-wide cost is large (an audio download plus a transcription per video). A leaf can also target the bucket by name for per-channel opt-in without flipping this switch. Optional: settings written before this field existed lack it; the sanitizer defaults it to false. | +| `order` | transcription `"listed"`<br>download `"listed"`<br>digest `"cheapest"`<br>backfill `"listed"` | Ordering within each rule (see AutoQueueOrder). Optional exactly like replaceAutoSubs: settings files written before this field existed lack it, and the sanitizer defaults them to "listed" (today's behaviour). | +| `snoozeUntil` | `null` | Epoch ms until which this runner idles WITHOUT stopping: next() returns null so the loop stays up, re-reads settings each iteration, and resumes by itself when the moment passes. null/absent/past = not snoozed. Survives a restart because it lives in settings.json, not in runner memory. | +| `held` | transcription `false`<br>download `false`<br>digest `false`<br>backfill `true` | THE LANE'S PAUSE GATE. Shut means the lane holds: every dispatch path asks lib/pauseGates.ts, whose limit()/guard returns 0 so runPool idle-waits. A hold, never a stop — see that file's header.<br><br>OPTIONAL IN THE TYPE, FILLED BY THE SANITIZER. Until slice 1.4 four separate settings fields carried this — `transcriptionsPaused`, `downloadsPaused`, `digest.digestsPaused` and (inverted) `backfill.enabled` — so `undefined` meant "ask the legacy field" and `sanitizePolicy` deliberately refused to default it: a default would have read a paused corpus as running. S0-pause deleted those four, on the precondition that the live settings.json already carried every `held` key, and the default came in with them (`defaultHeldFor` — free everywhere except backfill, whose field was inverted and shipped held).<br><br>It stays optional because a reader may be handed a PARTIAL settings object (laneGuards.test.ts casts one), and `isGateHeld` answers `false` for a lane that carries no key at all rather than throwing. | +| `root` | object — see below | The lane's rule tree: a group whose children are groups and leaves (see the node table). A missing root is the lane's default — empty for the runner lanes, one catch-all leaf for digest and backfill. While a channel-priority document exists, the four roots are compiled from it and not hand-edited. | + +#### `autoQueue.<lane>.root (tree nodes)` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `id` | Stable node id, unique within the lane's tree. Preserved on save when valid and not taken, so persisted fairness state survives an unrelated edit; a missing or duplicate id is replaced with a generated one. | +| `match` | LEAF ONLY: which videos this leaf owns (see the match table). | +| `weight` | Relative share under a weighted-fair parent. Default 1. Ignored otherwise. | +| `maxWorkers` | Optional ceiling on concurrent in-flight workers drawn from this node (and, for a group, its whole subtree). A capped node reads as "no work" and the parent falls through to the next sibling, like an HTB class ceiling. null = no cap. | +| `mode` | GROUP ONLY: how the children compete — "strict" (first child with work wins), "round-robin", or "weighted-fair" (by each child's `weight`). Unknown values read as "strict". | +| `children` | GROUP ONLY: the child nodes, in priority order for a strict group. A node with a `children` array is a group; any other node is a leaf. | + +#### `autoQueue.<lane>.root … .match` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `type` | What the leaf matches: "channel" (one channel slug in `value`), "platform" (a platform name in `value`), or "all". | +| `value` | Channel slug (type=channel) or platform name (type=platform). Ignored for type=all. A type=channel leaf with no value matches nothing. | +| `bucket` | Optional snapshot bucket this leaf draws from, narrowing the default for the runner kind (transcription → downloadedNoTranscript, download → undownloadedIds). E.g. bucket="failedListed" prioritizes retries. | +| `operation` | Optional OPERATION this leaf draws from — a registered backfill kind id, or "digest". Same meaning as `bucket` one level up: it narrows what the leaf claims, and it draws from ChannelWork.operations rather than ChannelWork.buckets.<br><br>It lives on the MATCH, beside `bucket`, and not on the node. A field on the node would need group inheritance — "this group is the digest subtree" — and inheritance is resolution logic buildPendingByLeaf does not have. Here it needs exactly one sanitizer and exactly one claiming path.<br><br>It is a SEPARATE id space from `bucket`, and the sanitizer enforces that a leaf names at most one of the two (operation wins): `defaultBuckets` is a priority-ordered union, so a name that meant a bucket to one leaf and an operation to another would silently mix two id spaces, and selectableBucketsForKind feeds the editor's bucket dropdown, where an operation must not appear as a bucket. | + +Default: + +```json +{ + "transcription": { + "enabled": false, + "maxWorkers": null, + "replaceAutoSubs": false, + "order": "listed", + "snoozeUntil": null, + "held": false, + "root": { + "id": "root", + "mode": "strict", + "weight": 1, + "maxWorkers": null, + "children": [] + } + }, + "download": { + "enabled": false, + "maxWorkers": null, + "replaceAutoSubs": false, + "order": "listed", + "snoozeUntil": null, + "held": false, + "root": { + "id": "root", + "mode": "strict", + "weight": 1, + "maxWorkers": null, + "children": [] + } + }, + "digest": { + "enabled": false, + "maxWorkers": null, + "replaceAutoSubs": false, + "order": "cheapest", + "snoozeUntil": null, + "held": false, + "root": { + "id": "root", + "mode": "strict", + "weight": 1, + "maxWorkers": null, + "children": [ + { + "id": "all", + "match": { + "type": "all" + }, + "weight": 1, + "maxWorkers": null + } + ] + } + }, + "backfill": { + "enabled": false, + "maxWorkers": null, + "replaceAutoSubs": false, + "order": "listed", + "snoozeUntil": null, + "held": true, + "root": { + "id": "root", + "mode": "strict", + "weight": 1, + "maxWorkers": null, + "children": [ + { + "id": "all", + "match": { + "type": "all" + }, + "weight": 1, + "maxWorkers": null + } + ] + } + } +} +``` + +## `channelPriority` + +THE OPERATOR-FACING PRIORITY MODEL: one tier per channel plus one corpus-wide focus selector. It is the SOURCE the four `autoQueue[lane].root` trees are compiled from (common/lib/channelPriority.ts), not a second mechanism beside them — and its `paused` tier is the one part that is not a tree shape, filtering the runner's channel list instead. An empty document (the default) is today's behaviour exactly: no focus, every channel normal, the stored trees stand. + +#### `channelPriority` + +| Key | Default | Description | +|---|---|---| +| `focus` | object — see below | The corpus-wide focus selector: none, one site's channels, or a list of channels. A focus is compiled into a leading `prio-focus` group in every lane's tree. | +| `channels` | `{}` | ONLY channels that differ from the default appear. An absent slug is `normal`, unranked — so the default document is empty and "absent document = today's behaviour" holds byte for byte. | + +#### `channelPriority.focus` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `kind` | "none" (no focus), "site" (the channels of one site, resolved at compile time so it tracks membership) or "channels" (an explicit list, from "Focus these"). | +| `siteId` | kind "site" only: the site whose channels are focused. A blank id reads as no focus; an unknown one survives and focuses nothing. | +| `slugs` | kind "channels" only: the focused channel slugs, trimmed and de-duplicated. An empty list reads as no focus. | + +#### `channelPriority.channels.<slug>` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `tier` | THE BASE TIER: what every operation gets unless an override says otherwise. | +| `rank` | Order WITHIN the tier, ascending. Absent = unranked, which sorts after every ranked sibling and then by slug. ONE rank per channel, not one per lane — the two hand-made lane orders collapse into this on migration. | +| `overrides` | PER-OPERATION OVERRIDES of the base tier. Only operations that DIFFER from the base appear: the sanitizer normalises an override equal to `tier` away, so the on-disk document stays a list of exceptions to a list of exceptions.<br><br>`{tier:"normal", overrides:{sync:"paused"}}` is "everything but sync" — the lossless reading of the retired `excludeFromSync`. Its inverse, `{tier:"paused", overrides:{sync:"normal"}}`, is "sync only": keep the playlist and metadata current, dispatch nothing. | +| `autoPaused` | PAUSED BY THE MACHINE, NOT BY THE OPERATOR, and what to put back.<br><br>Set when the drive a channel's media is on stops being there: the watch pass records the tier the channel HAD and forces `paused`, so nothing in any lane dispatches against a `data/` nobody can read. Cleared — and the tier restored — when the drive comes back.<br><br>WHY IT IS A FIELD AND NOT A DERIVED STATE. The lanes read `tier`; making them all ask a second question would be four more places to forget. And the tier the channel is to be RESTORED to is not derivable from anything once it has been overwritten — that is the whole content of this field.<br><br>OPTIONAL, and an older binary that drops it leaves the channel Paused with nothing lost but the automatic restore. The operator's own word always wins: a MANUAL tier change clears it (see clearAutoPause), so a drive coming back can never un-pause a channel somebody paused on purpose. | + +#### `channelPriority.channels.<slug>.autoPaused` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `reason` | One reason today. A union so a second one has somewhere to go, and so a surface can say WHICH machine decided rather than "automatic". | +| `since` | ISO, for "auto-paused — media unreachable since <date>". | +| `previousTier` | The base tier the channel had before the machine paused it; what a restore puts back. Never `paused` (that would restore to paused — a no-op dressed as a restore). | + +Default: + +```json +{ + "focus": { + "kind": "none" + }, + "channels": {} +} +``` + +## `socialLinks` + +Default social links applied to every site that doesn't define its own. A site inherits these unless its site.json carries an explicit `socialLinks` array — see Site.socialLinks / resolveSocialLinks in common/lib/site.ts. The one presentation field that lives globally so a shared footer doesn't have to be repeated per site. + +#### `socialLinks[]` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `label` | Visible name, also the accessible label of the icon. | +| `url` | Link target: http(s), mailto: or a site-relative path. | +| `svg` | Inline SVG markup. Normalized on save (width/height stripped, fill="currentColor", aria-hidden) and rejected when unsafe (script, foreignObject, event handlers, javascript: URLs) or when it has no viewBox. | + +Default: + +```json +[] +``` + +## `homepageUrl` + +Absolute public URL of the family hub/homepage (e.g. "https://archilyzer.pages.dev"). Every export site links back to it ("the family" backlink) when set. Empty = no hub link rendered. Normalized to a trailing-slash-free http(s) URL. + +Default: `""` + +## `savedVideoBackup` + +Backup configuration for the saved-video store (Phase 4 of the video-persistence feature). When enabled with a destination, the store is mirrored there (additively, no deletes) with a per-backup manifest, and the sync scheduler runs the backup on the configured cadence. See common/controller/backupSavedVideos.ts. + +#### `savedVideoBackup` + +| Key | Default | Description | +|---|---|---| +| `enabled` | `false` | Master switch for the scheduled backup. A backup can still be run manually when this is false, as long as a destination is set. | +| `dest` | `""` | Destination root the store is mirrored into (a local path or any rsync target). Empty disables both scheduled and manual backups. | +| `intervalMinutes` | `1440` | Cadence (minutes) for the scheduled backup when enabled. Clamped into the sync-interval window; default daily. | + +Default: + +```json +{ + "enabled": false, + "dest": "", + "intervalMinutes": 1440 +} +``` + +## `storage` + +Where a channel's downloaded media goes when it is relocated off the corpus disk. A DEFAULT ONLY: the relocate controller never reads it and always takes an explicit root, so this is the value the per-channel Storage panel prefills and the /channels bulk move falls back to. Blank = no default. See StorageSettings. + +#### `storage` + +| Key | Default | Description | +|---|---|---| +| `locations` | `[]` | The named storage locations a channel's media may be relocated to — one entry per root, each with an id, label, root, `autoRepoint` and the learned volume identity. Order is display order. Managed on /storage. | +| `defaultLocationId` | `""` | The location prefilled as the destination of a move. "" = no default. | +| `savedVideosLocationId` | absent | WHERE THE SAVED-VIDEO STORE IS, by location id. "" = in place, under the corpus at `paths.savedVideosDir`.<br><br>A RECORD OF WHAT IS ON DISK, never an intention — the same contract as a channel's `config.dataDir`. It is written by the move, on success, after the copy has verified and the symlink is in place; nothing else writes it, and a reader that disagrees with the disk trusts the disk. Optional so an older settings.json parses (and an older binary that drops it leaves a store that still works, because the symlink is what every reader follows). | + +#### `storage.locations[]` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `id` | /^[a-z0-9][a-z0-9-]{0,63}$/, unique within the list. Stable: it is what `defaultLocationId` and every form and action refer to. | +| `label` | Human name. Blank sanitizes to the id. | +| `root` | Absolute directory, trailing "/" stripped. NEVER existence-checked on read — the whole point of a cold location is a drive that may not be mounted when settings are parsed. | +| `autoRepoint` | Opt-in: when the volume is found mounted somewhere else, re-point without asking (if the preflight passes). Off by default — re-point rewrites every channel symlink on the location, and that is not something to do silently unless the operator asked for it. | +| `volume` | Identity learned at the last successful probe. Optional because a location may never have been probed, and because in a container block devices are invisible and identity is permanently unknown. | + +#### `storage.locations[].volume` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `uuid` | Filesystem UUID, the one stable name a disk has across mountpoints. This is what makes "the platter came up somewhere else" a recoverable situation. | +| `fstype` | Filesystem type reported by the probe (e.g. "ext4"). Informational; omitted when unknown. | +| `label` | Filesystem label reported by the probe. Informational; omitted when unknown. | +| `mountpoint` | Where the volume was mounted at the last successful probe, and the path of the location's root RELATIVE to that mountpoint. Invariant: `root === join(mountpoint, relPath)`. Keeping the two halves is what lets a probe compute a candidate root when the volume reappears elsewhere. | +| `relPath` | The location root's path RELATIVE to `mountpoint` (see there). Invariant: `root === join(mountpoint, relPath)`. | + +Default: + +```json +{ + "locations": [], + "defaultLocationId": "" +} +``` + +## `buildPipeline` + +How the static export is built: "basic" reuses the single export/ tree and serializes builds on one queue (the long-standing behavior); "docker" runs each site's build in an isolated container for safe parallelism. The Docker pipeline itself is a follow-up; this block persists the chosen mode plus the container/concurrency knobs the deploy page and the future orchestrator read. + +#### `buildPipeline` + +| Key | Default | Description | +|---|---|---| +| `mode` | `"basic"` | "basic" — `pnpm run build` in export/, serialized on the build queue (shared output tree, no safe parallelism). "docker" — isolated per-site container builds, parallel up to `maxParallelBuilds`. | +| `maxParallelBuilds` | `2` | Cap on concurrent per-site container builds in docker mode. Ignored in basic mode (which is always serial). Clamped to [1, BUILD_MAX_PARALLEL_MAX]. | +| `dockerImage` | `"yt-dlp-transcript-browser-build"` | Tag of the reusable build image (built once, reused for every site). | +| `dockerfile` | `"Dockerfile.build"` | Dockerfile path relative to the monorepo root, used to (re)build the image. | + +Default: + +```json +{ + "mode": "basic", + "maxParallelBuilds": 2, + "dockerImage": "yt-dlp-transcript-browser-build", + "dockerfile": "Dockerfile.build" +} +``` + +## `digest` + +AI digest generation (chapters + topic tags over the existing transcripts). Local-first: the metered lane is off by default. See DigestSettings. + +#### `digest` + +| Key | Default | Description | +|---|---|---| +| `remoteEnabled` | `false` | Master switch for the metered (remote-api) lane. OFF by default — an opt-in overflow for the long tail or a channel where local quality is poor, never the default path. | +| `longTailSeconds` | `14400` | Videos longer than this are "long tail": 8.2% of the corpus by count, 46% of all transcript tokens. The batch's duration-aware ordering and the optional remote overflow both key off it. | +| `localAppId` | `"ollama-direct"` | The engine each lane uses (ids from common/lib/digestApps.ts). | +| `remoteAppId` | `"claude-code"` | The engine the metered (remote-api) lane uses — an id from common/lib/digestApps.ts. Unknown ids degrade to the default app rather than failing. | +| `apps` | `{}` | Per-app config, keyed by app id — the same id-keyed sub-record shape as transcriptionApps. | +| `yieldToTranscription` | `true` | Yield the GPU to the transcription lane: while transcription is working, the digest batch's limit() returns 0 and the pool idle-waits. ON by default, because `digest:local` is deliberately on a different queue from TRANSCRIPTION_QUEUE and so would otherwise run ollama and the transcription engine on the same 8 GB card. See controller/digestYield.ts. | +| `yieldToCpuWorkers` | `false` | Whether a busy worker pinned to `device: "cpu"` counts as GPU contention.<br><br>OFF by default, which is the FIX for a real bug: the yield originally tested only `kind === "local"`, so on a box with one GPU worker and two CPU-pinned ones (this box, at parallelTranscriptions 2) the digest lane stopped dead for transcription that competes for zero GPU shaders.<br><br>Only an EXPLICIT "cpu" is treated as non-contending. A worker with no device set is using the engine binary's own default, which may be the GPU, so it still triggers the yield — the unknown case fails safe.<br><br>Composes with `yieldToTranscription`: that is the master switch, this only narrows which workers it reacts to. | +| `spendCapUsd` | `0` | Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. Only ever consulted for a metered app. | +| `sections` | list — see below | Which sections a sweep generates.<br><br>Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, and that is not a contradiction: a tag call sends the same transcript as the chapter call before it, so it hits the engine's cached prefix and pays essentially no prefill (+0.4 s across 4 extra calls, against 22.4 s for the first 4). All it pays is decode, and a tag list is ~30 output tokens where a chapter list is ~200–290.<br><br>The corollary matters more than the number: run them in the SAME pass. Tags generated later, on their own, pay full prefill again — measured at 44% of a whole chapters pass, i.e. 3–9× the marginal cost of just including them now. | +| `timestampMode` | `"chunk-local"` | How each chunk's transcript markers are numbered — see DigestTimestampMode. Was a scored variable in the bake-off rather than a pre-applied fix; the measurement is in and "chunk-local" is now the shipped default. | +| `promptVariant` | `""` | A free-text label for a non-default prompt shape, folded into the recorded provenance by digestPromptVariant(). Setting it invalidates every digest generated under a different label, which is exactly what makes a bake-off round re-run its sample instead of skipping it as fresh. Empty = default. | + +#### `digest.apps.<appId>` + +Per entry — each entry spells its own values. + +| Key | Description | +|---|---| +| `bin` | Binary path/name override (process-based apps only). | +| `baseUrl` | Base URL override (HTTP apps only). | +| `model` | Model id, e.g. "qwen2.5:7b" or "haiku". | +| `numCtx` | Context window in tokens. MUST reach the engine explicitly for ollama: its 4096 default silently truncates the input and the model then summarizes whatever fragment survived — measured, and the single easiest way to get quietly-wrong output at scale. | +| `temperature` | Sampling temperature. 0 for a structured extraction task. | +| `think` | Reasoning-model toggle (ollama's top-level `think`). Only sent when set, so a model that does not support thinking is never handed a field it rejects.<br><br>It matters for throughput, not correctness: measured on this box, qwen3:8b with thinking on spends most of its output budget on a `thinking` block before the JSON body the schema constrains. For an extraction task with a pinned schema that reasoning buys little and costs a multiple of the tokens, and tokens are what a multi-week sweep is priced in. | +| `timeoutMs` | Per-request wall-clock ceiling (ms). A wedged engine must not stall a sweep. | + +Default: + +```json +{ + "remoteEnabled": false, + "longTailSeconds": 14400, + "localAppId": "ollama-direct", + "remoteAppId": "claude-code", + "apps": {}, + "yieldToTranscription": true, + "yieldToCpuWorkers": false, + "spendCapUsd": 0, + "sections": [ + "chapters" + ], + "timestampMode": "chunk-local", + "promptVariant": "" +} +``` + +## `diarization` + +Speaker diarization captured right after transcription, while the audio is still on disk. OFF by default. See DiarizationSettings. + +#### `diarization` + +| Key | Default | Description | +|---|---|---| +| `enabled` | `false` | Master switch. OFF by default so a transcription batch can start before this lands, with diarization backfilled over the retained audio afterwards.<br><br>Turning it ON also arms the cleanup guard: the Clean-audio sweep stops deleting audio for a transcribed video that has no diarization.json yet. That is the point — it is what keeps the perishable input alive long enough to be captured — but it means enabling this holds disk. | +| `inlineAfterTranscribe` | `false` | Run diarization inline in the post-transcribe hook.<br><br>OFF by default, and that default is a MEASURED decision, not caution. Measured on this box: GPU transcription runs at 221 s/audio-hour (16.3x realtime, over 3,602 real videos), CPU diarization at ~500-680 s/audio-hour. Diarization is therefore ~2-3x SLOWER than the transcription it follows, so running it inline drops whole-pipeline throughput by roughly 3-4x and leaves the GPU idle while the CPU catches up.<br><br>The intended sequence for a large batch is the opposite: leave this off, let the batch transcribe at full GPU speed with `enabled` holding the audio, and diarize afterwards with the backfill pass. Turn it on for steady state, once the arrival rate is a few videos a day rather than a corpus. | +| `threshold` | `0.9` | Clustering threshold — the single most consequential knob, since it decides how many speakers come out. Larger merges more aggressively.<br><br>The default is 0.9, NOT sherpa-onnx's own 0.5, and that is measured on this corpus. On a 6-minute excerpt of a two-person interview (known ground truth: 2 speakers), sherpa's default produced 22 clusters; 0.9 produced 6, with the top two at 40%/40% of talk time — recognizably the two hosts. Sweep on the same clip: 0.4→23, 0.5→22, 0.6→17, 0.7→12, 0.8→10, 0.9→6.<br><br>It still over-splits, which is why this is a capture lane and not an answer: the turns are recorded with the threshold that produced them, so a later attribution pass can re-cluster or re-run without needing the audio back. | +| `threads` | `4` | Engine threads per diarize run. | +| `engine` | `"sherpa-onnx"` | Which engine runs. "sherpa-onnx" is the shipped default and what every sidecar on disk was produced by; "sortformer" is the ggml engine built by scripts/build-sortformer.sh.<br><br>CHANGING THIS RESTATES THE FRESHNESS IDENTITY (see diarizationTarget), so every sidecar written by the other engine becomes stale and the backfill lane offers to redo it. That is intended — the two disagree about how many speakers exist, and a corpus half-diarized by each is not one corpus — but on the retained audio it is weeks of work, not a toggle.<br><br>Why anyone would: on the same file, sherpa at its tuned threshold returns 13 speakers and sortformer returns 4, agreeing on the dominant speaker's share to within half a point (73.1% vs 73.5%). On the corpus's worst case sherpa returns 35 and sortformer 4. Over-splitting is the failure mode this lane has always had, and sortformer is end-to-end rather than clustered, so it does not have it. The cost is a hard ceiling of 4 speakers and ~1.8x the wall clock. | +| `backend` | `"vulkan"` | Compute device for the sortformer engine; ignored by sherpa-onnx, which has no Vulkan compute path on Linux.<br><br>"vulkan" is 1.5x faster than a thread-tuned CPU run (894 vs 1305 s/audio-hour, measured on this box) and holds 558 MB resident instead of 4.84 GB by keeping weights and activations in VRAM. It also takes ~4.4 GB of an 8 GB card, which is why the lane YIELDS to transcription rather than sharing — see controller/digestYield.ts. | +| `python` | `"python3"` | Python interpreter for the default sherpa-onnx engine. sherpa-onnx ships wheels only up to cp313, and this box's system python is 3.14 — so this usually points at a dedicated venv rather than `python3`. | +| `segModel` | `""` | ONNX model paths for the default engine. Empty = the lane cannot run, which is reported as a skip rather than a failure. | +| `embModel` | `""` | ONNX speaker-embedding model path for the sherpa-onnx engine. Empty = the lane cannot run, reported as a `not-configured` skip rather than a failure (same as `segModel`). | +| `sortformerBin` | `""` | Binary and model for the sortformer engine, both produced by scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the same "not-configured" skip as an unset segModel/embModel. | +| `sortformerModel` | `""` | Model for the sortformer engine, produced by scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the same `not-configured` skip as an unset `sortformerBin`. | +| `concurrency` | `1` | How many diarize runs may execute at once in the backfill pass. Kept low by default: diarization is CPU-bound and competes with GPU feeding and the digest sweep for the same 8 threads. | +| `maxAudioHours` | `0` | Videos longer than this are DEFERRED rather than diarized: reported as a third number that is never summed into reachable work, so a capped corpus can never read as finished.<br><br>THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering holds a pairwise distance matrix over speech-segment embeddings — O(n^2) in SEGMENT count — and speaker-turn density varies 40x across this corpus (33-1364 turns/hour), so duration does not actually predict the blowup: a sparse 7h42m video completed while a dense 6h12m one was OOM-killed. Duration is merely the only predictor available for free, from metadata already on disk, BEFORE spending 45 minutes to find out. n^2 at 30k segments is 6.7 GiB and at 40k is 11.9 GiB, which brackets the 10.6 GB and 9.6 GB peaks measured on this 16 GB box.<br><br>0 disables the cap. That is where this goes once windowed diarization lands: windowing divides per-window n by the window count, so the matrix falls by its square, and the cap stops being needed rather than being tuned. | + +Default: + +```json +{ + "enabled": false, + "inlineAfterTranscribe": false, + "threshold": 0.9, + "threads": 4, + "engine": "sherpa-onnx", + "backend": "vulkan", + "python": "python3", + "segModel": "", + "embModel": "", + "sortformerBin": "", + "sortformerModel": "", + "concurrency": 1, + "maxAudioHours": 0 +} +``` + +## `backfill` + +The generic catch-up lane for derived data the existing corpus predates. OFF by default, and idle-only when on. See BackfillSettings. + +#### `backfill` + +| Key | Default | Description | +|---|---|---| +| `concurrency` | `1` | Slots the lane may use when it is not standing aside. Kept at 1 by default for the same reason diarization.concurrency is: this is CPU-bound work competing with GPU feeding and the digest sweep for the same 8 threads. | +| `allowRedownload` | `false` | Re-acquire media for videos whose input is GONE (audio deleted after transcription). OFF by default and deliberately so: measured on this corpus, 836 videos still have media and ~76,270 would need a re-download — 91x the reachable work, against 45 GB free at 97% full. When on, each re-fetched file is removed in a `finally` as soon as the backfill has used it, unless the video is marked do-not-clean, or unless the auto-transcribe policy would replace its auto-captions (`replaceAutoSubs`, or a leaf on `downloadedAutoSubsOnly`), in which case the audio is kept for that runner.<br><br>WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: "youtube"` channel — which normally only fetches subtitles — the re-acquire applies a PER-VIDEO transcribe override so yt-dlp lands audio a diarizer can read; the channel's stored config is not changed. Without that override the fetch re-downloads the captions the video already has and lands nothing, which is what happened to ~16,000 videos on eight channels in 2026-08. | + +Default: + +```json +{ + "concurrency": 1, + "allowRedownload": false +} +``` + +## `attribution` + +Naming the speakers diarization found (or reconstructing them from the transcript when it found none). OFF by default. See AttributionSettings. + +#### `attribution` + +| Key | Default | Description | +|---|---|---| +| `enabled` | `false` | Master switch. Off means the backfill registry reports no attribution work at all — the feature gate every Operation has. | +| `appId` | `"ollama-direct"` | Which digest app runs the naming. Attribution IS a digest-app workload — constrained JSON decoding over transcript text — so it reuses that registry and that per-app config (settings.digest.apps[appId]) rather than growing a second copy of the ollama URL, context size and timeout. | +| `model` | `""` | Model override. Empty = the app's configured model, then its default. It is separate from the digest's because the two workloads may want different sizes, and because it is part of the freshness identity: sharing the digest's model field would make a digest bake-off invalidate every attribution record on disk as a side effect. | +| `diarizedEnabled` | `false` | The lanes, separately. Both default OFF even when `enabled` is on, so turning the feature on to look at it cannot start a corpus sweep.<br><br>They are not a fallback pair. `diarized` is one call per video and grounded in acoustic clustering; `textOnly` is ~30 calls and guesses at identity across chunk seams. An operator may reasonably want the first forever and the second never. | +| `textOnlyEnabled` | `false` | The text-only attribution lane: names speakers from the transcript alone (~30 model calls per video, guessing identity across chunk seams). Default OFF even when `enabled` is on. See `diarizedEnabled` — the two are separate lanes, not a fallback pair. | +| `promptVersion` | `1` | The prompt generation a record must match to count as fresh.<br><br>Defaults to (and is floored at) ATTRIBUTION_PROMPT_VERSION, the shipped constant. Raising it forces a corpus-wide regeneration without a code change, which is the honest way to redo everything after a prompt tweak. It cannot be set BELOW the shipped constant, and that floor is the lesson from digestPrompt.ts's version 1 -> 2 note: pinning freshness to an older generation freezes output from a superseded prompt into the corpus, looking identical to output from the current one. | + +Default: + +```json +{ + "enabled": false, + "appId": "ollama-direct", + "model": "", + "diarizedEnabled": false, + "textOnlyEnabled": false, + "promptVersion": 1 +} +``` diff --git a/SETUP.md b/SETUP.md @@ -252,8 +252,10 @@ channels. ## Configuration & environment variables Most configuration now lives in the editor's **/settings** page, persisted to -`settings.json` at the repo root (gitignored). `settings.json.example` is a minimal -starting template. Settings are optional — a missing/partial `settings.json` falls +`settings.json` at the repo root (gitignored). Every key, its default and what it +does is in [SETTINGS.md](SETTINGS.md); `settings.json.example` is the defaults as a +starting template. Both are generated from the settings schema +(`common/lib/settingsSchema.ts`). Settings are optional — a missing/partial `settings.json` falls back to built-in defaults, so the app runs out of the box. Paths and binaries resolve through `getPaths()` in `common/lib/paths.ts`. Override diff --git a/common/architecture.test.ts b/common/architecture.test.ts @@ -62,13 +62,6 @@ const ROOTS = [ // Today's back-edges, `<file> -> <imported module>`, each with why it is still // here. THIS LIST IS A DEBT LEDGER, not a policy. const ALLOWED: Record<string, string> = { - // The four auto-queue SANITIZERS are the auto-queue's half of the settings - // schema. They belong in lib/ with the rest of it; that move is phase 3 slice - // 4 (one schema library, one writer), not a rename. The tree/settings TYPES - // already moved (lib/autoQueueTypes.ts). - "lib/settings.ts -> jobs/autoQueuePolicy": - "defaultAutoQueue + three sanitizers; moves with the settings schema in phase 3", - // Three progress-parser factories for the transcription apps. Parsing a // subprocess's stdout is dispatch's job, not the model's; the fix is for the // app descriptor to name a parser the runner resolves, which is phase 1 work. diff --git a/common/bin/settings-example.ts b/common/bin/settings-example.ts @@ -0,0 +1,58 @@ +#!/usr/bin/env tsx +// WRITE settings.json.example AND SETTINGS.md FROM THE SETTINGS SCHEMA. +// +// Usage (from the repo root): +// pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts +// pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts --check +// +// `--check` writes nothing and exits 1 if either committed file differs from +// what the schema generates (the same claim common/lib/settingsDocs.test.ts +// makes). Becomes `archilyzer settings example` in one-core phase 4. +// +// Reads no settings.json and writes no settings.json: both outputs are +// functions of the schema alone. + +import { readFile, writeFile } from "node:fs/promises"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { + renderSettingsExample, + renderSettingsMarkdown, +} from "../lib/settingsDocs"; +import { parseFlags } from "./_parseFlags"; + +const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", ".."); + +const OUTPUTS: ReadonlyArray<[string, () => string]> = [ + ["settings.json.example", renderSettingsExample], + ["SETTINGS.md", renderSettingsMarkdown], +]; + +async function main(): Promise<number> { + const flags = parseFlags(process.argv.slice(2)); + const check = flags.check === "true"; + let stale = 0; + for (const [name, render] of OUTPUTS) { + const file = path.join(REPO, name); + const want = render(); + if (check) { + const have = await readFile(file, "utf8").catch(() => ""); + if (have !== want) { + console.error(`${name} is stale — regenerate it`); + stale++; + } + continue; + } + await writeFile(file, want); + console.log(`wrote ${name}`); + } + return stale > 0 ? 1 : 0; +} + +main().then( + (code) => process.exit(code), + (err) => { + console.error(err); + process.exit(1); + }, +); diff --git a/common/jobs/autoQueuePolicy.test.ts b/common/jobs/autoQueuePolicy.test.ts @@ -9,7 +9,6 @@ import { bucketIdsFrom, bucketLaneWorkIds, bucketsForKind, - defaultAutoQueue, defaultBucketsForPolicy, defaultDrawsForPolicy, isGroup, @@ -18,10 +17,17 @@ import { emptyAutoQueueRuntime, flattenLeaves, policyDrawsBucket, - sanitizeAutoQueue, - sanitizeAutoQueueOrder, selectNextWork, } from "./autoQueuePolicy"; +// REPOINTED, NOT REWRITTEN (one-core phase 3 slice 4a). The defaults and the +// sanitizer moved to lib/autoQueueSchema.ts; every assertion below is the one it +// was, which is the point — this file is the proof that the move changed no +// answer, from the `held` defaults to the tree normalisation. +import { + defaultAutoQueue, + sanitizeAutoQueue, + sanitizeAutoQueueOrder, +} from "../lib/autoQueueSchema"; import { bucketLaneOperationId } from "../lib/operations"; // Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/jobs/autoQueuePolicy.test.ts diff --git a/common/jobs/autoQueuePolicy.ts b/common/jobs/autoQueuePolicy.ts @@ -30,6 +30,22 @@ export type { }; export { LANES, isGroup }; +// THE SANITIZER MOVED DOWN A LAYER (one-core phase 3 slice 4a). Everything that +// turns a raw settings.json value into a legal `AutoQueueSettings` now lives in +// `lib/autoQueueSchema.ts`, with the rest of the settings schema — which is what +// let `lib/settings.ts` stop importing this file. Re-exported here so every +// existing `from "../jobs/autoQueuePolicy"` import still resolves, exactly as +// the TYPES above are. +export { + AUTO_QUEUE_MODES, + AUTO_QUEUE_ORDERS, + AUTO_QUEUE_MAX_WORKERS_MAX, + defaultAutoQueue, + defaultAutoQueuePolicy, + sanitizeAutoQueue, + sanitizeAutoQueueOrder, +} from "../lib/autoQueueSchema"; + // Pure, side-effect-free policy engine for the automatic priority queue. It // decides WHICH pending video to process next, across all channels, from a @@ -46,34 +62,6 @@ export { LANES, isGroup }; // falls through to the next-priority sibling — exactly like HTB's class ceil. -export const AUTO_QUEUE_MODES: ReadonlyArray<AutoQueueMode> = [ - "strict", - "round-robin", - "weighted-fair", -]; - -export const AUTO_QUEUE_ORDERS: ReadonlyArray<AutoQueueOrder> = [ - "listed", - "newest", - "oldest", - "cheapest", -]; - - -// Coerce a stored/raw value to a legal order. Anything unrecognised — including -// a missing field on a settings file written before the field existed — means -// "listed", i.e. today's behaviour. One sanitizer, because the same enum is -// stored on all four lane policies — it was stored in three MORE places before -// slice 1.3 folded digest.recencyOrder and backfill.order into them — and copies -// of this line would eventually disagree about what an absent field means. -export function sanitizeAutoQueueOrder(value: unknown): AutoQueueOrder { - return value === "newest" || value === "oldest" || value === "cheapest" - ? value - : "listed"; -} - - - // Buckets each runner kind draws from, in priority order. A leaf with no // explicit bucket draws from the whole list (union, deduped); the list order is // its internal priority. Single source of truth for the runner, the pending- @@ -263,7 +251,6 @@ export function policyDrawsBucket( ); } -export const AUTO_QUEUE_MAX_WORKERS_MAX = 64; // --- Runtime fairness state (persisted best-effort by autoQueueState.ts) ---- @@ -545,219 +532,3 @@ export function selectNextWork( return pick(root, pending, runtime, active, []); } -// --- Defaults + sanitization (defensive, like sanitizeSyncScheduler) -------- - -function clampMaxWorkers(value: unknown): number | null { - if (value == null) return null; - if (typeof value !== "number" || !Number.isFinite(value)) return null; - const n = Math.floor(value); - if (n < 1) return null; - return Math.min(n, AUTO_QUEUE_MAX_WORKERS_MAX); -} - -function clampWeight(value: unknown): number { - if (typeof value !== "number" || !Number.isFinite(value)) return 1; - const n = Math.floor(value); - return n < 1 ? 1 : Math.min(n, AUTO_QUEUE_MAX_WORKERS_MAX); -} - -function sanitizeMatch(value: unknown): AutoQueueMatch { - const r = (value ?? {}) as Record<string, unknown>; - const type: AutoQueueMatchType = - r.type === "channel" || r.type === "platform" || r.type === "all" - ? r.type - : "all"; - const out: AutoQueueMatch = { type }; - if (typeof r.value === "string" && r.value.trim()) out.value = r.value.trim(); - const operation = - typeof r.operation === "string" && r.operation.trim() - ? r.operation.trim() - : ""; - if (operation) { - // Coerce-to-legal, this file's existing style: a leaf naming BOTH an - // operation and a bucket is ambiguous, so the stored tree is not allowed to - // express it. Operation wins and the bucket is dropped, rather than the - // pair being kept and resolved differently by whichever reader looks first. - out.operation = operation; - return out; - } - if (typeof r.bucket === "string" && r.bucket.trim()) { - out.bucket = r.bucket.trim(); - } - return out; -} - -// Coerce a raw node, assigning a unique id (provided id preserved when valid and -// not already taken, so persisted fairness state survives an unrelated edit). -function sanitizeNode(value: unknown, seen: Set<string>): AutoQueueNode { - const r = (value ?? {}) as Record<string, unknown>; - const id = takeId(r.id, seen); - const weight = clampWeight(r.weight); - const maxWorkers = clampMaxWorkers(r.maxWorkers); - if (Array.isArray(r.children)) { - const mode: AutoQueueMode = AUTO_QUEUE_MODES.includes(r.mode as AutoQueueMode) - ? (r.mode as AutoQueueMode) - : "strict"; - return { - id, - mode, - weight, - maxWorkers, - children: r.children.map((c) => sanitizeNode(c, seen)), - }; - } - return { id, match: sanitizeMatch(r.match), weight, maxWorkers }; -} - -let idCounter = 0; -function takeId(raw: unknown, seen: Set<string>): string { - let id = typeof raw === "string" && raw.trim() ? raw.trim() : ""; - if (!id || seen.has(id)) { - do { - id = `node-${++idCounter}`; - } while (seen.has(id)); - } - seen.add(id); - return id; -} - -function sanitizeRoot(value: unknown, seen: Set<string>): AutoQueueGroup { - const node = sanitizeNode( - value && typeof value === "object" ? value : { mode: "strict", children: [] }, - seen, - ); - if (isGroup(node)) return node; - // A root that deserialized as a leaf is meaningless — wrap into an empty group. - return { id: node.id, mode: "strict", weight: 1, maxWorkers: null, children: [] }; -} - -function emptyRoot(): AutoQueueGroup { - return { id: "root", mode: "strict", weight: 1, maxWorkers: null, children: [] }; -} - -// THE DEFAULT TREE FOR A LANE, and the two answers are different on purpose. -// -// The runner lanes default to an EMPTY root: they have shipped that way since -// the auto-queue existed, an empty tree dispatches nothing, and a settings file -// that omits a root must keep meaning exactly that. -// -// The digest and backfill lanes default to one catch-all leaf, because their -// work list is an operation's `ids` and a lane with no leaf at all could never -// draw it. The leaf is inert while `enabled` is false — which is how they -// default, and what keeps gate B (never enable the backfill lane against -// ~66,540 missingInput videos by accident) a decision an operator still has to -// take. -function defaultRootFor(lane: AutoQueueKind): AutoQueueGroup { - if (lane === "transcription" || lane === "download") return emptyRoot(); - return { - id: "root", - mode: "strict", - weight: 1, - maxWorkers: null, - children: [{ id: "all", match: { type: "all" }, weight: 1, maxWorkers: null }], - }; -} - -// The digest lane's historical ordering is SHORTEST-FIRST, and it is not -// cosmetic: a 12-minute video is one chunk and a four-hour stream is thirty, so -// draining the cheap end first is what makes a multi-week sweep show progress. -// Defaulting the lane to "cheapest" is how that survives the move from the -// sweep to the tree. The comparator arrives with the runner (slice 1.2); until -// then next() has none for this order and falls back to today's. -function defaultOrderFor(lane: AutoQueueKind): AutoQueueOrder { - return lane === "digest" ? "cheapest" : "listed"; -} - -// THE DEFAULT GATE FOR A LANE, and only one lane ships held. -// -// It is not a new policy — it is the reading the four retired pause fields gave -// a file that named no gate, preserved. `transcriptionsPaused`, -// `downloadsPaused` and `digest.digestsPaused` all defaulted false (free); -// `backfill.enabled` defaulted FALSE and was INVERTED, so the backfill lane has -// shipped HELD since it existed. S0-pause deleted the fields, which is what -// makes defaulting this key correct — and required, because from slice 1.4 until -// S0-pause an absent `held` had somewhere else to ask, and now it has not. -// -// The backfill lane is therefore off twice over on a fresh install: unarmed -// (`enabled: false`) and held. That is gate B — never enable the backfill lane -// against ~66,540 missingInput videos by accident — kept as two deliberate acts. -function defaultHeldFor(lane: AutoQueueKind): boolean { - return lane === "backfill"; -} - -export function defaultAutoQueuePolicy( - lane: AutoQueueKind = "transcription", -): AutoQueuePolicy { - return { - enabled: false, - maxWorkers: null, - replaceAutoSubs: false, - order: defaultOrderFor(lane), - snoozeUntil: null, - held: defaultHeldFor(lane), - root: defaultRootFor(lane), - }; -} - -export function defaultAutoQueue(): AutoQueueSettings { - return Object.fromEntries( - LANES.map((lane) => [lane, defaultAutoQueuePolicy(lane)]), - ) as AutoQueueSettings; -} - -// A snooze that has already lapsed is not a snooze: normalizing it to null here -// means every reader (runner, status payload, UI) can treat "non-null" as "still -// snoozed" without repeating the clock comparison. Re-sanitized on every read of -// settings.json, so a stale value self-clears without anyone writing. -function sanitizeSnooze(value: unknown): number | null { - if (typeof value !== "number" || !Number.isFinite(value)) return null; - const at = Math.floor(value); - return at > Date.now() ? at : null; -} - -// A LANE THAT IS NOT IN THE FILE IS THE LANE'S DEFAULT, not an empty object. -// -// Every settings.json in existence carries exactly two lanes, so the digest and -// backfill blocks arrive `undefined` on every read until something writes them. -// Coercing that to `sanitizePolicy({})` would give them an empty root — a lane -// that can never draw anything even once an operator enables it — so the whole -// default policy is the fallback, and only the fields the file actually names -// override it. -function sanitizePolicy(value: unknown, lane: AutoQueueKind): AutoQueuePolicy { - if (value == null) return defaultAutoQueuePolicy(lane); - const r = (value ?? {}) as Record<string, unknown>; - const seen = new Set<string>(); - return { - enabled: r.enabled === true, - maxWorkers: clampMaxWorkers(r.maxWorkers), - // Opt-in only: anything but an explicit `true` (including a missing field on - // a pre-existing settings.json) leaves the lane off. - replaceAutoSubs: r.replaceAutoSubs === true, - // Anything unrecognised (including a missing field) means the lane's own - // default — "listed" for the runner lanes, "cheapest" for digest. - order: r.order === undefined - ? defaultOrderFor(lane) - : sanitizeAutoQueueOrder(r.order), - snoozeUntil: sanitizeSnooze(r.snoozeUntil), - // THE LANE'S PAUSE GATE, and the only spelling of one since S0-pause deleted - // the four legacy fields it migrated from. DEFAULTED, which it deliberately - // was not while those fields existed: an absent key used to mean "ask the - // retired field", so filling it in here would have read a paused corpus as - // running. There is nothing left to ask, and `defaultHeldFor` is the - // reading those fields gave a file that named no gate — free everywhere - // except backfill, whose field was inverted and defaulted to held. - held: typeof r.held === "boolean" ? r.held : defaultHeldFor(lane), - root: - r.root === undefined - ? defaultRootFor(lane) - : sanitizeRoot(r.root, seen), - }; -} - -export function sanitizeAutoQueue(value: unknown): AutoQueueSettings { - if (!value || typeof value !== "object") return defaultAutoQueue(); - const r = value as Record<string, unknown>; - return Object.fromEntries( - LANES.map((lane) => [lane, sanitizePolicy(r[lane], lane)]), - ) as AutoQueueSettings; -} diff --git a/common/lib/autoQueueSchema.ts b/common/lib/autoQueueSchema.ts @@ -0,0 +1,284 @@ +// THE AUTO-QUEUE'S HALF OF THE SETTINGS SCHEMA. +// +// Four lane policies live under `settings.autoQueue`, and until one-core phase 3 +// slice 4a their defaults and their sanitizer lived in `jobs/autoQueuePolicy.ts` +// beside the PICKER that reads them. That was the one back-edge `lib/settings.ts` +// still carried (`common/architecture.test.ts`'s allow-list entry +// `lib/settings.ts -> jobs/autoQueuePolicy`, now deleted): the model layer +// importing dispatch to learn the shape of its own file. +// +// The split is by ROLE, not by size. What is here is everything that turns a +// raw JSON value into a legal `AutoQueueSettings` — the defaults, the clamps, +// the tree normalisation, the lane gate. What stays in `jobs/autoQueuePolicy.ts` +// is everything that CHOOSES with it: bucket lists, pending-set construction and +// the SWRR resolver. `autoQueuePolicy.ts` re-exports every name moved here, so +// no existing import site changed. +// +// It never throws. Every function is total over `unknown`, because its input is +// a file an operator may have hand-edited and a settings read may not fail. +// +// NO ZOD HERE, deliberately. Four `"use client"` forms import constants from +// this module through `jobs/autoQueuePolicy` (LadderRung reads +// AUTO_QUEUE_MODES), so anything this file imports can land in a browser +// bundle. The zod field seam that wraps `sanitizeAutoQueue` lives in +// `lib/settingsFieldSchemas.ts`, which only server code imports. + +import { + type AutoQueueGroup, + type AutoQueueKind, + type AutoQueueMatch, + type AutoQueueMatchType, + type AutoQueueMode, + type AutoQueueNode, + type AutoQueueOrder, + type AutoQueuePolicy, + type AutoQueueSettings, + LANES, + isGroup, +} from "./autoQueueTypes"; + +export const AUTO_QUEUE_MODES: ReadonlyArray<AutoQueueMode> = [ + "strict", + "round-robin", + "weighted-fair", +]; + +export const AUTO_QUEUE_ORDERS: ReadonlyArray<AutoQueueOrder> = [ + "listed", + "newest", + "oldest", + "cheapest", +]; + + +// Coerce a stored/raw value to a legal order. Anything unrecognised — including +// a missing field on a settings file written before the field existed — means +// "listed", i.e. today's behaviour. One sanitizer, because the same enum is +// stored on all four lane policies — it was stored in three MORE places before +// slice 1.3 folded digest.recencyOrder and backfill.order into them — and copies +// of this line would eventually disagree about what an absent field means. +export function sanitizeAutoQueueOrder(value: unknown): AutoQueueOrder { + return value === "newest" || value === "oldest" || value === "cheapest" + ? value + : "listed"; +} + +export const AUTO_QUEUE_MAX_WORKERS_MAX = 64; + +// --- Defaults + sanitization (defensive, like sanitizeSyncScheduler) -------- + +function clampMaxWorkers(value: unknown): number | null { + if (value == null) return null; + if (typeof value !== "number" || !Number.isFinite(value)) return null; + const n = Math.floor(value); + if (n < 1) return null; + return Math.min(n, AUTO_QUEUE_MAX_WORKERS_MAX); +} + +function clampWeight(value: unknown): number { + if (typeof value !== "number" || !Number.isFinite(value)) return 1; + const n = Math.floor(value); + return n < 1 ? 1 : Math.min(n, AUTO_QUEUE_MAX_WORKERS_MAX); +} + +function sanitizeMatch(value: unknown): AutoQueueMatch { + const r = (value ?? {}) as Record<string, unknown>; + const type: AutoQueueMatchType = + r.type === "channel" || r.type === "platform" || r.type === "all" + ? r.type + : "all"; + const out: AutoQueueMatch = { type }; + if (typeof r.value === "string" && r.value.trim()) out.value = r.value.trim(); + const operation = + typeof r.operation === "string" && r.operation.trim() + ? r.operation.trim() + : ""; + if (operation) { + // Coerce-to-legal, this file's existing style: a leaf naming BOTH an + // operation and a bucket is ambiguous, so the stored tree is not allowed to + // express it. Operation wins and the bucket is dropped, rather than the + // pair being kept and resolved differently by whichever reader looks first. + out.operation = operation; + return out; + } + if (typeof r.bucket === "string" && r.bucket.trim()) { + out.bucket = r.bucket.trim(); + } + return out; +} + +// Coerce a raw node, assigning a unique id (provided id preserved when valid and +// not already taken, so persisted fairness state survives an unrelated edit). +function sanitizeNode(value: unknown, seen: Set<string>): AutoQueueNode { + const r = (value ?? {}) as Record<string, unknown>; + const id = takeId(r.id, seen); + const weight = clampWeight(r.weight); + const maxWorkers = clampMaxWorkers(r.maxWorkers); + if (Array.isArray(r.children)) { + const mode: AutoQueueMode = AUTO_QUEUE_MODES.includes(r.mode as AutoQueueMode) + ? (r.mode as AutoQueueMode) + : "strict"; + return { + id, + mode, + weight, + maxWorkers, + children: r.children.map((c) => sanitizeNode(c, seen)), + }; + } + return { id, match: sanitizeMatch(r.match), weight, maxWorkers }; +} + +let idCounter = 0; +function takeId(raw: unknown, seen: Set<string>): string { + let id = typeof raw === "string" && raw.trim() ? raw.trim() : ""; + if (!id || seen.has(id)) { + do { + id = `node-${++idCounter}`; + } while (seen.has(id)); + } + seen.add(id); + return id; +} + +function sanitizeRoot(value: unknown, seen: Set<string>): AutoQueueGroup { + const node = sanitizeNode( + value && typeof value === "object" ? value : { mode: "strict", children: [] }, + seen, + ); + if (isGroup(node)) return node; + // A root that deserialized as a leaf is meaningless — wrap into an empty group. + return { id: node.id, mode: "strict", weight: 1, maxWorkers: null, children: [] }; +} + +function emptyRoot(): AutoQueueGroup { + return { id: "root", mode: "strict", weight: 1, maxWorkers: null, children: [] }; +} + +// THE DEFAULT TREE FOR A LANE, and the two answers are different on purpose. +// +// The runner lanes default to an EMPTY root: they have shipped that way since +// the auto-queue existed, an empty tree dispatches nothing, and a settings file +// that omits a root must keep meaning exactly that. +// +// The digest and backfill lanes default to one catch-all leaf, because their +// work list is an operation's `ids` and a lane with no leaf at all could never +// draw it. The leaf is inert while `enabled` is false — which is how they +// default, and what keeps gate B (never enable the backfill lane against +// ~66,540 missingInput videos by accident) a decision an operator still has to +// take. +function defaultRootFor(lane: AutoQueueKind): AutoQueueGroup { + if (lane === "transcription" || lane === "download") return emptyRoot(); + return { + id: "root", + mode: "strict", + weight: 1, + maxWorkers: null, + children: [{ id: "all", match: { type: "all" }, weight: 1, maxWorkers: null }], + }; +} + +// The digest lane's historical ordering is SHORTEST-FIRST, and it is not +// cosmetic: a 12-minute video is one chunk and a four-hour stream is thirty, so +// draining the cheap end first is what makes a multi-week sweep show progress. +// Defaulting the lane to "cheapest" is how that survives the move from the +// sweep to the tree. The comparator arrives with the runner (slice 1.2); until +// then next() has none for this order and falls back to today's. +function defaultOrderFor(lane: AutoQueueKind): AutoQueueOrder { + return lane === "digest" ? "cheapest" : "listed"; +} + +// THE DEFAULT GATE FOR A LANE, and only one lane ships held. +// +// It is not a new policy — it is the reading the four retired pause fields gave +// a file that named no gate, preserved. `transcriptionsPaused`, +// `downloadsPaused` and `digest.digestsPaused` all defaulted false (free); +// `backfill.enabled` defaulted FALSE and was INVERTED, so the backfill lane has +// shipped HELD since it existed. S0-pause deleted the fields, which is what +// makes defaulting this key correct — and required, because from slice 1.4 until +// S0-pause an absent `held` had somewhere else to ask, and now it has not. +// +// The backfill lane is therefore off twice over on a fresh install: unarmed +// (`enabled: false`) and held. That is gate B — never enable the backfill lane +// against ~66,540 missingInput videos by accident — kept as two deliberate acts. +function defaultHeldFor(lane: AutoQueueKind): boolean { + return lane === "backfill"; +} + +export function defaultAutoQueuePolicy( + lane: AutoQueueKind = "transcription", +): AutoQueuePolicy { + return { + enabled: false, + maxWorkers: null, + replaceAutoSubs: false, + order: defaultOrderFor(lane), + snoozeUntil: null, + held: defaultHeldFor(lane), + root: defaultRootFor(lane), + }; +} + +export function defaultAutoQueue(): AutoQueueSettings { + return Object.fromEntries( + LANES.map((lane) => [lane, defaultAutoQueuePolicy(lane)]), + ) as AutoQueueSettings; +} + +// A snooze that has already lapsed is not a snooze: normalizing it to null here +// means every reader (runner, status payload, UI) can treat "non-null" as "still +// snoozed" without repeating the clock comparison. Re-sanitized on every read of +// settings.json, so a stale value self-clears without anyone writing. +function sanitizeSnooze(value: unknown): number | null { + if (typeof value !== "number" || !Number.isFinite(value)) return null; + const at = Math.floor(value); + return at > Date.now() ? at : null; +} + +// A LANE THAT IS NOT IN THE FILE IS THE LANE'S DEFAULT, not an empty object. +// +// Every settings.json in existence carries exactly two lanes, so the digest and +// backfill blocks arrive `undefined` on every read until something writes them. +// Coercing that to `sanitizePolicy({})` would give them an empty root — a lane +// that can never draw anything even once an operator enables it — so the whole +// default policy is the fallback, and only the fields the file actually names +// override it. +function sanitizePolicy(value: unknown, lane: AutoQueueKind): AutoQueuePolicy { + if (value == null) return defaultAutoQueuePolicy(lane); + const r = (value ?? {}) as Record<string, unknown>; + const seen = new Set<string>(); + return { + enabled: r.enabled === true, + maxWorkers: clampMaxWorkers(r.maxWorkers), + // Opt-in only: anything but an explicit `true` (including a missing field on + // a pre-existing settings.json) leaves the lane off. + replaceAutoSubs: r.replaceAutoSubs === true, + // Anything unrecognised (including a missing field) means the lane's own + // default — "listed" for the runner lanes, "cheapest" for digest. + order: r.order === undefined + ? defaultOrderFor(lane) + : sanitizeAutoQueueOrder(r.order), + snoozeUntil: sanitizeSnooze(r.snoozeUntil), + // THE LANE'S PAUSE GATE, and the only spelling of one since S0-pause deleted + // the four legacy fields it migrated from. DEFAULTED, which it deliberately + // was not while those fields existed: an absent key used to mean "ask the + // retired field", so filling it in here would have read a paused corpus as + // running. There is nothing left to ask, and `defaultHeldFor` is the + // reading those fields gave a file that named no gate — free everywhere + // except backfill, whose field was inverted and defaulted to held. + held: typeof r.held === "boolean" ? r.held : defaultHeldFor(lane), + root: + r.root === undefined + ? defaultRootFor(lane) + : sanitizeRoot(r.root, seen), + }; +} + +export function sanitizeAutoQueue(value: unknown): AutoQueueSettings { + if (!value || typeof value !== "object") return defaultAutoQueue(); + const r = value as Record<string, unknown>; + return Object.fromEntries( + LANES.map((lane) => [lane, sanitizePolicy(r[lane], lane)]), + ) as AutoQueueSettings; +} + diff --git a/common/lib/autoQueueTypes.ts b/common/lib/autoQueueTypes.ts @@ -9,6 +9,8 @@ // // Enforced by ../architecture.test.ts. +import type { FieldDocs } from "./fieldDocs"; + // --- Tree types ------------------------------------------------------------- export type AutoQueueMode = "strict" | "round-robin" | "weighted-fair"; @@ -33,40 +35,48 @@ export type AutoQueueMatchType = "channel" | "platform" | "all"; // which is what makes it safe to name here before anything computes it. export type AutoQueueOrder = "listed" | "newest" | "oldest" | "cheapest"; +// Each field is documented in AUTO_QUEUE_MATCH_FIELD_DOCS below (rendered into SETTINGS.md). export type AutoQueueMatch = { type: AutoQueueMatchType; - // Channel slug (type=channel) or platform name (type=platform). Ignored for - // type=all. A type=channel leaf with no value matches nothing. value?: string; - // Optional snapshot bucket this leaf draws from, narrowing the default for the - // runner kind (transcription → downloadedNoTranscript, download → - // undownloadedIds). E.g. bucket="failedListed" prioritizes retries. bucket?: string; - // Optional OPERATION this leaf draws from — a registered backfill kind id, or - // "digest". Same meaning as `bucket` one level up: it narrows what the leaf - // claims, and it draws from ChannelWork.operations rather than - // ChannelWork.buckets. - // - // It lives on the MATCH, beside `bucket`, and not on the node. A field on the - // node would need group inheritance — "this group is the digest subtree" — - // and inheritance is resolution logic buildPendingByLeaf does not have. Here - // it needs exactly one sanitizer and exactly one claiming path. - // - // It is a SEPARATE id space from `bucket`, and the sanitizer enforces that a - // leaf names at most one of the two (operation wins): `defaultBuckets` is a - // priority-ordered union, so a name that meant a bucket to one leaf and an - // operation to another would silently mix two id spaces, and - // selectableBucketsForKind feeds the editor's bucket dropdown, where an - // operation must not appear as a bucket. operation?: string; }; +export const AUTO_QUEUE_MATCH_FIELD_DOCS: FieldDocs<AutoQueueMatch> = { + type: + "What the leaf matches: \"channel\" (one channel slug in `value`), \"platform\" (a platform name in `value`), or \"all\".", + value: + "Channel slug (type=channel) or platform name (type=platform). Ignored " + + "for type=all. A type=channel leaf with no value matches nothing.", + bucket: + "Optional snapshot bucket this leaf draws from, narrowing the default " + + "for the runner kind (transcription → downloadedNoTranscript, download " + + "→ undownloadedIds). E.g. bucket=\"failedListed\" prioritizes retries.", + operation: + "Optional OPERATION this leaf draws from — a registered backfill kind " + + "id, or \"digest\". Same meaning as `bucket` one level up: it narrows " + + "what the leaf claims, and it draws from ChannelWork.operations rather " + + "than ChannelWork.buckets.\n\n" + + "It lives on the MATCH, beside `bucket`, and not on the node. A field " + + "on the node would need group inheritance — \"this group is the digest " + + "subtree\" — and inheritance is resolution logic buildPendingByLeaf does" + + " not have. Here it needs exactly one sanitizer and exactly one " + + "claiming path.\n\n" + + "It is a SEPARATE id space from `bucket`, and the sanitizer enforces " + + "that a leaf names at most one of the two (operation wins): " + + "`defaultBuckets` is a priority-ordered union, so a name that meant a " + + "bucket to one leaf and an operation to another would silently mix two " + + "id spaces, and selectableBucketsForKind feeds the editor's bucket " + + "dropdown, where an operation must not appear as a bucket.", +}; + +// A node of a lane's rule tree is a leaf or a group; both are documented in +// AUTO_QUEUE_NODE_FIELD_DOCS below (rendered into SETTINGS.md). export type AutoQueueLeaf = { id: string; match: AutoQueueMatch; - // Relative share under a weighted-fair parent. Default 1. Ignored otherwise. weight?: number; - // Optional ceiling on concurrent in-flight workers drawn from this leaf. maxWorkers?: number | null; }; @@ -80,57 +90,96 @@ export type AutoQueueGroup = { export type AutoQueueNode = AutoQueueLeaf | AutoQueueGroup; +export const AUTO_QUEUE_NODE_FIELD_DOCS: FieldDocs<AutoQueueNode> = { + id: + "Stable node id, unique within the lane's tree. Preserved on save when " + + "valid and not taken, so persisted fairness state survives an unrelated " + + "edit; a missing or duplicate id is replaced with a generated one.", + match: + "LEAF ONLY: which videos this leaf owns (see the match table).", + weight: + "Relative share under a weighted-fair parent. Default 1. Ignored " + + "otherwise.", + maxWorkers: + "Optional ceiling on concurrent in-flight workers drawn from this node " + + "(and, for a group, its whole subtree). A capped node reads as \"no " + + "work\" and the parent falls through to the next sibling, like an HTB " + + "class ceiling. null = no cap.", + mode: + "GROUP ONLY: how the children compete — \"strict\" (first child with " + + "work wins), \"round-robin\", or \"weighted-fair\" (by each child's " + + "`weight`). Unknown values read as \"strict\".", + children: + "GROUP ONLY: the child nodes, in priority order for a strict group. A " + + "node with a `children` array is a group; any other node is a leaf.", +}; + export function isGroup(node: AutoQueueNode): node is AutoQueueGroup { return Array.isArray((node as AutoQueueGroup).children); } // --- Settings (persisted in settings.json under `autoQueue`) ---------------- +// Each field is documented in AUTO_QUEUE_POLICY_FIELD_DOCS below (rendered into SETTINGS.md). export type AutoQueuePolicy = { - // Master switch for this runner (transcription / download independently). enabled: boolean; - // Overall ceiling on concurrent in-flight workers for this runner. null = no - // runner-level cap (the worker pool / platform queues are the real throttle). maxWorkers: number | null; - // Opt in to the lowest-priority "replace YouTube auto-captions" lane: append - // this kind's opt-in buckets (autoSubsOnly / downloadedAutoSubsOnly) to the - // tail of the default union, so videos whose only transcript is YouTube ASR - // get re-done with our own engine whenever nothing more important is pending. - // Default false — the corpus-wide cost is large (an audio download plus a - // transcription per video). A leaf can also target the bucket by name for - // per-channel opt-in without flipping this switch. Optional: settings written - // before this field existed lack it; the sanitizer defaults it to false. replaceAutoSubs?: boolean; - // Ordering within each rule (see AutoQueueOrder). Optional exactly like - // replaceAutoSubs: settings files written before this field existed lack it, - // and the sanitizer defaults them to "listed" (today's behaviour). order?: AutoQueueOrder; - // Epoch ms until which this runner idles WITHOUT stopping: next() returns null - // so the loop stays up, re-reads settings each iteration, and resumes by - // itself when the moment passes. null/absent/past = not snoozed. Survives a - // restart because it lives in settings.json, not in runner memory. snoozeUntil?: number | null; - // THE LANE'S PAUSE GATE. Shut means the lane holds: every dispatch path asks - // lib/pauseGates.ts, whose limit()/guard returns 0 so runPool idle-waits. A - // hold, never a stop — see that file's header. - // - // OPTIONAL IN THE TYPE, FILLED BY THE SANITIZER. Until slice 1.4 four separate - // settings fields carried this — `transcriptionsPaused`, `downloadsPaused`, - // `digest.digestsPaused` and (inverted) `backfill.enabled` — so `undefined` - // meant "ask the legacy field" and `sanitizePolicy` deliberately refused to - // default it: a default would have read a paused corpus as running. S0-pause - // deleted those four, on the precondition that the live settings.json already - // carried every `held` key, and the default came in with them - // (`defaultHeldFor` — free everywhere except backfill, whose field was - // inverted and shipped held). - // - // It stays optional because a reader may be handed a PARTIAL settings object - // (laneGuards.test.ts casts one), and `isGateHeld` answers `false` for a lane - // that carries no key at all rather than throwing. held?: boolean; root: AutoQueueGroup; }; +export const AUTO_QUEUE_POLICY_FIELD_DOCS: FieldDocs<AutoQueuePolicy> = { + enabled: + "Master switch for this runner (transcription / download " + + "independently).", + maxWorkers: + "Overall ceiling on concurrent in-flight workers for this runner. null " + + "= no runner-level cap (the worker pool / platform queues are the real " + + "throttle).", + replaceAutoSubs: + "Opt in to the lowest-priority \"replace YouTube auto-captions\" lane: " + + "append this kind's opt-in buckets (autoSubsOnly / " + + "downloadedAutoSubsOnly) to the tail of the default union, so videos " + + "whose only transcript is YouTube ASR get re-done with our own engine " + + "whenever nothing more important is pending. Default false — the " + + "corpus-wide cost is large (an audio download plus a transcription per " + + "video). A leaf can also target the bucket by name for per-channel opt-" + + "in without flipping this switch. Optional: settings written before " + + "this field existed lack it; the sanitizer defaults it to false.", + order: + "Ordering within each rule (see AutoQueueOrder). Optional exactly like " + + "replaceAutoSubs: settings files written before this field existed lack" + + " it, and the sanitizer defaults them to \"listed\" (today's behaviour).", + snoozeUntil: + "Epoch ms until which this runner idles WITHOUT stopping: next() " + + "returns null so the loop stays up, re-reads settings each iteration, " + + "and resumes by itself when the moment passes. null/absent/past = not " + + "snoozed. Survives a restart because it lives in settings.json, not in " + + "runner memory.", + held: + "THE LANE'S PAUSE GATE. Shut means the lane holds: every dispatch path " + + "asks lib/pauseGates.ts, whose limit()/guard returns 0 so runPool idle-" + + "waits. A hold, never a stop — see that file's header.\n\n" + + "OPTIONAL IN THE TYPE, FILLED BY THE SANITIZER. Until slice 1.4 four " + + "separate settings fields carried this — `transcriptionsPaused`, " + + "`downloadsPaused`, `digest.digestsPaused` and (inverted) " + + "`backfill.enabled` — so `undefined` meant \"ask the legacy field\" and " + + "`sanitizePolicy` deliberately refused to default it: a default would " + + "have read a paused corpus as running. S0-pause deleted those four, on " + + "the precondition that the live settings.json already carried every " + + "`held` key, and the default came in with them (`defaultHeldFor` — free" + + " everywhere except backfill, whose field was inverted and shipped " + + "held).\n\n" + + "It stays optional because a reader may be handed a PARTIAL settings " + + "object (laneGuards.test.ts casts one), and `isGateHeld` answers " + + "`false` for a lane that carries no key at all rather than throwing.", + root: + "The lane's rule tree: a group whose children are groups and leaves (see the node table). A missing root is the lane's default — empty for the runner lanes, one catch-all leaf for digest and backfill. While a channel-priority document exists, the four roots are compiled from it and not hand-edited.", +}; + export type AutoQueueSettings = Record<AutoQueueKind, AutoQueuePolicy>; // --- Lanes ------------------------------------------------------------------ diff --git a/common/lib/channelPriority.ts b/common/lib/channelPriority.ts @@ -58,6 +58,7 @@ import { type AutoQueueNode, isGroup, } from "./autoQueueTypes"; +import type { FieldDocs } from "./fieldDocs"; // --- The vocabulary --------------------------------------------------------- @@ -119,56 +120,98 @@ export type ChannelFocus = | { kind: "site"; siteId: string } | { kind: "channels"; slugs: string[] }; +export const CHANNEL_FOCUS_FIELD_DOCS: FieldDocs<ChannelFocus> = { + kind: + "\"none\" (no focus), \"site\" (the channels of one site, resolved at " + + "compile time so it tracks membership) or \"channels\" (an explicit " + + "list, from \"Focus these\").", + siteId: + "kind \"site\" only: the site whose channels are focused. A blank id " + + "reads as no focus; an unknown one survives and focuses nothing.", + slugs: + "kind \"channels\" only: the focused channel slugs, trimmed and " + + "de-duplicated. An empty list reads as no focus.", +}; + +// Each field is documented in CHANNEL_PRIORITY_ENTRY_FIELD_DOCS below (rendered into SETTINGS.md). export type ChannelPriorityEntry = { - // THE BASE TIER: what every operation gets unless an override says otherwise. tier: StoredChannelTier; - // Order WITHIN the tier, ascending. Absent = unranked, which sorts after - // every ranked sibling and then by slug. ONE rank per channel, not one per - // lane — the two hand-made lane orders collapse into this on migration. rank?: number; - // PER-OPERATION OVERRIDES of the base tier. Only operations that DIFFER from - // the base appear: the sanitizer normalises an override equal to `tier` away, - // so the on-disk document stays a list of exceptions to a list of exceptions. - // - // `{tier:"normal", overrides:{sync:"paused"}}` is "everything but sync" — the - // lossless reading of the retired `excludeFromSync`. Its inverse, - // `{tier:"paused", overrides:{sync:"normal"}}`, is "sync only": keep the - // playlist and metadata current, dispatch nothing. overrides?: Partial<Record<PriorityOperation, StoredChannelTier>>; - // PAUSED BY THE MACHINE, NOT BY THE OPERATOR, and what to put back. - // - // Set when the drive a channel's media is on stops being there: the watch - // pass records the tier the channel HAD and forces `paused`, so nothing in - // any lane dispatches against a `data/` nobody can read. Cleared — and the - // tier restored — when the drive comes back. - // - // WHY IT IS A FIELD AND NOT A DERIVED STATE. The lanes read `tier`; making - // them all ask a second question would be four more places to forget. And - // the tier the channel is to be RESTORED to is not derivable from anything - // once it has been overwritten — that is the whole content of this field. - // - // OPTIONAL, and an older binary that drops it leaves the channel Paused with - // nothing lost but the automatic restore. The operator's own word always - // wins: a MANUAL tier change clears it (see clearAutoPause), so a drive - // coming back can never un-pause a channel somebody paused on purpose. - autoPaused?: { - // One reason today. A union so a second one has somewhere to go, and so a - // surface can say WHICH machine decided rather than "automatic". - reason: "storage"; - // ISO, for "auto-paused — media unreachable since <date>". - since: string; - previousTier: StoredChannelTier; - }; + autoPaused?: ChannelAutoPause; }; +export const CHANNEL_PRIORITY_ENTRY_FIELD_DOCS: FieldDocs<ChannelPriorityEntry> = { + tier: + "THE BASE TIER: what every operation gets unless an override says " + + "otherwise.", + rank: + "Order WITHIN the tier, ascending. Absent = unranked, which sorts after" + + " every ranked sibling and then by slug. ONE rank per channel, not one " + + "per lane — the two hand-made lane orders collapse into this on " + + "migration.", + overrides: + "PER-OPERATION OVERRIDES of the base tier. Only operations that DIFFER " + + "from the base appear: the sanitizer normalises an override equal to " + + "`tier` away, so the on-disk document stays a list of exceptions to a " + + "list of exceptions.\n\n" + + "`{tier:\"normal\", overrides:{sync:\"paused\"}}` is \"everything but sync\" " + + "— the lossless reading of the retired `excludeFromSync`. Its inverse, " + + "`{tier:\"paused\", overrides:{sync:\"normal\"}}`, is \"sync only\": keep the" + + " playlist and metadata current, dispatch nothing.", + autoPaused: + "PAUSED BY THE MACHINE, NOT BY THE OPERATOR, and what to put back.\n\n" + + "Set when the drive a channel's media is on stops being there: the " + + "watch pass records the tier the channel HAD and forces `paused`, so " + + "nothing in any lane dispatches against a `data/` nobody can read. " + + "Cleared — and the tier restored — when the drive comes back.\n\n" + + "WHY IT IS A FIELD AND NOT A DERIVED STATE. The lanes read `tier`; " + + "making them all ask a second question would be four more places to " + + "forget. And the tier the channel is to be RESTORED to is not derivable" + + " from anything once it has been overwritten — that is the whole " + + "content of this field.\n\n" + + "OPTIONAL, and an older binary that drops it leaves the channel Paused " + + "with nothing lost but the automatic restore. The operator's own word " + + "always wins: a MANUAL tier change clears it (see clearAutoPause), so a" + + " drive coming back can never un-pause a channel somebody paused on " + + "purpose.", +}; + +// The machine's pause record on a channel entry (see `autoPaused` above). +// Each field is documented in CHANNEL_AUTO_PAUSE_FIELD_DOCS below (rendered into SETTINGS.md). +export type ChannelAutoPause = { + reason: "storage"; + since: string; + previousTier: StoredChannelTier; +}; + +export const CHANNEL_AUTO_PAUSE_FIELD_DOCS: FieldDocs<ChannelAutoPause> = { + reason: + "One reason today. A union so a second one has somewhere to go, and so " + + "a surface can say WHICH machine decided rather than \"automatic\".", + since: + "ISO, for \"auto-paused — media unreachable since <date>\".", + previousTier: + "The base tier the channel had before the machine paused it; what a " + + "restore puts back. Never `paused` (that would restore to paused — a " + + "no-op dressed as a restore).", +}; + +// Each field is documented in CHANNEL_PRIORITY_FIELD_DOCS below (rendered into SETTINGS.md). export type ChannelPriority = { focus: ChannelFocus; - // ONLY channels that differ from the default appear. An absent slug is - // `normal`, unranked — so the default document is empty and "absent document - // = today's behaviour" holds byte for byte. channels: Record<string, ChannelPriorityEntry>; }; +export const CHANNEL_PRIORITY_FIELD_DOCS: FieldDocs<ChannelPriority> = { + focus: + "The corpus-wide focus selector: none, one site's channels, or a list of channels. A focus is compiled into a leading `prio-focus` group in every lane's tree.", + channels: + "ONLY channels that differ from the default appear. An absent slug is " + + "`normal`, unranked — so the default document is empty and \"absent " + + "document = today's behaviour\" holds byte for byte.", +}; + export function defaultChannelPriority(): ChannelPriority { return { focus: { kind: "none" }, channels: {} }; } diff --git a/common/lib/digest.ts b/common/lib/digest.ts @@ -17,6 +17,8 @@ // NEVER rename these to `transcript.<x>.<y>` — SUB_FILE_RE in videoStatus.ts // would claim such a file as a subtitle track. +import type { FieldDocs } from "./fieldDocs"; + export const DIGEST_FILENAME = "ai-digest.json"; export const DIGEST_OVERRIDES_FILENAME = "ai-digest.overrides.json"; @@ -144,33 +146,46 @@ export const DEFAULT_DIGEST_APP_ID = OLLAMA_DIGEST_APP_ID; // Per-app configuration persisted under settings.digest.apps[id]. Every field is // optional; an app falls back to its own defaults. +// Each field is documented in DIGEST_APP_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md). export type DigestAppConfig = { - // Binary path/name override (process-based apps only). bin?: string; - // Base URL override (HTTP apps only). baseUrl?: string; - // Model id, e.g. "qwen2.5:7b" or "haiku". model?: string; - // Context window in tokens. MUST reach the engine explicitly for ollama: its - // 4096 default silently truncates the input and the model then summarizes - // whatever fragment survived — measured, and the single easiest way to get - // quietly-wrong output at scale. numCtx?: number; - // Sampling temperature. 0 for a structured extraction task. temperature?: number; - // Reasoning-model toggle (ollama's top-level `think`). Only sent when set, so - // a model that does not support thinking is never handed a field it rejects. - // - // It matters for throughput, not correctness: measured on this box, qwen3:8b - // with thinking on spends most of its output budget on a `thinking` block - // before the JSON body the schema constrains. For an extraction task with a - // pinned schema that reasoning buys little and costs a multiple of the tokens, - // and tokens are what a multi-week sweep is priced in. think?: boolean; - // Per-request wall-clock ceiling (ms). A wedged engine must not stall a sweep. timeoutMs?: number; }; +export const DIGEST_APP_CONFIG_FIELD_DOCS: FieldDocs<DigestAppConfig> = { + bin: + "Binary path/name override (process-based apps only).", + baseUrl: + "Base URL override (HTTP apps only).", + model: + "Model id, e.g. \"qwen2.5:7b\" or \"haiku\".", + numCtx: + "Context window in tokens. MUST reach the engine explicitly for ollama:" + + " its 4096 default silently truncates the input and the model then " + + "summarizes whatever fragment survived — measured, and the single " + + "easiest way to get quietly-wrong output at scale.", + temperature: + "Sampling temperature. 0 for a structured extraction task.", + think: + "Reasoning-model toggle (ollama's top-level `think`). Only sent when " + + "set, so a model that does not support thinking is never handed a field" + + " it rejects.\n\n" + + "It matters for throughput, not correctness: measured on this box, " + + "qwen3:8b with thinking on spends most of its output budget on a " + + "`thinking` block before the JSON body the schema constrains. For an " + + "extraction task with a pinned schema that reasoning buys little and " + + "costs a multiple of the tokens, and tokens are what a multi-week sweep" + + " is priced in.", + timeoutMs: + "Per-request wall-clock ceiling (ms). A wedged engine must not stall a " + + "sweep.", +}; + // Whether an engine-reported model resolution looks like a DIFFERENT model // rather than a benign tag completion. "qwen2.5" resolving to "qwen2.5:7b" or // "qwen2.5:latest" is ollama filling in a tag; "qwen2.5:7b" coming back as diff --git a/common/lib/fieldDocs.ts b/common/lib/fieldDocs.ts @@ -0,0 +1,14 @@ +// A DESCRIPTION FOR EVERY KEY OF A SETTINGS BLOCK, checked by the compiler. +// +// Each settings.json block type carries a `<TYPE>_FIELD_DOCS: FieldDocs<Type>` +// record beside it. The mapped type requires one entry per key — optional keys +// included, and every member's keys when the type is a union — so adding a +// field without documenting it is a tsc error, and a stale entry for a removed +// field is an excess-property error. The records are rendered into SETTINGS.md +// by lib/settingsDocs.ts; they are the one home of each field's documentation. +// +// Pure, no imports: a `"use client"` module may carry a record. + +type AllKeys<T> = T extends unknown ? keyof T : never; + +export type FieldDocs<T> = { readonly [K in AllKeys<T> & string]: string }; diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -1,11 +1,30 @@ +// THE ONE READER AND THE ONE WRITER OF settings.json. +// +// The shape — every field, its default, its clamp, its documentation — is +// `siteSettingsSchema` in ./settingsSchema.ts (one-core phase 3 slice 4a), and +// everything that module exports is re-exported here, so the ~200 importers of +// `lib/settings` (types, constants, clamps, sanitizers) did not move. +// +// What is left in this file is exactly what a schema cannot do: +// +// - READ STAYS LENIENT. A settings.json is whatever an operator, an older +// build, or a half-finished write left behind. getSettings never throws: +// an unreadable or non-object file reads as the empty one, and every field +// is total. Three migrations are keyed on a field's ABSENCE in the raw file +// — which a parsed object cannot see, because parsing folds the default in +// — so they run around the parse, each handed the RAW object. +// - WRITE STAYS STRICT. writeSettings derives the worker shadow, runs the two +// validators that THROW (a worker list that cannot transcribe, a social +// link whose SVG is unsafe), parses through the same schema — which is what +// drops every key it does not name, retired fields included — and writes +// atomically (tmp + rename). +// +// One schema, both directions: the only differences between what a read and a +// write produce are those migrations and those two validators. + import fs from "node:fs"; import path from "node:path"; import { getPaths } from "./paths"; -import { SYNC_INTERVAL_MAX_MINUTES } from "./channelConfig"; -import { - isDownloadFormatPreset, - type DownloadFormatPreset, -} from "../ytdlp/downloadFormat"; import { type AppInstanceConfig, DEFAULT_TRANSCRIBE_ARGS, @@ -13,1587 +32,98 @@ import { TRANSCRIPTION_APPS, } from "./transcriptionApps"; import { - type Worker, defaultWorkersFromApps, - sanitizeWorkerConfig, sanitizeWorkers, validateWorkers, } from "./workers"; -import type { AutoQueueSettings } from "./autoQueueTypes"; -import { - defaultChannelPriority, - sanitizeChannelPriority, - type ChannelPriority, -} from "./channelPriority"; import { migrateSweepsToLanes } from "./laneMigration"; +import { migrateMediaRootToLocations } from "./storageLocations"; import { - INTERNAL_LOCATION_ID, - migrateMediaRootToLocations, - type StorageLocation, - type StorageSettings, - type StorageVolume, -} from "./storageLocations"; -// The four SANITIZERS still come from the engine. They are the auto-queue's -// half of the settings schema and belong in lib/ with the rest of it, but that -// move is phase 3 slice 4 (one schema, one writer) — not a rename. Recorded in -// ../architecture.test.ts's allow-list until then. -import { - defaultAutoQueue, - sanitizeAutoQueue, -} from "../jobs/autoQueuePolicy"; -import { - DEFAULT_DIARIZATION_ENGINE, - DEFAULT_DIARIZATION_THRESHOLD, - DIARIZATION_BACKENDS, - DIARIZATION_ENGINE_IDS, - type DiarizationBackend, - type DiarizationEngineId, -} from "./diarization"; -import { ATTRIBUTION_PROMPT_VERSION } from "./attribution"; -import { - DEFAULT_COOKIE_MODE, - isCookieMode, - type CookieMode, -} from "./cookiePolicy"; -// From the CLIENT-SAFE digest module, deliberately — digestApps.ts imports execa, -// and settings.ts must stay reachable from anywhere. -import { - CLAUDE_DIGEST_APP_ID, - DEFAULT_DIGEST_APP_ID, - DEFAULT_DIGEST_TIMESTAMP_MODE, - DIGEST_SECTION_KINDS, - DIGEST_TIMESTAMP_MODES, - isDigestSectionKind, - isDigestTimestampMode, - type DigestAppConfig, - type DigestSectionKind, - type DigestTimestampMode, -} from "./digest"; - -export type { Worker } from "./workers"; -export type { AutoQueueSettings } from "./autoQueueTypes"; -export type { ChannelPriority } from "./channelPriority"; - -// Transcribe placeholder/arg helpers now live with the whisper-cpp app in -// transcriptionApps.ts. Re-exported here so existing import sites keep working. -export { - type AppInstanceConfig, - TRANSCRIBE_PLACEHOLDER_AUDIO, - TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE, - TRANSCRIBE_PLACEHOLDER_MODEL, - TRANSCRIBE_KNOWN_PLACEHOLDERS, - DEFAULT_TRANSCRIBE_ARGS, - validateTranscribeArgs, -} from "./transcriptionApps"; - -// Global, OPERATIONAL settings shared across every site this editor powers. -// Per-site presentation (branding, social links, channel groups, membership) -// lives in sites/<siteId>/site.json — see common/lib/site.ts. -export type SiteSettings = { - // Title for the EDITOR admin shell only (the editor manages all sites and so - // is not tied to any one site's branding). Public sites get their own titles - // from site.json. - adminTitle: string; - maxTranscriptPageBytes: number; - // Active transcription app id (key into TRANSCRIPTION_APPS, e.g. "whisper-cpp" - // or "chough"). Selected globally; see common/lib/transcriptionApps.ts. - transcriptionApp: string; - // Per-app configuration, keyed by app id. Each app reads only its own block; - // a missing block means "use the app's defaults". DEPRECATED in favor of - // `workers` (each local worker carries its own config); kept one release to - // drive migration and allow rollback. See common/lib/workers.ts. - transcriptionApps: Record<string, AppInstanceConfig>; - // Configured transcription workers (named processing slots). The scheduler - // distributes each video to the highest-priority free worker. A settings.json - // predating this field is migrated to a single enabled worker from the active - // app (see defaultWorkersFromApps). See common/lib/workers.ts. - workers: Worker[]; - // Browser spec (e.g. "firefox", "chrome:Default") passed to - // `yt-dlp --cookies-from-browser`. WHEN it is passed is governed by - // `cookieMode` below. Empty string = no cookies configured. Per-channel - // override available (ChannelConfig.cookiesFromBrowser). - cookiesFromBrowser: string; - // How yt-dlp invocations use the configured cookies (see - // common/lib/cookiePolicy.ts): "always" passes them on every invocation, - // "when-required" (default; the historical behavior) only to retry an - // auth/age failure, "defer" never in normal runs — auth-gated videos are - // excluded from batches and collected into the per-channel "Needs cookies" - // bucket for a manual cookie run. Per-channel override available - // (ChannelConfig.cookieMode). - cookieMode: CookieMode; - // Pause (seconds) inserted between per-video yt-dlp invocations in - // managed batch downloads. yt-dlp's own `-t sleep` only paces requests - // within one invocation, so without this the managed loop hammers the - // source IP back-to-back. 0 disables. Per-channel override available. - sleepBetweenDownloadsSeconds: number; - // Default yt-dlp `-f` download format for every channel that doesn't set its - // own (ChannelConfig.downloadFormat). "auto" picks per-source: `original` for - // Odysee (whose HLS rungs are CDN-truncated), `bestaudio/worst` elsewhere. See - // common/ytdlp/downloadFormat.ts. - downloadFormat: DownloadFormatPreset; - // Minimum free disk space (GB) required on the transcripts data directory for - // downloads to run. When free space is below this floor, a download job is - // prevented from starting and a running batch stops launching new videos - // (the in-flight one finishes). 0 disables the gate. See common/lib/diskSpace.ts. - minFreeDiskGB: number; - // Extra headroom (GB) above minFreeDiskGB that a stopped pipeline must see - // before it resumes. Resuming at the same number we stopped at flaps — the - // first restarted download pushes free space back under the floor. This is - // the hysteresis margin, so "resumed" means the operator actually freed - // something rather than a scratch file being cleaned up. 0 disables the - // hysteresis (resume at the floor). See diskGate() in common/lib/diskSpace.ts. - resumeMarginGB: number; - // Default number of videos transcribed in parallel when a "Transcribe - // missing" / bucket run doesn't specify its own concurrency. The per-run - // Concurrency input in the channel UI overrides this for a single run. - parallelTranscriptions: number; - // When true, the no-subs fallback in the managed downloader runs whisper - // inline immediately after the audio download succeeds. When false - // (default), audio is left for the next "Transcribe missing" pass so a - // batch download finishes faster and whisper can parallelize. - inlineTranscribeOnFallback: boolean; - // When true (default), managed downloads skip videos that are currently live - // or scheduled/upcoming, decided from a metadata-only prefetch pass. Finished - // livestream VODs (was_live) are NOT skipped and download normally. A skip is - // recorded but not archived, so the next sync/download-missing retries the - // video once the stream ends. Per-channel override available - // (ChannelConfig.skipLiveDownloads). - skipLiveDownloads: boolean; - // Whether the transcribed-audio cleanup sweep checks each candidate is still - // available upstream before deleting its audio, pinning (do-not-clean) any - // video found permanently gone. The delete is irreversible and a gone video's - // audio is irreplaceable, so this defaults to true. Turn it off for an offline - // or URL-less setup, where the check can never resolve and cleanup would - // otherwise never delete anything. See verifyBeforeClean.ts. - verifyAvailabilityBeforeClean: boolean; - // Whether site builds generate downloadable transcript/live-chat archive zips - // (into public/archives, linked on the Downloads page). Global default; a site - // can opt out via site.json `archives: false`, and a single build can skip via - // the "Skip archive zips" build control. Opt-out: default true. - buildArchives: boolean; - // Overflow object storage (Cloudflare R2) for archive zips that exceed the - // Pages per-file size cap (see Site.archiveMaxBytes). When both fields are set, - // an oversize archive is uploaded here on deploy — via `wrangler r2 object put`, - // keyed `<siteId>/archives/<file>` — instead of being dropped, and the Downloads - // page links to `<publicBaseUrl>/<key>`. Blank/absent → no overflow, so oversize - // archives stay unavailable ("Too large to host"). - archiveStorage?: { bucket: string; publicBaseUrl: string }; - // Debounce preset for the global snapshot scheduler: how long it waits after - // the last report-changing action before regenerating affected channel - // reports. See REPORT_DEBOUNCE_PRESETS. Default "fast" (~1s, no cap). - reportDebouncePreset: ReportDebouncePreset; - // How often (seconds) the editor UI passively re-fetches the current page's - // server-rendered data via router.refresh(), so sidebar badges and reports - // stay live without a manual reload. Mounted globally; pauses while the tab is - // hidden. 0 disables passive refresh entirely. See AUTO_REFRESH_INTERVAL_*. - autoRefreshIntervalSeconds: number; - // Global configuration for the scheduled (cron-driven) channel sync system. - // The per-channel cadence lives on ChannelConfig.syncIntervalMinutes; this - // block holds the defaults and guard rails the scheduler applies across all - // channels. See common/jobs/syncScheduler.ts. - syncScheduler: SyncSchedulerSettings; - // Configuration for the automatic priority-queue runners (auto-transcribe / - // auto-download). Each holds a tree policy that decides which channel's video - // to process next, cross-channel, by priority/round-robin/weighted-fair rules. - // Independent of syncScheduler (which decides staleness, not work order). See - // common/jobs/autoQueuePolicy.ts. - autoQueue: AutoQueueSettings; - // THE OPERATOR-FACING PRIORITY MODEL: one tier per channel plus one - // corpus-wide focus selector. It is the SOURCE the four `autoQueue[lane].root` - // trees are compiled from (common/lib/channelPriority.ts), not a second - // mechanism beside them — and its `paused` tier is the one part that is not a - // tree shape, filtering the runner's channel list instead. An empty document - // (the default) is today's behaviour exactly: no focus, every channel normal, - // the stored trees stand. - channelPriority: ChannelPriority; - // Default social links applied to every site that doesn't define its own. - // A site inherits these unless its site.json carries an explicit - // `socialLinks` array — see Site.socialLinks / resolveSocialLinks in - // common/lib/site.ts. The one presentation field that lives globally so a - // shared footer doesn't have to be repeated per site. - socialLinks: SocialLink[]; - // Absolute public URL of the family hub/homepage (e.g. "https://archilyzer.pages.dev"). - // Every export site links back to it ("the family" backlink) when set. Empty = - // no hub link rendered. Normalized to a trailing-slash-free http(s) URL. - homepageUrl: string; - // Backup configuration for the saved-video store (Phase 4 of the - // video-persistence feature). When enabled with a destination, the store is - // mirrored there (additively, no deletes) with a per-backup manifest, and the - // sync scheduler runs the backup on the configured cadence. See - // common/controller/backupSavedVideos.ts. - savedVideoBackup: SavedVideoBackupSettings; - // Where a channel's downloaded media goes when it is relocated off the corpus - // disk. A DEFAULT ONLY: the relocate controller never reads it and always - // takes an explicit root, so this is the value the per-channel Storage panel - // prefills and the /channels bulk move falls back to. Blank = no default. - // See StorageSettings. - storage: StorageSettings; - // How the static export is built: "basic" reuses the single export/ tree and - // serializes builds on one queue (the long-standing behavior); "docker" runs - // each site's build in an isolated container for safe parallelism. The Docker - // pipeline itself is a follow-up; this block persists the chosen mode plus the - // container/concurrency knobs the deploy page and the future orchestrator read. - buildPipeline: BuildPipelineSettings; - // AI digest generation (chapters + topic tags over the existing transcripts). - // Local-first: the metered lane is off by default. See DigestSettings. - digest: DigestSettings; - // Speaker diarization captured right after transcription, while the audio is - // still on disk. OFF by default. See DiarizationSettings. - diarization: DiarizationSettings; - // The generic catch-up lane for derived data the existing corpus predates. - // OFF by default, and idle-only when on. See BackfillSettings. - backfill: BackfillSettings; - // Naming the speakers diarization found (or reconstructing them from the - // transcript when it found none). OFF by default. See AttributionSettings. - attribution: AttributionSettings; -}; - -// Configuration for speaker attribution — putting names to the speaker turns. -// -// OFF by default, and that default is doing real work rather than being -// cautious. The text-only lane costs roughly one model call per transcript -// CHUNK, which on this corpus is ~194,000 calls, the same order as the digest -// sweep — and the digest sweep has completed 0.17% of its own. Arming both at -// once puts two multi-week lanes on the same 8 GB card with nothing to arbitrate -// between them (the backfill lane's yield deliberately watches only the -// transcription lane). Nothing here arms anything; a pilot decides whether the -// corpus-wide text-only pass is worth 25-55 GPU-days at all. -export type AttributionSettings = { - // Master switch. Off means the backfill registry reports no attribution work - // at all — the feature gate every Operation has. - enabled: boolean; - // Which digest app runs the naming. Attribution IS a digest-app workload — - // constrained JSON decoding over transcript text — so it reuses that registry - // and that per-app config (settings.digest.apps[appId]) rather than growing a - // second copy of the ollama URL, context size and timeout. - appId: string; - // Model override. Empty = the app's configured model, then its default. It is - // separate from the digest's because the two workloads may want different - // sizes, and because it is part of the freshness identity: sharing the digest's - // model field would make a digest bake-off invalidate every attribution record - // on disk as a side effect. - model: string; - // The lanes, separately. Both default OFF even when `enabled` is on, so - // turning the feature on to look at it cannot start a corpus sweep. - // - // They are not a fallback pair. `diarized` is one call per video and grounded - // in acoustic clustering; `textOnly` is ~30 calls and guesses at identity - // across chunk seams. An operator may reasonably want the first forever and - // the second never. - diarizedEnabled: boolean; - textOnlyEnabled: boolean; - // The prompt generation a record must match to count as fresh. - // - // Defaults to (and is floored at) ATTRIBUTION_PROMPT_VERSION, the shipped - // constant. Raising it forces a corpus-wide regeneration without a code - // change, which is the honest way to redo everything after a prompt tweak. - // It cannot be set BELOW the shipped constant, and that floor is the lesson - // from digestPrompt.ts's version 1 -> 2 note: pinning freshness to an older - // generation freezes output from a superseded prompt into the corpus, looking - // identical to output from the current one. - promptVersion: number; -}; - -// Configuration for the backfill lane — the generic answer to "a derived-data -// feature landed and 77,000 existing videos do not have it". -// -// WHAT THE LANE GETS is its own queueKey (concurrency with transcription) plus a -// limit() that returns 0 to stand aside — the same mechanism the digest yield -// uses, which fails OPEN so a bad read costs contention rather than a deadlock. -// There is no priority system to join: the registry submits every named queue at -// concurrency 1 and SchedulerTier only orders work within a single key. -// -// THE SHARE IS `concurrency` AND THE LANE'S `autoQueue.backfill.maxWorkers`; the -// yield is the operation's declared `contendsFor`. Slice 1.3 retired the -// `weight` scalar that used to mean both — see backfillLimit(). -export type BackfillSettings = { - // Slots the lane may use when it is not standing aside. Kept at 1 by default - // for the same reason diarization.concurrency is: this is CPU-bound work - // competing with GPU feeding and the digest sweep for the same 8 threads. - concurrency: number; - // Re-acquire media for videos whose input is GONE (audio deleted after - // transcription). OFF by default and deliberately so: measured on this corpus, - // 836 videos still have media and ~76,270 would need a re-download — 91x the - // reachable work, against 45 GB free at 97% full. When on, each re-fetched - // file is removed in a `finally` as soon as the backfill has used it, unless - // the video is marked do-not-clean, or unless the auto-transcribe policy would - // replace its auto-captions (`replaceAutoSubs`, or a leaf on - // `downloadedAutoSubsOnly`), in which case the audio is kept for that runner. - // - // WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: "youtube"` - // channel — which normally only fetches subtitles — the re-acquire applies a - // PER-VIDEO transcribe override so yt-dlp lands audio a diarizer can read; - // the channel's stored config is not changed. Without that override the fetch - // re-downloads the captions the video already has and lands nothing, which is - // what happened to ~16,000 videos on eight channels in 2026-08. - allowRedownload: boolean; -}; - -// Configuration for the speaker-diarization capture lane. -// -// This exists because AUDIO IS THE ONE PERISHABLE INPUT in the pipeline: -// cleanAudioFromTranscribed deletes it once a video is transcribed, so -// diarization has to happen while the audio is still there or not at all. The -// capture half is deliberately all that ships here — attribution, LLM speaker -// naming, viewer badges and quote filtering can all be redone later from the -// saved JSON, whereas the audio cannot. -export type DiarizationSettings = { - // Master switch. OFF by default so a transcription batch can start before this - // lands, with diarization backfilled over the retained audio afterwards. - // - // Turning it ON also arms the cleanup guard: the Clean-audio sweep stops - // deleting audio for a transcribed video that has no diarization.json yet. - // That is the point — it is what keeps the perishable input alive long enough - // to be captured — but it means enabling this holds disk. - enabled: boolean; - // Run diarization inline in the post-transcribe hook. - // - // OFF by default, and that default is a MEASURED decision, not caution. - // Measured on this box: GPU transcription runs at 221 s/audio-hour (16.3x - // realtime, over 3,602 real videos), CPU diarization at ~500-680 s/audio-hour. - // Diarization is therefore ~2-3x SLOWER than the transcription it follows, so - // running it inline drops whole-pipeline throughput by roughly 3-4x and leaves - // the GPU idle while the CPU catches up. - // - // The intended sequence for a large batch is the opposite: leave this off, let - // the batch transcribe at full GPU speed with `enabled` holding the audio, and - // diarize afterwards with the backfill pass. Turn it on for steady state, once - // the arrival rate is a few videos a day rather than a corpus. - inlineAfterTranscribe: boolean; - // Clustering threshold — the single most consequential knob, since it decides - // how many speakers come out. Larger merges more aggressively. - // - // The default is 0.9, NOT sherpa-onnx's own 0.5, and that is measured on this - // corpus. On a 6-minute excerpt of a two-person interview (known ground truth: - // 2 speakers), sherpa's default produced 22 clusters; 0.9 produced 6, with the - // top two at 40%/40% of talk time — recognizably the two hosts. Sweep on the - // same clip: 0.4→23, 0.5→22, 0.6→17, 0.7→12, 0.8→10, 0.9→6. - // - // It still over-splits, which is why this is a capture lane and not an answer: - // the turns are recorded with the threshold that produced them, so a later - // attribution pass can re-cluster or re-run without needing the audio back. - threshold: number; - // Engine threads per diarize run. - threads: number; - // Which engine runs. "sherpa-onnx" is the shipped default and what every - // sidecar on disk was produced by; "sortformer" is the ggml engine built by - // scripts/build-sortformer.sh. - // - // CHANGING THIS RESTATES THE FRESHNESS IDENTITY (see diarizationTarget), so - // every sidecar written by the other engine becomes stale and the backfill lane - // offers to redo it. That is intended — the two disagree about how many - // speakers exist, and a corpus half-diarized by each is not one corpus — but on - // the retained audio it is weeks of work, not a toggle. - // - // Why anyone would: on the same file, sherpa at its tuned threshold returns 13 - // speakers and sortformer returns 4, agreeing on the dominant speaker's share - // to within half a point (73.1% vs 73.5%). On the corpus's worst case sherpa - // returns 35 and sortformer 4. Over-splitting is the failure mode this lane has - // always had, and sortformer is end-to-end rather than clustered, so it does - // not have it. The cost is a hard ceiling of 4 speakers and ~1.8x the wall - // clock. - engine: DiarizationEngineId; - // Compute device for the sortformer engine; ignored by sherpa-onnx, which has - // no Vulkan compute path on Linux. - // - // "vulkan" is 1.5x faster than a thread-tuned CPU run (894 vs 1305 - // s/audio-hour, measured on this box) and holds 558 MB resident instead of - // 4.84 GB by keeping weights and activations in VRAM. It also takes ~4.4 GB of - // an 8 GB card, which is why the lane YIELDS to transcription rather than - // sharing — see controller/digestYield.ts. - backend: DiarizationBackend; - // Python interpreter for the default sherpa-onnx engine. sherpa-onnx ships - // wheels only up to cp313, and this box's system python is 3.14 — so this - // usually points at a dedicated venv rather than `python3`. - python: string; - // ONNX model paths for the default engine. Empty = the lane cannot run, which - // is reported as a skip rather than a failure. - segModel: string; - embModel: string; - // Binary and model for the sortformer engine, both produced by - // scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the - // same "not-configured" skip as an unset segModel/embModel. - sortformerBin: string; - sortformerModel: string; - // How many diarize runs may execute at once in the backfill pass. Kept low by - // default: diarization is CPU-bound and competes with GPU feeding and the - // digest sweep for the same 8 threads. - concurrency: number; - // Videos longer than this are DEFERRED rather than diarized: reported as a - // third number that is never summed into reachable work, so a capped corpus - // can never read as finished. - // - // THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering holds a - // pairwise distance matrix over speech-segment embeddings — O(n^2) in SEGMENT - // count — and speaker-turn density varies 40x across this corpus (33-1364 - // turns/hour), so duration does not actually predict the blowup: a sparse - // 7h42m video completed while a dense 6h12m one was OOM-killed. Duration is - // merely the only predictor available for free, from metadata already on disk, - // BEFORE spending 45 minutes to find out. n^2 at 30k segments is 6.7 GiB and - // at 40k is 11.9 GiB, which brackets the 10.6 GB and 9.6 GB peaks measured on - // this 16 GB box. - // - // 0 disables the cap. That is where this goes once windowed diarization lands: - // windowing divides per-window n by the window count, so the matrix falls by - // its square, and the cap stops being needed rather than being tuned. - maxAudioHours: number; -}; - -// Configuration for the derived-corpus digest layer. Local-first by decision: -// `remoteEnabled` gates the metered lane and defaults to false, so nothing here -// can spend money until it is explicitly turned on. -export type DigestSettings = { - // Master switch for the metered (remote-api) lane. OFF by default — an opt-in - // overflow for the long tail or a channel where local quality is poor, never - // the default path. - remoteEnabled: boolean; - // Videos longer than this are "long tail": 8.2% of the corpus by count, 46% of - // all transcript tokens. The batch's duration-aware ordering and the optional - // remote overflow both key off it. - longTailSeconds: number; - // The engine each lane uses (ids from common/lib/digestApps.ts). - localAppId: string; - remoteAppId: string; - // Per-app config, keyed by app id — the same id-keyed sub-record shape as - // transcriptionApps. - apps: Record<string, DigestAppConfig>; - // Yield the GPU to the transcription lane: while transcription is working, the - // digest batch's limit() returns 0 and the pool idle-waits. ON by default, - // because `digest:local` is deliberately on a different queue from - // TRANSCRIPTION_QUEUE and so would otherwise run ollama and the transcription - // engine on the same 8 GB card. See controller/digestYield.ts. - yieldToTranscription: boolean; - // Whether a busy worker pinned to `device: "cpu"` counts as GPU contention. - // - // OFF by default, which is the FIX for a real bug: the yield originally tested - // only `kind === "local"`, so on a box with one GPU worker and two CPU-pinned - // ones (this box, at parallelTranscriptions 2) the digest lane stopped dead for - // transcription that competes for zero GPU shaders. - // - // Only an EXPLICIT "cpu" is treated as non-contending. A worker with no device - // set is using the engine binary's own default, which may be the GPU, so it - // still triggers the yield — the unknown case fails safe. - // - // Composes with `yieldToTranscription`: that is the master switch, this only - // narrows which workers it reacts to. - yieldToCpuWorkers: boolean; - // Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. Only ever - // consulted for a metered app. - spendCapUsd: number; - // Which sections a sweep generates. - // - // Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, and that - // is not a contradiction: a tag call sends the same transcript as the chapter - // call before it, so it hits the engine's cached prefix and pays essentially no - // prefill (+0.4 s across 4 extra calls, against 22.4 s for the first 4). All it - // pays is decode, and a tag list is ~30 output tokens where a chapter list is - // ~200–290. - // - // The corollary matters more than the number: run them in the SAME pass. Tags - // generated later, on their own, pay full prefill again — measured at 44% of a - // whole chapters pass, i.e. 3–9× the marginal cost of just including them now. - sections: DigestSectionKind[]; - // How each chunk's transcript markers are numbered — see DigestTimestampMode. - // Was a scored variable in the bake-off rather than a pre-applied fix; the - // measurement is in and "chunk-local" is now the shipped default. - timestampMode: DigestTimestampMode; - // A free-text label for a non-default prompt shape, folded into the recorded - // provenance by digestPromptVariant(). Setting it invalidates every digest - // generated under a different label, which is exactly what makes a bake-off - // round re-run its sample instead of skipping it as fresh. Empty = default. - promptVariant: string; -}; - -// "basic" — `pnpm run build` in export/, serialized on the build queue (shared -// output tree → no safe parallelism). -// "docker" — isolated per-site container builds (follow-up); enables real -// parallel multi-site builds capped by maxParallelBuilds. -export type BuildMode = "basic" | "docker"; - -export type BuildPipelineSettings = { - mode: BuildMode; - // Cap on concurrent per-site container builds in docker mode. Ignored in basic - // mode (which is always serial). Clamped to [1, BUILD_MAX_PARALLEL_MAX]. - maxParallelBuilds: number; - // Tag of the reusable build image (built once, reused for every site). - dockerImage: string; - // Dockerfile path relative to the monorepo root, used to (re)build the image. - dockerfile: string; -}; - -export type SavedVideoBackupSettings = { - // Master switch for the scheduled backup. A backup can still be run manually - // when this is false, as long as a destination is set. - enabled: boolean; - // Destination root the store is mirrored into (a local path or any rsync - // target). Empty disables both scheduled and manual backups. - dest: string; - // Cadence (minutes) for the scheduled backup when enabled. Clamped into the - // sync-interval window; default daily. - intervalMinutes: number; -}; - -export type SyncSchedulerSettings = { - // Master switch. When false, a tick selects nothing (manual sync still works). - enabled: boolean; - // Fallback cadence (minutes) for channels with no per-channel override. - defaultIntervalMinutes: number; - // Cap on sync jobs running/queued at once. A tick queues at most - // (cap - currently-active) channels; the rest roll to the next tick. This is - // also the stagger mechanism that keeps a big due-batch from hitting the - // source all at once. - maxConcurrentSyncs: number; - // Optional local-clock quiet window during which auto-sync is suppressed. - // Both null = always allowed. The window may wrap past midnight - // (e.g. start=22, end=6). Hours are [0,23]; the window is [start, end). - quietHoursStart: number | null; - quietHoursEnd: number | null; - // Failure backoff bounds. After N consecutive failed scheduled syncs a - // channel waits min(base * 2^(N-1), max) minutes before it's eligible again. - backoffBaseMinutes: number; - backoffMaxMinutes: number; - // Cadence (seconds) for the editor's in-process heartbeat — the internal timer - // armed by the instrumentation hook (editor/instrumentation.ts) that calls the - // scheduler tick directly, so no external cron is needed. 0 = off: rely on the - // external `pnpm sync:tick` heartbeat instead. Any positive value is clamped to - // [SYNC_HEARTBEAT_MIN_SECONDS, SYNC_HEARTBEAT_MAX_SECONDS]. The env var - // SYNC_HEARTBEAT_SECONDS overrides this at runtime. See SCHEDULED_SYNC.md. - heartbeatSeconds: number; - // Cadence (minutes) for the scheduled keep-latest deletion check. For each - // channel with ChannelConfig.keepLatest > 0, the tick re-probes the kept - // window for source deletion (checkKeptDeletedAction) at most this often and - // pins any gone videos as do-not-clean. Clamped into the sync-interval window; - // default daily. The check shares the same concurrency cap and quiet-hours - // window as scheduled syncs. See editor/app/scheduler/runTick.ts. - keepLatestCheckIntervalMinutes: number; - // Default cadence (minutes) for the sync FULL SWEEP — the deep pass that - // re-enumerates a channel's whole listing in one yt-dlp spawn, refreshes the - // stored `playlist` file, and flags videos that have left the listing into - // maybe-missing.json. Ordinary syncs stay on the cheap newest-first paged - // walk; a sync only upgrades itself to a sweep when this interval has elapsed - // since the channel's lastFullSweepAt. Per-channel override: - // ChannelConfig.fullSweepIntervalMinutes. 0 = never sweep. Default daily. - // See common/jobs/deepSync.ts. - fullSweepIntervalMinutes: number; - // Upper bound on how many maybe-missing suspects a full sweep will resolve - // in-line with the per-video availability probe (deleted vs private vs - // unlisted). At or under the cap the sweep runs the targeted check itself, so - // "Sync all" surfaces upstream deletions with no extra clicks; over it, the - // suspects are flagged and left for a manual check rather than firing hundreds - // of probes inside a sync. 0 = never auto-confirm. - fullSweepConfirmMaxSuspects: number; - // Shrink guard: how far a fresh listing may fall below the stored one before - // it is treated as suspect rather than acted on. Expressed as a percentage of - // the previous count, floored at SHRINK_ABS_FLOOR entries so ordinary churn on - // a small channel doesn't trip it. A suspect listing does not rewrite - // `playlist` or maybe-missing.json and does not count as a sweep — but a - // SECOND enumeration reporting a similar count confirms it and is accepted, so - // a genuine mass deletion costs at most one cadence period. 0 = off (the - // empty-listing rejection still applies). See controller/acceptListing.ts. - fullSweepShrinkGuardPercent: number; -}; - -export type SocialLink = { - label: string; - url: string; - svg: string; -}; - -export const SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS = 600; -export const SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS = 10; - -export const MIN_FREE_DISK_GB_DEFAULT = 5; -export const MIN_FREE_DISK_GB_MAX = 100000; - -// Hysteresis margin for the low-disk gate. 2 GB is deliberately larger than any -// single scratch file the pipeline writes, so cleaning one up cannot by itself -// reopen the gate. -export const RESUME_MARGIN_GB_DEFAULT = 2; -export const RESUME_MARGIN_GB_MAX = 1000; - -export const PARALLEL_TRANSCRIPTIONS_MAX = 16; -export const PARALLEL_TRANSCRIPTIONS_DEFAULT = 2; - -// Passive UI auto-refresh cadence. 0 is a sentinel meaning "disabled"; any other -// value is clamped into [MIN, MAX] seconds. -export const AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS = 5; -export const AUTO_REFRESH_INTERVAL_MIN_SECONDS = 1; -export const AUTO_REFRESH_INTERVAL_MAX_SECONDS = 600; - -// Global snapshot-scheduler debounce presets. `debounceMs` is the quiet-period -// window after the last report-changing action; `maxWaitMs` caps the total -// delay under continuous activity (null = no cap, fire purely on the quiet -// period). Consumed by common/jobs/snapshotScheduler.ts and surfaced in the -// Settings form. -export type ReportDebouncePreset = "fast" | "balanced" | "lazy"; - -export const REPORT_DEBOUNCE_PRESETS: Record< - ReportDebouncePreset, - { debounceMs: number; maxWaitMs: number | null } -> = { - fast: { debounceMs: 1000, maxWaitMs: null }, - balanced: { debounceMs: 3000, maxWaitMs: 30000 }, - lazy: { debounceMs: 10000, maxWaitMs: 60000 }, -}; - -export const DEFAULT_REPORT_DEBOUNCE_PRESET: ReportDebouncePreset = "fast"; - -export function isReportDebouncePreset(v: unknown): v is ReportDebouncePreset { - return v === "fast" || v === "balanced" || v === "lazy"; -} - -export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024; -export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024; -export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024; - -export const DEFAULT_ADMIN_TITLE = "Transcript Browser Admin"; - -// Sync-scheduler bounds + defaults. Default cadence is daily; concurrency is -// conservative so a tick doesn't fan out into the source provider all at once. -export const SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES = 1440; -export const SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT = 2; -export const SYNC_SCHEDULER_MAX_CONCURRENT_MAX = 16; -export const SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES = 30; -export const SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES = 1440; -export const KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES = 1440; -// Full-sweep defaults. Daily: a sweep is one full enumeration of the channel, -// far more expensive than the 50-entry page an ordinary sync fetches. The -// confirm cap keeps an unattended sweep from fanning out into hundreds of -// per-video probes when a channel's listing changes wholesale. -export const FULL_SWEEP_DEFAULT_INTERVAL_MINUTES = 1440; -export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT = 25; -export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX = 10000; -// Shrink-guard default: a listing that has lost more than a tenth of its -// entries (and more than SHRINK_ABS_FLOOR of them) needs a second opinion. -export const FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT = 10; -export const FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX = 100; -export const SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES = 1440; - -// Internal-heartbeat cadence bounds. 0 means "off" (use an external cron -// heartbeat); any other value is clamped into [MIN, MAX] seconds. The floor -// keeps the in-process timer from busy-looping; the ceiling is one hour. -export const SYNC_HEARTBEAT_DEFAULT_SECONDS = 0; -export const SYNC_HEARTBEAT_MIN_SECONDS = 15; -export const SYNC_HEARTBEAT_MAX_SECONDS = 3600; - -export function defaultSyncScheduler(): SyncSchedulerSettings { - return { - enabled: false, - defaultIntervalMinutes: SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES, - maxConcurrentSyncs: SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT, - quietHoursStart: null, - quietHoursEnd: null, - backoffBaseMinutes: SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES, - backoffMaxMinutes: SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES, - heartbeatSeconds: SYNC_HEARTBEAT_DEFAULT_SECONDS, - keepLatestCheckIntervalMinutes: KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES, - fullSweepIntervalMinutes: FULL_SWEEP_DEFAULT_INTERVAL_MINUTES, - fullSweepConfirmMaxSuspects: FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT, - fullSweepShrinkGuardPercent: FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT, - }; -} - -// Clamp an internal-heartbeat cadence: 0 (off) passes through; any positive -// value is clamped up into [MIN, MAX]; junk falls back to the default. -export function clampHeartbeatSeconds(value: unknown): number { - if (typeof value !== "number" || !Number.isFinite(value)) { - return SYNC_HEARTBEAT_DEFAULT_SECONDS; + clampParallelTranscriptions, + defaultStorage, + normalizeSocialSvg, + parseSocialLinks, + sanitizeTranscriptionApps, + siteSettingsSchema, + type SiteSettings, + type SocialLink, +} from "./settingsSchema"; + +export * from "./settingsSchema"; + +type RawSettings = Record<string, unknown>; + +// The file as JSON, or `undefined` when it is missing or not JSON. Never +// throws: a settings read is on every request path. +function readRawSettings(file: string): unknown { + try { + return JSON.parse(fs.readFileSync(file, "utf8")); + } catch { + return undefined; } - const n = Math.floor(value); - if (n <= 0) return 0; - if (n < SYNC_HEARTBEAT_MIN_SECONDS) return SYNC_HEARTBEAT_MIN_SECONDS; - if (n > SYNC_HEARTBEAT_MAX_SECONDS) return SYNC_HEARTBEAT_MAX_SECONDS; - return n; -} - -function clampHourOrNull(value: unknown): number | null { - if (typeof value !== "number" || !Number.isFinite(value)) return null; - const n = Math.floor(value); - if (n < 0 || n > 23) return null; - return n; -} - -// Like clampPositiveInt, but 0 survives as a sentinel ("off"/"never"). Used by -// the cadences whose disabled state is expressed as a zero rather than a -// separate boolean. -function clampIntAllowZero(value: unknown, fallback: number, max: number): number { - const n = - typeof value === "number" && Number.isFinite(value) - ? Math.floor(value) - : fallback; - if (n <= 0) return 0; - if (n > max) return max; - return n; } -function clampPositiveInt(value: unknown, fallback: number, max: number): number { - const n = - typeof value === "number" && Number.isFinite(value) - ? Math.floor(value) - : fallback; - if (n < 1) return 1; - if (n > max) return max; - return n; +// Only a plain object is a settings file. `null`, `[]`, `3` and a truncated +// write all read as the empty file — every field its default — rather than as +// an exception. (Before slice 4a a file containing `null` threw here.) +function rawObject(raw: unknown): RawSettings { + return raw && typeof raw === "object" && !Array.isArray(raw) + ? (raw as RawSettings) + : {}; } -// Coerce a raw settings.syncScheduler value into a clean SyncSchedulerSettings, -// falling back to defaults for missing/ill-typed fields. Quiet hours are only -// honored when BOTH endpoints are valid hours; otherwise the window is cleared. -export function sanitizeSyncScheduler(value: unknown): SyncSchedulerSettings { - const d = defaultSyncScheduler(); - if (!value || typeof value !== "object") return d; - const r = value as Record<string, unknown>; - const start = clampHourOrNull(r.quietHoursStart); - const end = clampHourOrNull(r.quietHoursEnd); - const backoffBase = clampPositiveInt( - r.backoffBaseMinutes, - d.backoffBaseMinutes, - SYNC_INTERVAL_MAX_MINUTES, - ); - return { - enabled: r.enabled === true, - defaultIntervalMinutes: clampPositiveInt( - r.defaultIntervalMinutes, - d.defaultIntervalMinutes, - SYNC_INTERVAL_MAX_MINUTES, - ), - maxConcurrentSyncs: clampPositiveInt( - r.maxConcurrentSyncs, - d.maxConcurrentSyncs, - SYNC_SCHEDULER_MAX_CONCURRENT_MAX, - ), - quietHoursStart: start !== null && end !== null ? start : null, - quietHoursEnd: start !== null && end !== null ? end : null, - backoffBaseMinutes: backoffBase, - // Cap can't sit below the base, or backoff would never grow. - backoffMaxMinutes: Math.max( - backoffBase, - clampPositiveInt( - r.backoffMaxMinutes, - d.backoffMaxMinutes, - SYNC_INTERVAL_MAX_MINUTES, - ), - ), - heartbeatSeconds: clampHeartbeatSeconds(r.heartbeatSeconds), - keepLatestCheckIntervalMinutes: clampPositiveInt( - r.keepLatestCheckIntervalMinutes, - d.keepLatestCheckIntervalMinutes, - SYNC_INTERVAL_MAX_MINUTES, - ), - fullSweepIntervalMinutes: clampIntAllowZero( - r.fullSweepIntervalMinutes, - d.fullSweepIntervalMinutes, - SYNC_INTERVAL_MAX_MINUTES, - ), - fullSweepConfirmMaxSuspects: clampIntAllowZero( - r.fullSweepConfirmMaxSuspects, - d.fullSweepConfirmMaxSuspects, - FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX, - ), - fullSweepShrinkGuardPercent: clampIntAllowZero( - r.fullSweepShrinkGuardPercent, - d.fullSweepShrinkGuardPercent, - FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX, - ), - }; -} - -export function defaultSavedVideoBackup(): SavedVideoBackupSettings { - return { - enabled: false, - dest: "", - intervalMinutes: SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES, - }; -} - -// Coerce a raw settings.savedVideoBackup value into a clean -// SavedVideoBackupSettings. A missing destination forces enabled off, since a -// backup with nowhere to go is meaningless. -export function sanitizeSavedVideoBackup( - value: unknown, -): SavedVideoBackupSettings { - const d = defaultSavedVideoBackup(); - if (!value || typeof value !== "object") return d; - const r = value as Record<string, unknown>; - const dest = typeof r.dest === "string" ? r.dest.trim() : ""; - return { - enabled: dest !== "" && r.enabled === true, - dest, - intervalMinutes: clampPositiveInt( - r.intervalMinutes, - d.intervalMinutes, - SYNC_INTERVAL_MAX_MINUTES, - ), - }; -} - -// Where relocated channel media goes: the named locations. -// -// This used to be ONE FIELD, `mediaRoot` — a single absolute string, the cold -// drive, typed once. It grew into a list of entities because a root alone -// cannot answer the two questions the operator actually has: is that disk here, -// and if it came up somewhere else, how do I point the channels at it without -// ssh and hand edits? A location carries an id, a label, the root, an opt-in -// `autoRepoint`, and the volume identity learned at its last probe. -// -// Still NOT a policy: a channel on a location is not thereby deprioritized, and -// nothing auto-relocates anything because a location exists. +// THE TWO ABSENCE-KEYED MIGRATIONS THAT REWRITE AN INPUT BLOCK, applied to the +// raw object BEFORE the parse. Each is handed the raw file, never a parsed one, +// because "the file does not spell this key" is the whole trigger. // -// AVAILABILITY IS NEVER STORED HERE. A refresh that wrote "available" would -// rewrite settings.json — and so bump the pulse revision — every few seconds. -// The probe (common/lib/storageVolumes.ts) is computed per request; only the -// `volume` identity is ever written back, and only when it changed. +// - autoQueue: `migrateSweepsToLanes` fills `autoQueue.digest` / `.backfill` +// from the retired sweep fields when — and only when — the file does not +// already carry that lane. It never enables a lane the sweep flag did not. +// See lib/laneMigration.ts. +// - storage: `storage.mediaRoot` (one absolute string) becomes a one-entry +// location list when the file has no `locations` key. An absent `storage` +// block is the default block, which has a `locations` key and so does not +// migrate. See lib/storageLocations.ts. // -// The types live in lib/storageLocations.ts, which is pure: a `"use client"` -// file may import them, and must not reach storageVolumes.ts (execa). -export type { StorageLocation, StorageVolume, StorageSettings }; - -export function defaultStorage(): StorageSettings { - return { locations: [], defaultLocationId: "" }; -} - -const LOCATION_ID_RE = /^[a-z0-9][a-z0-9-]{0,63}$/; - -// "internal" IS TAKEN. It is the synthetic /storage row for the corpus volume -// (INTERNAL_LOCATION_ID), and the regex above admits it — so a hand-edited -// settings.json, or an operator typing the obvious word into the New location -// form, could store a real location under the one id the page assembles for -// itself. The row would then be built twice, the rollup would count channels -// into whichever assembled last, and `locationOfDataDir` would start matching -// unrelocated channels against it. -function isReservedLocationId(id: string): boolean { - return id === INTERNAL_LOCATION_ID; -} - -function sanitizeVolume(value: unknown): StorageVolume | undefined { - if (!value || typeof value !== "object") return undefined; - const v = value as Record<string, unknown>; - const uuid = typeof v.uuid === "string" ? v.uuid.trim() : ""; - const mountpoint = - typeof v.mountpoint === "string" ? v.mountpoint.trim() : ""; - // No uuid is no identity, and no mountpoint means `root === join(mountpoint, - // relPath)` cannot hold — either way the record is not usable for finding the - // volume again, so it is dropped rather than half-kept. - if (!uuid || !mountpoint) return undefined; - const relPath = typeof v.relPath === "string" ? v.relPath.trim() : ""; - const fstype = typeof v.fstype === "string" ? v.fstype.trim() : ""; - const label = typeof v.label === "string" ? v.label.trim() : ""; +// Both return RAW values; the schema's sanitizers then normalize them exactly as +// they normalize a hand-edited file. +function premigrateRaw(raw: RawSettings): RawSettings { return { - uuid, - ...(fstype ? { fstype } : {}), - ...(label ? { label } : {}), - mountpoint, - relPath, + ...raw, + autoQueue: migrateSweepsToLanes(raw), + storage: migrateMediaRootToLocations(raw.storage ?? defaultStorage()), }; } -// Coerce a raw settings.storage value into a clean StorageSettings. -// -// EXISTENCE IS NOT CHECKED, on purpose: the whole point of a cold location is -// that it is a drive that may not be mounted when settings are read, and a -// sanitizer that dropped the root on an unmounted platter would silently erase -// the operator's choice on the next save. +// THE TWO ABSENCE-KEYED MIGRATIONS THAT NEED THE PARSED RESULT, in this order: // -// ABSOLUTENESS *IS* checked, and a location with a relative root is DROPPED -// rather than resolved. Resolving it would anchor the location to whatever cwd -// the reader booted in — a different directory under docker, under a worktree, -// and under `pnpm dev` — so the same settings.json would name three different -// drives. The location form rejects a relative path with a message before it -// ever gets here; this is the last line, not the only one. -// -// NESTED ROOTS ARE ALLOWED. "/mnt/platter" and "/mnt/platter/archive" may both -// be locations; `locationOfDataDir` resolves a channel to the LONGEST matching -// root. Nothing here rejects the nesting, because the operator who arranges a -// disk that way means it. -// -// A STALE `mediaRoot` SITTING BESIDE `locations` IS IGNORED — it is not merged -// back in as an extra location. `migrateMediaRootToLocations` reads it exactly -// once, when `locations` is absent; after that the list is the whole truth, and -// resurrecting a root the operator deleted would be a bug, not a kindness. -// -// ROLLBACK: an older binary sanitizes this block to `{ mediaRoot: "" }` — the -// locations are dropped and the single cold root comes back blank. One string -// lost, nothing on disk moved. `cp settings.json settings.json.pre-storage- -// locations` before the upgrade and a downgrade is a file copy. -export function sanitizeStorage(value: unknown): StorageSettings { - const d = defaultStorage(); - if (!value || typeof value !== "object") return d; - const r = value as Record<string, unknown>; - const rawList = Array.isArray(r.locations) ? r.locations : []; - const locations: StorageLocation[] = []; - const seen = new Set<string>(); - for (const entry of rawList) { - if (!entry || typeof entry !== "object") continue; - const e = entry as Record<string, unknown>; - const id = typeof e.id === "string" ? e.id.trim() : ""; - if (!LOCATION_ID_RE.test(id) || isReservedLocationId(id) || seen.has(id)) { - continue; - } - const rawRoot = typeof e.root === "string" ? e.root.trim() : ""; - if (!path.isAbsolute(rawRoot)) continue; - // "/mnt/platter/" and "/mnt/platter" are one root; "/" stays "/". - const stripped = rawRoot.replace(/\/+$/, ""); - const root = stripped === "" ? "/" : stripped; - const label = typeof e.label === "string" ? e.label.trim() : ""; - const volume = sanitizeVolume(e.volume); - seen.add(id); - locations.push({ - id, - label: label || id, - root, - autoRepoint: e.autoRepoint === true, - ...(volume ? { volume } : {}), - }); - } - const wanted = - typeof r.defaultLocationId === "string" ? r.defaultLocationId.trim() : ""; - // A default naming a location that is gone falls back to the first one, not - // to "": with a location configured, "no default" is never the answer the - // operator wanted, and a blank default silently disables every prefill. - const defaultLocationId = locations.some((l) => l.id === wanted) - ? wanted - : (locations[0]?.id ?? ""); - // THE SAVED-VIDEO STORE'S LOCATION IS NOT FALLEN BACK, and the asymmetry - // with `defaultLocationId` above is deliberate. That one is a PREFERENCE, so - // picking another location when the named one is gone is helpful. This one is - // a RECORD OF WHERE BYTES ARE: pointing it at a different location because - // the recorded one was deleted would claim the store had moved when nothing - // had. A dangling id sanitizes to "" — "in place" — which is what the disk - // says as soon as anybody looks, and the symlink (if any) keeps working - // regardless, because the store is reached through it and not through this. - const savedWanted = - typeof r.savedVideosLocationId === "string" - ? r.savedVideosLocationId.trim() - : ""; - const savedVideosLocationId = locations.some((l) => l.id === savedWanted) - ? savedWanted - : ""; - return { - locations, - defaultLocationId, - ...(savedVideosLocationId ? { savedVideosLocationId } : {}), - }; -} - -// 4 hours. Measured: videos over this are 8.2% of the corpus by count but hold -// 46% of all transcript tokens, so they are where a sweep's wall-clock actually -// goes and where chunk-seam bugs live. -export const DIGEST_LONG_TAIL_DEFAULT_SECONDS = 4 * 3600; -export const DIGEST_LONG_TAIL_MAX_SECONDS = 24 * 3600; - -export function defaultDigest(): DigestSettings { - return { - // OFF. The metered lane is built but never the default — see PLAN.md. - remoteEnabled: false, - longTailSeconds: DIGEST_LONG_TAIL_DEFAULT_SECONDS, - localAppId: DEFAULT_DIGEST_APP_ID, - remoteAppId: CLAUDE_DIGEST_APP_ID, - // Empty on purpose: every per-app knob falls through to its own default - // constant (resolveNumCtx -> DEFAULT_DIGEST_NUM_CTX, now 8192, and - // maxCuesForContext sizes the chunk to it). Seeding a copy of those values - // here would give the same number two homes and let them drift. - apps: {}, - // ON. Real GPU contention with the transcription engine is a genuine cost - // (re-priced: 11.2 s/chunk idle against 24.9 s/chunk on a contended box), so - // the safe default is to step aside; turning it off is the deliberate choice. - yieldToTranscription: true, - // OFF. A CPU-pinned worker is not GPU contention, and treating it as such - // stalled the digest lane for nothing. See DigestSettings.yieldToCpuWorkers. - yieldToCpuWorkers: false, - spendCapUsd: 0, - sections: ["chapters"], - timestampMode: DEFAULT_DIGEST_TIMESTAMP_MODE, - promptVariant: "", - }; -} - -// Coerce a raw settings.digest.apps value into a clean keyed map of -// DigestAppConfig. Mirrors sanitizeTranscriptionApps — INCLUDING its -// Array.isArray guard, without which a JSON array would pass the typeof check and -// produce numeric-keyed garbage. -export function sanitizeDigestApps( - value: unknown, -): Record<string, DigestAppConfig> { - if (!value || typeof value !== "object" || Array.isArray(value)) return {}; - const out: Record<string, DigestAppConfig> = {}; - for (const [id, raw] of Object.entries(value as Record<string, unknown>)) { - if (!raw || typeof raw !== "object") continue; - const r = raw as Record<string, unknown>; - const cfg: DigestAppConfig = {}; - if (typeof r.bin === "string" && r.bin.trim()) cfg.bin = r.bin.trim(); - if (typeof r.baseUrl === "string" && r.baseUrl.trim()) { - cfg.baseUrl = r.baseUrl.trim(); - } - if (typeof r.model === "string" && r.model.trim()) cfg.model = r.model.trim(); - if (typeof r.numCtx === "number" && r.numCtx > 0) { - cfg.numCtx = Math.floor(r.numCtx); - } - if (typeof r.temperature === "number" && r.temperature >= 0) { - cfg.temperature = r.temperature; - } - if (typeof r.timeoutMs === "number" && r.timeoutMs > 0) { - cfg.timeoutMs = Math.floor(r.timeoutMs); - } - // Only carried when explicitly set — see DigestAppConfig.think. - if (typeof r.think === "boolean") cfg.think = r.think; - out[id] = cfg; - } - return out; -} - -export function sanitizeDigest(value: unknown): DigestSettings { - const d = defaultDigest(); - if (!value || typeof value !== "object") return d; - const r = value as Record<string, unknown>; - const sections = Array.isArray(r.sections) - ? (r.sections.filter(isDigestSectionKind) as DigestSectionKind[]) - : []; - return { - remoteEnabled: r.remoteEnabled === true, - longTailSeconds: clampPositiveInt( - r.longTailSeconds, - d.longTailSeconds, - DIGEST_LONG_TAIL_MAX_SECONDS, - ), - // Unknown app ids are not rejected here: getDigestApp() is total and falls - // back to the local default, so a stale id degrades rather than breaking. - localAppId: - typeof r.localAppId === "string" && r.localAppId.trim() - ? r.localAppId.trim() - : d.localAppId, - remoteAppId: - typeof r.remoteAppId === "string" && r.remoteAppId.trim() - ? r.remoteAppId.trim() - : d.remoteAppId, - apps: sanitizeDigestApps(r.apps), - // Defaults to ON when absent — `=== false` rather than `!== true`, so a - // settings file written before this field existed keeps the GPU-safe - // behaviour instead of silently opting into contention. - yieldToTranscription: r.yieldToTranscription !== false, - // The OPPOSITE idiom, and deliberately so: `=== true`, so absence falls to - // OFF. The field's absence means a settings file written before the CPU-worker - // bug was found, and for those files OFF is the FIXED behaviour, not a silent - // change of intent — nobody ever asked to stall the digest lane for a CPU - // transcription. `yieldToTranscription` still gates the whole thing, so the - // GPU-safe default is untouched. - yieldToCpuWorkers: r.yieldToCpuWorkers === true, - spendCapUsd: - typeof r.spendCapUsd === "number" && r.spendCapUsd > 0 - ? Math.round(r.spendCapUsd * 100) / 100 - : 0, - // An empty/garbage list would silently generate nothing, so fall back to the - // default rather than honoring it. - sections: sections.length > 0 ? sections : d.sections, - timestampMode: isDigestTimestampMode(r.timestampMode) - ? r.timestampMode - : d.timestampMode, - // Trimmed and length-capped: it goes into provenance on every record, and a - // runaway value would bloat 119k sidecars. - promptVariant: - typeof r.promptVariant === "string" - ? r.promptVariant.trim().slice(0, 40) - : d.promptVariant, - }; -} - -// Every known section kind, for the settings UI's checkbox list. -export const DIGEST_SECTION_OPTIONS = DIGEST_SECTION_KINDS; -export const DIGEST_TIMESTAMP_MODE_OPTIONS = DIGEST_TIMESTAMP_MODES; - -export const BUILD_MAX_PARALLEL_DEFAULT = 2; -export const BUILD_MAX_PARALLEL_MAX = 16; -export const DEFAULT_BUILD_IMAGE = "yt-dlp-transcript-browser-build"; -export const DEFAULT_BUILD_DOCKERFILE = "Dockerfile.build"; - -export function isBuildMode(v: unknown): v is BuildMode { - return v === "basic" || v === "docker"; -} - -export function defaultBuildPipeline(): BuildPipelineSettings { - return { - mode: "basic", - maxParallelBuilds: BUILD_MAX_PARALLEL_DEFAULT, - dockerImage: DEFAULT_BUILD_IMAGE, - dockerfile: DEFAULT_BUILD_DOCKERFILE, - }; -} - -// Coerce a raw settings.buildPipeline value into a clean BuildPipelineSettings, -// falling back to defaults for missing/ill-typed fields. -export function sanitizeBuildPipeline(value: unknown): BuildPipelineSettings { - const d = defaultBuildPipeline(); - if (!value || typeof value !== "object") return d; - const r = value as Record<string, unknown>; - const dockerImage = - typeof r.dockerImage === "string" && r.dockerImage.trim() - ? r.dockerImage.trim() - : d.dockerImage; - const dockerfile = - typeof r.dockerfile === "string" && r.dockerfile.trim() - ? r.dockerfile.trim() - : d.dockerfile; - return { - mode: isBuildMode(r.mode) ? r.mode : d.mode, - maxParallelBuilds: clampPositiveInt( - r.maxParallelBuilds, - d.maxParallelBuilds, - BUILD_MAX_PARALLEL_MAX, - ), - dockerImage, - dockerfile, - }; -} - -function defaults(): SiteSettings { - return { - adminTitle: DEFAULT_ADMIN_TITLE, - maxTranscriptPageBytes: TRANSCRIPT_PAGE_DEFAULT_BYTES, - transcriptionApp: DEFAULT_TRANSCRIPTION_APP_ID, - transcriptionApps: {}, - workers: [], - cookiesFromBrowser: "", - cookieMode: DEFAULT_COOKIE_MODE, - sleepBetweenDownloadsSeconds: SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS, - downloadFormat: "auto", - minFreeDiskGB: MIN_FREE_DISK_GB_DEFAULT, - resumeMarginGB: RESUME_MARGIN_GB_DEFAULT, - parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT, - inlineTranscribeOnFallback: false, - skipLiveDownloads: true, - verifyAvailabilityBeforeClean: true, - buildArchives: true, - archiveStorage: { bucket: "", publicBaseUrl: "" }, - reportDebouncePreset: DEFAULT_REPORT_DEBOUNCE_PRESET, - autoRefreshIntervalSeconds: AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS, - syncScheduler: defaultSyncScheduler(), - autoQueue: defaultAutoQueue(), - channelPriority: defaultChannelPriority(), - socialLinks: [], - homepageUrl: "", - savedVideoBackup: defaultSavedVideoBackup(), - storage: defaultStorage(), - buildPipeline: defaultBuildPipeline(), - // Must be listed here or the allowlist loop in getSettings() drops the key - // entirely and the whole section is never read from disk. - digest: defaultDigest(), - diarization: defaultDiarization(), - backfill: defaultBackfill(), - attribution: defaultAttribution(), - }; -} - -// The whole default settings object, without touching disk. Exported so a test -// (or any caller that needs a settings-SHAPED value rather than the operator's -// actual configuration) can build one without a settings.json. -export function defaultSiteSettings(): SiteSettings { - return defaults(); -} - -export function defaultBackfill(): BackfillSettings { - return { - concurrency: 1, - // See BackfillSettings.allowRedownload — this one holds disk. - allowRedownload: false, - }; -} - -export function sanitizeBackfill(value: unknown): BackfillSettings { - const d = defaultBackfill(); - if (!value || typeof value !== "object") return d; - const r = value as Record<string, unknown>; - return { - // Clamped rather than rejected: a hand-edited 5 means "as much as possible", - // and reading it as 0 would be the opposite of the intent. - concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16), - allowRedownload: r.allowRedownload === true, - }; -} - -export function defaultAttribution(): AttributionSettings { - return { - // OFF, and both lanes OFF under it. See AttributionSettings. - enabled: false, - appId: DEFAULT_DIGEST_APP_ID, - model: "", - diarizedEnabled: false, - textOnlyEnabled: false, - promptVersion: ATTRIBUTION_PROMPT_VERSION, - }; -} - -export function sanitizeAttribution(value: unknown): AttributionSettings { - const d = defaultAttribution(); - if (!value || typeof value !== "object") return d; - const r = value as Record<string, unknown>; - const str = (v: unknown, fallback: string) => - typeof v === "string" && v.trim() ? v.trim() : fallback; - return { - enabled: r.enabled === true, - appId: str(r.appId, d.appId), - // NOT `str(...)` with a fallback: empty is a meaningful value here ("use the - // app's own model"), so an empty string must survive rather than reverting - // to a default that is also empty by coincidence. - model: typeof r.model === "string" ? r.model.trim() : d.model, - diarizedEnabled: r.diarizedEnabled === true, - textOnlyEnabled: r.textOnlyEnabled === true, - // FLOORED at the shipped constant, never merely defaulted. A hand-edited - // value below it would pin freshness to a superseded prompt generation and - // freeze its output into the corpus — see AttributionSettings.promptVersion. - promptVersion: - typeof r.promptVersion === "number" && Number.isFinite(r.promptVersion) - ? Math.max(ATTRIBUTION_PROMPT_VERSION, Math.floor(r.promptVersion)) - : d.promptVersion, - }; -} - -export function defaultDiarization(): DiarizationSettings { - return { - // OFF. Capture is opt-in: turning it on makes the cleanup sweep start - // refusing to delete audio for transcribed-but-undiarized videos, which is - // correct but is a disk-pressure decision an operator should make. - enabled: false, - // OFF — see DiarizationSettings.inlineAfterTranscribe. Diarization is slower - // than the transcription it would follow, so inline is the exception. - inlineAfterTranscribe: false, - // Measured, not sherpa-onnx's 0.5 — see DiarizationSettings.threshold. The - // constant lives in lib/diarization.ts because isDiarizationFresh needs it - // to normalize an absent recorded threshold; importing it keeps the default - // and the comparator from drifting apart. - threshold: DEFAULT_DIARIZATION_THRESHOLD, - threads: 4, - // The engine every sidecar on disk was produced by. Switching is an explicit - // decision that restates the freshness identity — see DiarizationSettings. - engine: DEFAULT_DIARIZATION_ENGINE, - // Only consulted when engine is "sortformer". Defaulting to the GPU is safe - // because the lane yields the card to transcription rather than sharing it. - backend: "vulkan", - python: "python3", - segModel: "", - embModel: "", - sortformerBin: "", - sortformerModel: "", - concurrency: 1, - // OFF, because windowing made it unnecessary — which is what it was always - // for. It shipped at 4 hours as a stopgap while long recordings were being - // OOM-killed; the engine now processes them in windows and the 6h12m file - // that was being killed at 10.5 GB completes at ~2 GB. The knob stays, and - // stays honest about what it does, for a machine smaller than this one or a - // recording longer than anything measured here. - maxAudioHours: 0, - }; -} - -export function sanitizeDiarization(value: unknown): DiarizationSettings { - const d = defaultDiarization(); - if (!value || typeof value !== "object") return d; - const r = value as Record<string, unknown>; - const str = (v: unknown, fallback: string) => - typeof v === "string" && v.trim() ? v.trim() : fallback; - return { - enabled: r.enabled === true, - inlineAfterTranscribe: r.inlineAfterTranscribe === true, - threshold: - typeof r.threshold === "number" && - Number.isFinite(r.threshold) && - r.threshold > 0 - ? r.threshold - : d.threshold, - threads: clampPositiveInt(r.threads, d.threads, 64), - // An unknown engine falls back to the default rather than disabling the lane: - // a typo in settings.json must not silently stop diarization, and the default - // is the one every existing sidecar already matches. - engine: DIARIZATION_ENGINE_IDS.includes(r.engine as DiarizationEngineId) - ? (r.engine as DiarizationEngineId) - : d.engine, - backend: DIARIZATION_BACKENDS.includes(r.backend as DiarizationBackend) - ? (r.backend as DiarizationBackend) - : d.backend, - python: str(r.python, d.python), - segModel: str(r.segModel, d.segModel), - embModel: str(r.embModel, d.embModel), - sortformerBin: str(r.sortformerBin, d.sortformerBin), - sortformerModel: str(r.sortformerModel, d.sortformerModel), - concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16), - // 0 is meaningful here (cap off), so this cannot use clampPositiveInt. - // Fractional hours are allowed — the knob is a duration, not a count. - maxAudioHours: - typeof r.maxAudioHours === "number" && - Number.isFinite(r.maxAudioHours) && - r.maxAudioHours >= 0 - ? r.maxAudioHours - : d.maxAudioHours, - }; -} - -const SOCIAL_URL_RE = /^(https?:\/\/|mailto:|\/)/i; - -// Normalize the family hub URL into a trailing-slash-free absolute http(s) URL. -// Returns "" for anything that isn't a usable absolute URL (the "no hub" state). -// Kept local to settings.ts to avoid a settings<->homepage import cycle; mirrors -// parseHomepageUrl() in homepage.ts. -export function normalizeHomepageUrl(input: unknown): string { - if (typeof input !== "string") return ""; - const trimmed = input.trim().replace(/\/+$/, ""); - return /^https?:\/\/\S+/i.test(trimmed) ? trimmed : ""; -} - -export function parseSocialLinks(input: unknown): SocialLink[] { - if (!Array.isArray(input)) return []; - const out: SocialLink[] = []; - for (const raw of input) { - if (!raw || typeof raw !== "object") continue; - const r = raw as Record<string, unknown>; - const label = typeof r.label === "string" ? r.label.trim() : ""; - const url = typeof r.url === "string" ? r.url.trim() : ""; - const svg = typeof r.svg === "string" ? r.svg : ""; - if (!label || !url || !svg) continue; - if (!SOCIAL_URL_RE.test(url)) continue; - out.push({ label, url, svg }); - } - return out; -} - -// Normalize an admin-provided SVG snippet for inline use in the export -// footer. Returns null on anything that looks unsafe or unrenderable. -// Steps: trim, allowlist-check, strip width/height, force fill="currentColor" -// + aria-hidden on the root <svg>. Requires a viewBox so the icon scales. -export function normalizeSocialSvg(raw: string): string | null { - if (typeof raw !== "string") return null; - const trimmed = raw.trim(); - if (!trimmed.startsWith("<svg") || !trimmed.endsWith("</svg>")) return null; - if (/<script\b/i.test(trimmed)) return null; - if (/<foreignObject\b/i.test(trimmed)) return null; - if (/<iframe\b/i.test(trimmed)) return null; - if (/javascript:/i.test(trimmed)) return null; - if (/\son[a-z]+\s*=/i.test(trimmed)) return null; - if (/<\?|<!ENTITY/i.test(trimmed)) return null; - - const openEnd = trimmed.indexOf(">"); - if (openEnd < 0) return null; - let opening = trimmed.slice(0, openEnd); - const rest = trimmed.slice(openEnd); - - if (!/\sviewBox\s*=\s*"/i.test(opening)) return null; - - opening = opening.replace(/\s(width|height)\s*=\s*"[^"]*"/gi, ""); - opening = opening.replace(/\s(width|height)\s*=\s*'[^']*'/gi, ""); - - if (!/\sfill\s*=/i.test(opening)) { - opening = opening.replace(/^<svg/i, '<svg fill="currentColor"'); - } - if (!/\saria-hidden\s*=/i.test(opening)) { - opening = opening.replace(/^<svg/i, '<svg aria-hidden="true"'); - } - return opening + rest; -} - -// normalizeSocialSvg() deliberately STRIPS width/height so the icon scales to its -// wrapper. The cost is that a viewBox-only <svg> has no intrinsic size, so before -// the stylesheet loads on a static host it paints at the replaced-element default -// (huge) — the "flash of giant social icons" FOUC. sizeSocialSvg() re-injects an -// intrinsic pixel size at RENDER time (existing site.json files already have the -// attributes stripped, so this must run on read, not just on write). The size is -// an *attribute*, not inline style, so a wrapper's `w-*`/`h-*` utilities still win -// once CSS loads — it only governs the pre-CSS first paint. -export function sizeSocialSvg(svg: string, px = 20): string { - if (typeof svg !== "string") return svg; - if (/^<svg[^>]*\swidth\s*=/i.test(svg)) return svg; // already sized - return svg.replace(/^<svg\b/i, `<svg width="${px}" height="${px}"`); -} - -export function clampSleepBetweenDownloadsSeconds(value: unknown): number { - const n = - typeof value === "number" && Number.isFinite(value) - ? Math.floor(value) - : SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS; - if (n < 0) return 0; - if (n > SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS) { - return SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS; - } - return n; -} - -export function clampMinFreeDiskGB(value: unknown): number { - const n = - typeof value === "number" && Number.isFinite(value) - ? Math.floor(value) - : MIN_FREE_DISK_GB_DEFAULT; - if (n < 0) return 0; - if (n > MIN_FREE_DISK_GB_MAX) return MIN_FREE_DISK_GB_MAX; - return n; -} - -export function clampResumeMarginGB(value: unknown): number { - const n = - typeof value === "number" && Number.isFinite(value) - ? Math.floor(value) - : RESUME_MARGIN_GB_DEFAULT; - if (n < 0) return 0; - if (n > RESUME_MARGIN_GB_MAX) return RESUME_MARGIN_GB_MAX; - return n; -} - -export function clampParallelTranscriptions(value: unknown): number { - const n = - typeof value === "number" && Number.isFinite(value) - ? Math.floor(value) - : PARALLEL_TRANSCRIPTIONS_DEFAULT; - if (n < 1) return 1; - if (n > PARALLEL_TRANSCRIPTIONS_MAX) return PARALLEL_TRANSCRIPTIONS_MAX; - return n; -} - -// 0 means "disabled" and is preserved as-is. Anything else is clamped into the -// [MIN, MAX] window; a non-finite value falls back to the default cadence. -export function clampAutoRefreshIntervalSeconds(value: unknown): number { - if (typeof value !== "number" || !Number.isFinite(value)) { - return AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS; - } - const n = Math.floor(value); - if (n <= 0) return 0; - if (n < AUTO_REFRESH_INTERVAL_MIN_SECONDS) { - return AUTO_REFRESH_INTERVAL_MIN_SECONDS; - } - if (n > AUTO_REFRESH_INTERVAL_MAX_SECONDS) { - return AUTO_REFRESH_INTERVAL_MAX_SECONDS; - } - return n; -} - -export function getSettings(): SiteSettings { - const file = getPaths().settingsFile; - let parsed: Partial<SiteSettings> = {}; - try { - parsed = JSON.parse(fs.readFileSync(file, "utf8")) as Partial<SiteSettings>; - } catch { - parsed = {}; - } - // Pick only known operational keys — a pre-multi-site settings.json may still - // carry siteTitle/groups/socialLinks, which now live per-site in site.json. - const merged: SiteSettings = { ...defaults() }; - const knownKeys = Object.keys(merged) as (keyof SiteSettings)[]; - for (const key of knownKeys) { - if (parsed[key] !== undefined) { - (merged as Record<string, unknown>)[key] = parsed[key]; - } - } - if (typeof merged.adminTitle !== "string" || !merged.adminTitle.trim()) { - merged.adminTitle = DEFAULT_ADMIN_TITLE; - } - merged.maxTranscriptPageBytes = clampPageBytes(merged.maxTranscriptPageBytes); - merged.transcriptionApps = sanitizeTranscriptionApps(merged.transcriptionApps); - // Migrate a pre-multi-app settings.json (transcribeBin/transcribeArgs/ - // transcribeModel, with no transcriptionApp key) onto the app registry. - if (parsed.transcriptionApp === undefined) { - migrateLegacyTranscription(parsed as LegacyTranscribeFields, merged); - } - if ( - typeof merged.transcriptionApp !== "string" || - !TRANSCRIPTION_APPS[merged.transcriptionApp] - ) { - merged.transcriptionApp = DEFAULT_TRANSCRIPTION_APP_ID; - } - if (typeof merged.cookiesFromBrowser !== "string") { - merged.cookiesFromBrowser = ""; - } - // A settings.json predating cookieMode (or carrying junk) gets the default, - // which preserves the historical retry-only behavior. - if (!isCookieMode(merged.cookieMode)) { - merged.cookieMode = DEFAULT_COOKIE_MODE; - } - merged.sleepBetweenDownloadsSeconds = clampSleepBetweenDownloadsSeconds( - merged.sleepBetweenDownloadsSeconds, - ); - if (!isDownloadFormatPreset(merged.downloadFormat)) { - merged.downloadFormat = "auto"; - } - merged.minFreeDiskGB = clampMinFreeDiskGB(merged.minFreeDiskGB); - merged.resumeMarginGB = clampResumeMarginGB(merged.resumeMarginGB); - merged.parallelTranscriptions = clampParallelTranscriptions( - merged.parallelTranscriptions, - ); - if (typeof merged.inlineTranscribeOnFallback !== "boolean") { - merged.inlineTranscribeOnFallback = false; - } - if (typeof merged.skipLiveDownloads !== "boolean") { - merged.skipLiveDownloads = true; - } - if (typeof merged.verifyAvailabilityBeforeClean !== "boolean") { - merged.verifyAvailabilityBeforeClean = true; - } - if (typeof merged.buildArchives !== "boolean") { - merged.buildArchives = true; - } - { - const s = merged.archiveStorage; - merged.archiveStorage = { - bucket: s && typeof s.bucket === "string" ? s.bucket : "", - publicBaseUrl: - s && typeof s.publicBaseUrl === "string" ? s.publicBaseUrl : "", - }; - } - if (!isReportDebouncePreset(merged.reportDebouncePreset)) { - merged.reportDebouncePreset = DEFAULT_REPORT_DEBOUNCE_PRESET; - } - merged.autoRefreshIntervalSeconds = clampAutoRefreshIntervalSeconds( - merged.autoRefreshIntervalSeconds, - ); - merged.syncScheduler = sanitizeSyncScheduler(merged.syncScheduler); - // THE SWEEPS' SCOPE, ON READ. `migrateSweepsToLanes` fills in - // `autoQueue.digest` / `.backfill` from the retired sweep fields when — and - // only when — the FILE does not already spell them, which is why it is handed - // `parsed` rather than `merged`: `merged` has had defaults folded in and can - // no longer tell "absent" from "default". It never enables a lane the sweep - // flag did not. See lib/laneMigration.ts. - merged.autoQueue = sanitizeAutoQueue(migrateSweepsToLanes(parsed)); - // No migration beside it: the legacy read (`channelPriorityFromLegacy`) needs - // 68 config.json files and getSettings is synchronous and reads one. It is a - // one-shot offline script instead, and an absent document sanitizes to the - // empty one, which means today's behaviour. - merged.channelPriority = sanitizeChannelPriority(merged.channelPriority); - merged.socialLinks = parseSocialLinks(merged.socialLinks); - merged.homepageUrl = normalizeHomepageUrl(merged.homepageUrl); - merged.savedVideoBackup = sanitizeSavedVideoBackup(merged.savedVideoBackup); - // THE COLD ROOT, ON READ. `storage.mediaRoot` — one absolute string — becomes - // a one-entry location list. Handed `parsed.storage` rather than - // `merged.storage` for the same reason `migrateSweepsToLanes` is handed - // `parsed`: `merged` has had `defaultStorage()` folded in and can no longer - // tell "the file has no locations key" from "the file has an empty list", and - // the migration must only fire on the former. See lib/storageLocations.ts. - merged.storage = sanitizeStorage( - migrateMediaRootToLocations( - (parsed as Record<string, unknown>).storage ?? merged.storage, - ), - ); - merged.buildPipeline = sanitizeBuildPipeline(merged.buildPipeline); - merged.digest = sanitizeDigest(merged.digest); - merged.diarization = sanitizeDiarization(merged.diarization); - merged.backfill = sanitizeBackfill(merged.backfill); - merged.attribution = sanitizeAttribution(merged.attribution); - // Workers. When the file predates the worker model (no `workers` key), - // synthesize a default list from the (now-settled) active app + per-app - // configs so existing installs behave identically. Otherwise sanitize the - // stored list. - if (parsed.workers === undefined) { - merged.workers = defaultWorkersFromApps( - merged.transcriptionApp, - merged.transcriptionApps, - merged.parallelTranscriptions, +// 1. A pre-multi-app file (transcribeBin/transcribeArgs/transcribeModel, no +// `transcriptionApp` key) is migrated onto the app registry. It reads the +// already-sanitized `transcriptionApps`, so it runs after the parse. +// 2. A file predating the worker model (no `workers` key) gets a worker list +// synthesized from the now-settled active app, so existing installs behave +// identically. A file that HAS the key keeps its sanitized list, even an +// empty one. +function finishRawMigrations( + parsed: SiteSettings, + raw: RawSettings, +): SiteSettings { + if (raw.transcriptionApp === undefined) { + migrateLegacyTranscription(raw as LegacyTranscribeFields, parsed); + } + if (raw.workers === undefined) { + parsed.workers = defaultWorkersFromApps( + parsed.transcriptionApp, + parsed.transcriptionApps, + parsed.parallelTranscriptions, ); - } else { - merged.workers = sanitizeWorkers(merged.workers); } - return merged; + return parsed; } -function clampPageBytes(value: unknown): number { - const n = - typeof value === "number" && Number.isFinite(value) - ? value - : TRANSCRIPT_PAGE_DEFAULT_BYTES; - if (n < TRANSCRIPT_PAGE_MIN_BYTES) return TRANSCRIPT_PAGE_MIN_BYTES; - if (n > TRANSCRIPT_PAGE_HARD_CAP_BYTES) return TRANSCRIPT_PAGE_HARD_CAP_BYTES; - return Math.floor(n); +export function getSettings(): SiteSettings { + const raw = rawObject(readRawSettings(getPaths().settingsFile)); + return finishRawMigrations(siteSettingsSchema.parse(premigrateRaw(raw)), raw); } type LegacyTranscribeFields = { @@ -1602,20 +132,6 @@ type LegacyTranscribeFields = { transcribeModel?: unknown; }; -// Coerce a raw settings.transcriptionApps value into a clean keyed map of -// AppInstanceConfig, dropping unknown/ill-typed fields. -export function sanitizeTranscriptionApps( - value: unknown, -): Record<string, AppInstanceConfig> { - if (!value || typeof value !== "object" || Array.isArray(value)) return {}; - const out: Record<string, AppInstanceConfig> = {}; - for (const [id, raw] of Object.entries(value as Record<string, unknown>)) { - if (!raw || typeof raw !== "object") continue; - out[id] = sanitizeWorkerConfig(raw); - } - return out; -} - function argsAreDefault(args: string[]): boolean { return ( args.length === DEFAULT_TRANSCRIBE_ARGS.length && @@ -1659,10 +175,16 @@ function migrateLegacyTranscription( merged.transcriptionApps = { ...merged.transcriptionApps, [appId]: cfg }; } -export async function writeSettings(next: SiteSettings): Promise<void> { - // Workers are the source of truth. A caller that still sets only the - // deprecated transcriptionApp/transcriptionApps (no `workers`) gets a list - // synthesized from them, so old call sites keep working during the transition. +// THE WORKER SHADOW, derived on every save. +// +// Workers are the source of truth. A caller that still sets only the deprecated +// transcriptionApp/transcriptionApps (no `workers`) gets a list synthesized from +// them, so old call sites keep working. The list is then VALIDATED — this throws +// — and the deprecated pair is rewritten as a faithful shadow of it (used only +// for rollback to a pre-worker build; a file with a `workers` key is never +// re-migrated): the active app is the first enabled local worker, and the per-app +// map mirrors each local worker's config. +function deriveWorkerShadow(next: SiteSettings): SiteSettings { let workers = sanitizeWorkers(next.workers); if (workers.length === 0) { const fallbackAppId = @@ -1679,14 +201,10 @@ export async function writeSettings(next: SiteSettings): Promise<void> { const workersErr = validateWorkers(workers); if (workersErr) throw new Error(workersErr); - // Keep the deprecated transcriptionApp/transcriptionApps as a faithful shadow - // of the workers (used only for rollback to a pre-worker build — a file with a - // `workers` key is never re-migrated). The active app is the first enabled - // local worker; the per-app map mirrors each local worker's config. const firstLocal = workers.find((w) => w.enabled && w.kind === "local") ?? workers.find((w) => w.kind === "local"); - const appId = + const transcriptionApp = firstLocal?.appId && TRANSCRIPTION_APPS[firstLocal.appId] ? firstLocal.appId : DEFAULT_TRANSCRIPTION_APP_ID; @@ -1696,87 +214,36 @@ export async function writeSettings(next: SiteSettings): Promise<void> { transcriptionApps[w.appId] = w.config ?? {}; } } - const socialLinks: SocialLink[] = []; - for (const link of parseSocialLinks(next.socialLinks)) { + return { ...next, workers, transcriptionApp, transcriptionApps }; +} + +// Every social link's SVG normalized for inline use, or a THROW naming the +// first one that is not safe to inline. The schema's own `parseSocialLinks` +// only checks shape; this is the write-side half. +function validatedSocialLinks(value: unknown): SocialLink[] { + const out: SocialLink[] = []; + for (const link of parseSocialLinks(value)) { const svg = normalizeSocialSvg(link.svg); if (svg === null) { throw new Error(`Social link "${link.label}" has an invalid SVG`); } - socialLinks.push({ ...link, svg }); + out.push({ ...link, svg }); } - const file = getPaths().settingsFile; - // Build the output from ONLY the known operational fields. Do NOT spread - // `next`: getSettings() spreads the raw file, so a settings.json still - // carrying pre-multi-site keys (siteTitle/groups/socialLinks) would otherwise - // smuggle those stale keys back onto disk on every save. - // - // THIS IS ALSO WHAT RETIRES A FIELD. The four legacy pause flags - // (`transcriptionsPaused`, `downloadsPaused`, `digest.digestsPaused` and the - // inverted `backfill.enabled`) are gone from the type, from the sanitizers - // and from this literal, so a settings.json that still spells one is read - // past on load and loses it on the next write. The gate is + return out; +} + +export async function writeSettings(next: SiteSettings): Promise<void> { + // THE SCHEMA IS ALSO WHAT RETIRES A FIELD. It names only the known + // operational fields and strips the rest, so a settings.json still carrying + // pre-multi-site keys (siteTitle/groups) or the four retired pause flags + // (`transcriptionsPaused`, `downloadsPaused`, `digest.digestsPaused`, the + // inverted `backfill.enabled`) loses them on this write. The gate is // `autoQueue[lane].held` and nothing else — see lib/pauseGates.ts. - const merged: SiteSettings = { - adminTitle: - typeof next.adminTitle === "string" && next.adminTitle.trim() - ? next.adminTitle.trim() - : DEFAULT_ADMIN_TITLE, - maxTranscriptPageBytes: clampPageBytes(next.maxTranscriptPageBytes), - transcriptionApp: appId, - transcriptionApps, - workers, - cookiesFromBrowser: - typeof next.cookiesFromBrowser === "string" - ? next.cookiesFromBrowser.trim() - : "", - cookieMode: isCookieMode(next.cookieMode) - ? next.cookieMode - : DEFAULT_COOKIE_MODE, - sleepBetweenDownloadsSeconds: clampSleepBetweenDownloadsSeconds( - next.sleepBetweenDownloadsSeconds, - ), - downloadFormat: isDownloadFormatPreset(next.downloadFormat) - ? next.downloadFormat - : "auto", - minFreeDiskGB: clampMinFreeDiskGB(next.minFreeDiskGB), - resumeMarginGB: clampResumeMarginGB(next.resumeMarginGB), - parallelTranscriptions: clampParallelTranscriptions( - next.parallelTranscriptions, - ), - inlineTranscribeOnFallback: next.inlineTranscribeOnFallback === true, - skipLiveDownloads: next.skipLiveDownloads !== false, - verifyAvailabilityBeforeClean: - next.verifyAvailabilityBeforeClean !== false, - buildArchives: next.buildArchives !== false, - archiveStorage: { - bucket: - typeof next.archiveStorage?.bucket === "string" - ? next.archiveStorage.bucket.trim() - : "", - publicBaseUrl: - typeof next.archiveStorage?.publicBaseUrl === "string" - ? next.archiveStorage.publicBaseUrl.trim() - : "", - }, - reportDebouncePreset: isReportDebouncePreset(next.reportDebouncePreset) - ? next.reportDebouncePreset - : DEFAULT_REPORT_DEBOUNCE_PRESET, - autoRefreshIntervalSeconds: clampAutoRefreshIntervalSeconds( - next.autoRefreshIntervalSeconds, - ), - syncScheduler: sanitizeSyncScheduler(next.syncScheduler), - autoQueue: sanitizeAutoQueue(next.autoQueue), - channelPriority: sanitizeChannelPriority(next.channelPriority), - socialLinks, - homepageUrl: normalizeHomepageUrl(next.homepageUrl), - savedVideoBackup: sanitizeSavedVideoBackup(next.savedVideoBackup), - storage: sanitizeStorage(next.storage), - buildPipeline: sanitizeBuildPipeline(next.buildPipeline), - digest: sanitizeDigest(next.digest), - diarization: sanitizeDiarization(next.diarization), - backfill: sanitizeBackfill(next.backfill), - attribution: sanitizeAttribution(next.attribution), - }; + const merged = siteSettingsSchema.parse({ + ...deriveWorkerShadow(next), + socialLinks: validatedSocialLinks(next.socialLinks), + }); + const file = getPaths().settingsFile; const tmp = `${file}.tmp-${process.pid}`; await fs.promises.writeFile(tmp, JSON.stringify(merged, null, 2) + "\n"); await fs.promises.rename(tmp, file); diff --git a/common/lib/settingsDocs.test.ts b/common/lib/settingsDocs.test.ts @@ -0,0 +1,63 @@ +import { readFileSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { renderSettingsExample, renderSettingsMarkdown } from "./settingsDocs"; + +// Run with: node_modules/.bin/tsx --test common/lib/settingsDocs.test.ts +// +// settings.json.example and SETTINGS.md are GENERATED from the settings schema +// (common/bin/settings-example.ts). This is what keeps them generated: a hand +// edit to either file, or a schema change without a regenerate, fails here. + +const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", ".."); + +for (const [name, render] of [ + ["settings.json.example", renderSettingsExample], + ["SETTINGS.md", renderSettingsMarkdown], +] as const) { + test(`${name} is what the schema generates`, () => { + const committed = readFileSync(path.join(REPO, name), "utf8"); + assert.equal( + committed, + render(), + `${name} is stale: run pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts`, + ); + }); +} + +test("the example parses back to the defaults", async () => { + const { siteSettingsSchema, defaultSiteSettings } = await import("./settingsSchema"); + const parsed = siteSettingsSchema.parse(JSON.parse(renderSettingsExample())); + assert.deepEqual(parsed, defaultSiteSettings()); +}); + +// EVERY FIELD, NOT ONLY THE 31 TOP-LEVEL ONES. The *_FIELD_DOCS records are +// complete by type (FieldDocs<T> requires one entry per key); these two check +// the wiring — that every object-valued block has a key table, and that a +// block's table names every key its default actually carries. +test("every object-valued block has a nested key table", async () => { + const { defaultSiteSettings } = await import("./settingsSchema"); + const { blockTables } = await import("./settingsDocs"); + const d = defaultSiteSettings() as Record<string, unknown>; + const tables = blockTables(defaultSiteSettings()) as Record<string, unknown[]>; + for (const [key, value] of Object.entries(d)) { + if (value === null || typeof value !== "object") continue; + assert.ok((tables[key]?.length ?? 0) > 0, `${key} has no key table`); + } +}); + +test("a block table documents every key its default carries", async () => { + const { defaultSiteSettings } = await import("./settingsSchema"); + const { blockTables } = await import("./settingsDocs"); + const d = defaultSiteSettings() as Record<string, unknown>; + for (const [key, list] of Object.entries(blockTables(defaultSiteSettings()))) { + const first = list?.[0]; + if (!first?.defaults || key === "autoQueue") continue; + const value = d[key] as Record<string, unknown>; + for (const field of Object.keys(value)) { + assert.ok(field in first.docs, `${key}.${field} is undocumented`); + } + } +}); diff --git a/common/lib/settingsDocs.ts b/common/lib/settingsDocs.ts @@ -0,0 +1,305 @@ +// THE TWO FILES GENERATED FROM THE SETTINGS SCHEMA: settings.json.example and +// the key table SETTINGS.md, both at the repo root. +// +// Pure renderers — `common/bin/settings-example.ts` writes them, and +// `settingsDocs.test.ts` asserts the committed files are byte-identical to what +// these return, so neither can be edited by hand without the test failing. +// +// Everything comes from `siteSettingsSchema` (./settingsSchema.ts): the keys and +// their order from its shape, the defaults from `defaultSiteSettings()`, the +// prose from each field's `.describe()`. Changing a default or a description is +// a schema edit followed by regenerating, never an edit here. + +import { + ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS, + ATTRIBUTION_SETTINGS_FIELD_DOCS, + BACKFILL_SETTINGS_FIELD_DOCS, + BUILD_PIPELINE_SETTINGS_FIELD_DOCS, + DIARIZATION_SETTINGS_FIELD_DOCS, + DIGEST_SETTINGS_FIELD_DOCS, + SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS, + SOCIAL_LINK_FIELD_DOCS, + SYNC_SCHEDULER_SETTINGS_FIELD_DOCS, + defaultSiteSettings, + siteSettingsSchema, + type SiteSettings, +} from "./settingsSchema"; +import { + AUTO_QUEUE_MATCH_FIELD_DOCS, + AUTO_QUEUE_NODE_FIELD_DOCS, + AUTO_QUEUE_POLICY_FIELD_DOCS, + LANES, +} from "./autoQueueTypes"; +import { + CHANNEL_AUTO_PAUSE_FIELD_DOCS, + CHANNEL_FOCUS_FIELD_DOCS, + CHANNEL_PRIORITY_ENTRY_FIELD_DOCS, + CHANNEL_PRIORITY_FIELD_DOCS, +} from "./channelPriority"; +import { + LLM_WORKER_CONFIG_FIELD_DOCS, + REMOTE_WORKER_CONFIG_FIELD_DOCS, + WORKER_FIELD_DOCS, +} from "./workers"; +import { APP_INSTANCE_CONFIG_FIELD_DOCS } from "./transcriptionApps"; +import { DIGEST_APP_CONFIG_FIELD_DOCS } from "./digest"; +import { + STORAGE_LOCATION_FIELD_DOCS, + STORAGE_SETTINGS_FIELD_DOCS, + STORAGE_VOLUME_FIELD_DOCS, +} from "./storageLocations"; + +// `workers` IS LEFT OUT OF THE EXAMPLE, and that is the one place the example +// is not the literal default object. Its default is `[]`, and a settings.json +// that SPELLS `workers: []` READS as "no transcription workers" — until the +// next save, when writeSettings' worker shadow synthesizes one from +// `transcriptionApp`; in between, auto-transcribe does nothing, silently. A file +// that does not name the key gets that worker list synthesized on read (see +// getSettings), which is what a template copied to settings.json should give. +export const EXAMPLE_OMITTED_KEYS = ["workers"] as const; + +export function renderSettingsExample(): string { + const d = defaultSiteSettings() as Record<string, unknown>; + for (const key of EXAMPLE_OMITTED_KEYS) delete d[key]; + return JSON.stringify(d, null, 2) + "\n"; +} + +function isScalar(v: unknown): boolean { + return v === null || typeof v !== "object"; +} + +// The default as a table cell: a scalar inline, an empty container inline, +// anything larger by reference to its section. +function defaultCell(v: unknown): string { + if (isScalar(v)) return "`" + JSON.stringify(v) + "`"; + const json = JSON.stringify(v); + if (json === "[]" || json === "{}") return "`" + json + "`"; + return Array.isArray(v) ? "list — see below" : "object — see below"; +} + +// A description inside a table cell: one line, pipes escaped, paragraphs kept. +function cell(text: string): string { + return text.replace(/\|/g, "\\|").replace(/\n\n/g, "<br><br>").replace(/\n/g, " "); +} + +// One nested key table. `defaults(key)` answers the Default column; a table of +// per-entry fields (list items, map values, tree nodes) has no defaults — each +// entry spells its own — and says so. +type KeyTable = { + path: string; + docs: Readonly<Record<string, string>>; + defaults?: (key: string) => string; +}; + +function fromObject(obj: unknown): (key: string) => string { + const r = (obj ?? {}) as Record<string, unknown>; + return (key) => (key in r ? defaultCell(r[key]) : "absent"); +} + +// A lane-policy field's default can differ per lane (`held`, `order`, `root`), +// and that difference is exactly what a reader needs to see. +function perLane(d: SiteSettings): (key: string) => string { + return (key) => { + const cells = LANES.map((lane) => { + const policy = d.autoQueue[lane] as Record<string, unknown>; + return key in policy ? defaultCell(policy[key]) : "absent"; + }); + if (cells.every((c) => c === cells[0])) return cells[0]; + return LANES.map((lane, i) => `${lane} ${cells[i]}`).join("<br>"); + }; +} + +export function blockTables(d: SiteSettings): Partial<Record<keyof SiteSettings, KeyTable[]>> { + return { + transcriptionApps: [ + { path: "transcriptionApps.<appId>", docs: APP_INSTANCE_CONFIG_FIELD_DOCS }, + ], + workers: [ + { path: "workers[]", docs: WORKER_FIELD_DOCS }, + { path: "workers[].config", docs: APP_INSTANCE_CONFIG_FIELD_DOCS }, + { path: "workers[].remote", docs: REMOTE_WORKER_CONFIG_FIELD_DOCS }, + { path: "workers[].llm", docs: LLM_WORKER_CONFIG_FIELD_DOCS }, + ], + archiveStorage: [ + { + path: "archiveStorage", + docs: ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS, + defaults: fromObject(d.archiveStorage), + }, + ], + syncScheduler: [ + { + path: "syncScheduler", + docs: SYNC_SCHEDULER_SETTINGS_FIELD_DOCS, + defaults: fromObject(d.syncScheduler), + }, + ], + autoQueue: [ + { + path: "autoQueue.<lane>", + docs: AUTO_QUEUE_POLICY_FIELD_DOCS, + defaults: perLane(d), + }, + { path: "autoQueue.<lane>.root (tree nodes)", docs: AUTO_QUEUE_NODE_FIELD_DOCS }, + { path: "autoQueue.<lane>.root … .match", docs: AUTO_QUEUE_MATCH_FIELD_DOCS }, + ], + channelPriority: [ + { + path: "channelPriority", + docs: CHANNEL_PRIORITY_FIELD_DOCS, + defaults: fromObject(d.channelPriority), + }, + { path: "channelPriority.focus", docs: CHANNEL_FOCUS_FIELD_DOCS }, + { path: "channelPriority.channels.<slug>", docs: CHANNEL_PRIORITY_ENTRY_FIELD_DOCS }, + { + path: "channelPriority.channels.<slug>.autoPaused", + docs: CHANNEL_AUTO_PAUSE_FIELD_DOCS, + }, + ], + socialLinks: [{ path: "socialLinks[]", docs: SOCIAL_LINK_FIELD_DOCS }], + savedVideoBackup: [ + { + path: "savedVideoBackup", + docs: SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS, + defaults: fromObject(d.savedVideoBackup), + }, + ], + storage: [ + { + path: "storage", + docs: STORAGE_SETTINGS_FIELD_DOCS, + defaults: fromObject(d.storage), + }, + { path: "storage.locations[]", docs: STORAGE_LOCATION_FIELD_DOCS }, + { path: "storage.locations[].volume", docs: STORAGE_VOLUME_FIELD_DOCS }, + ], + buildPipeline: [ + { + path: "buildPipeline", + docs: BUILD_PIPELINE_SETTINGS_FIELD_DOCS, + defaults: fromObject(d.buildPipeline), + }, + ], + digest: [ + { path: "digest", docs: DIGEST_SETTINGS_FIELD_DOCS, defaults: fromObject(d.digest) }, + { path: "digest.apps.<appId>", docs: DIGEST_APP_CONFIG_FIELD_DOCS }, + ], + diarization: [ + { + path: "diarization", + docs: DIARIZATION_SETTINGS_FIELD_DOCS, + defaults: fromObject(d.diarization), + }, + ], + backfill: [ + { + path: "backfill", + docs: BACKFILL_SETTINGS_FIELD_DOCS, + defaults: fromObject(d.backfill), + }, + ], + attribution: [ + { + path: "attribution", + docs: ATTRIBUTION_SETTINGS_FIELD_DOCS, + defaults: fromObject(d.attribution), + }, + ], + }; +} + +function renderTable(out: string[], table: KeyTable): void { + out.push(`#### \`${table.path}\``); + out.push(""); + if (table.defaults) { + out.push("| Key | Default | Description |"); + out.push("|---|---|---|"); + for (const [key, text] of Object.entries(table.docs)) { + out.push(`| \`${key}\` | ${table.defaults(key)} | ${cell(text)} |`); + } + } else { + out.push("Per entry — each entry spells its own values."); + out.push(""); + out.push("| Key | Description |"); + out.push("|---|---|"); + for (const [key, text] of Object.entries(table.docs)) { + out.push(`| \`${key}\` | ${cell(text)} |`); + } + } + out.push(""); +} + +export function renderSettingsMarkdown(): string { + const d = defaultSiteSettings(); + const values = d as Record<string, unknown>; + const shape = siteSettingsSchema.shape as Record< + string, + { description?: string } + >; + const tables = blockTables(d); + const keys = Object.keys(shape); + const out: string[] = []; + out.push("# settings.json keys"); + out.push(""); + out.push( + "<!-- GENERATED by common/bin/settings-example.ts from common/lib/settingsSchema.ts and the *_FIELD_DOCS records beside each block type — do not edit by hand. -->", + ); + out.push(""); + out.push( + "Global operational settings shared by every site this editor powers, " + + "persisted to `settings.json` at the repo root (or `$SETTINGS_FILE`). " + + "Per-site presentation lives in `sites/<id>/site.json`. Every key is " + + "optional: a missing key reads as its default, an ill-typed one is " + + "coerced to its default or clamped, and an unknown one is dropped on the " + + "next save.", + ); + out.push(""); + out.push( + "Regenerate this file and `settings.json.example` with " + + "`pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts`.", + ); + out.push(""); + out.push( + "`settings.json.example` is the default object with one key left out, " + + "`workers`: a file that does not name it gets a worker list synthesized " + + "from `transcriptionApp` on read. A file that spells `workers: []` READS " + + "as no transcription at all — until the next save, when the writer " + + "synthesizes a worker the same way.", + ); + out.push(""); + out.push( + "A copied example PINS every default it spells — including each lane's " + + "`autoQueue.<lane>.held` — so a default changed in a later release will " + + "not reach that file. Delete any key you would rather have track the " + + "defaults.", + ); + out.push(""); + out.push("| Key | Default |"); + out.push("|---|---|"); + for (const key of keys) { + out.push(`| [\`${key}\`](#${key.toLowerCase()}) | ${defaultCell(values[key])} |`); + } + out.push(""); + for (const key of keys) { + out.push(`## \`${key}\``); + out.push(""); + out.push(shape[key].description ?? ""); + out.push(""); + const v = values[key]; + if (isScalar(v)) { + out.push(`Default: \`${JSON.stringify(v)}\``); + out.push(""); + } else { + for (const table of tables[key as keyof SiteSettings] ?? []) { + renderTable(out, table); + } + out.push("Default:"); + out.push(""); + out.push("```json"); + out.push(JSON.stringify(v, null, 2)); + out.push("```"); + out.push(""); + } + } + return out.join("\n"); +} diff --git a/common/lib/settingsFieldSchemas.ts b/common/lib/settingsFieldSchemas.ts @@ -0,0 +1,63 @@ +// THE ZOD SEAMS FOR THE THREE SANITIZERS THAT LIVE OUTSIDE settings.ts. +// +// `settings.json` has three blocks whose parsers have homes of their own and +// callers of their own: `workers` (lib/workers.ts — the pool and the settings +// form), `channelPriority` (lib/channelPriority.ts — storageWatch and the +// channels actions call `sanitizeChannelPriority(value: unknown)` directly) and +// `autoQueue` (lib/autoQueueSchema.ts — the picker's tests pin it). Each keeps +// its home and its signature. What this file adds is ONE schema per block, so +// `lib/settingsSchema.ts` can compose them with the other twenty-eight fields +// the same way it composes everything: zod supplies the plumbing (the key, the +// strip of unknown siblings), the existing sanitizer supplies the arithmetic — +// and the totality. Nothing was re-implemented, so nothing could drift. +// +// WHY A SEPARATE FILE, and not a `workersSchema` export beside each sanitizer: +// all three homes are imported AS VALUES by `"use client"` forms +// (WorkersConfigForm, ChannelTierSelect, LadderRung, …). A module-level +// `import { z } from "zod"` there would put zod in a browser bundle for no +// reason. Only server code imports this file — lib/settingsSchema.ts, which +// itself is only reached through lib/settings.ts (node:fs at module scope). + +import { z } from "zod"; +import { sanitizeWorkers, type Worker } from "./workers"; +import { + sanitizeChannelPriority, + type ChannelPriority, +} from "./channelPriority"; +import { sanitizeAutoQueue } from "./autoQueueSchema"; +import type { AutoQueueSettings } from "./autoQueueTypes"; + +// A settings FIELD: any JSON value in, a legal value out. +// +// TOTALITY COMES FROM THE COERCION, NOT FROM ZOD. `z.unknown()` accepts every +// input, so its `.catch(undefined)` never fires, and zod does not guard the +// transform: a coercion that threw would throw out of `parse`. What makes a +// settings read never throw is that every coercion passed here — each clamp and +// sanitizer — is itself total over `unknown`. The `.catch` is kept only so the +// field keeps its shape if `z.unknown()` is ever swapped for a validating +// schema. +// +// What zod does contribute: `z.unknown()` accepts a missing key, and zod 4 +// still runs the transform for it and emits the key — so an absent field is +// DEFAULTED, not dropped, exactly as `defaults()` used to fill it — and unknown +// sibling keys are stripped by the enclosing object. +// +// There is deliberately no `.default()` anywhere: a default applies only to +// `undefined`, and every coercion here already decides that case itself — the +// place "a zero limit is a hold" lives is inside the clamp, not in a fallback +// zod would apply around it. +export function settingsField<T>(coerce: (value: unknown) => T) { + return z.unknown().catch(undefined).transform((value): T => coerce(value)); +} + +export const workersSchema = settingsField( + (value): Worker[] => sanitizeWorkers(value), +); + +export const channelPrioritySchema = settingsField( + (value): ChannelPriority => sanitizeChannelPriority(value), +); + +export const autoQueueSchema = settingsField( + (value): AutoQueueSettings => sanitizeAutoQueue(value), +); diff --git a/common/lib/settingsSchema.test.ts b/common/lib/settingsSchema.test.ts @@ -0,0 +1,425 @@ +// THE SETTINGS SCHEMA: its shape, its boundaries, its migrations, and the +// promise that a read never throws. +// +// Run with: node_modules/.bin/tsx --test common/lib/settingsSchema.test.ts +// +// SETTINGS SEAM, same arrangement as ./settingsWrite.test.ts: `getPaths()` +// memoizes at module scope, so SETTINGS_FILE is set before anything imports +// ./settings, and the module is imported dynamically below. + +import { mkdtempSync, writeFileSync } from "node:fs"; +import { rm } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { test, after } from "node:test"; +import assert from "node:assert/strict"; + +const ROOT = mkdtempSync(path.join(os.tmpdir(), "settings-schema-")); +process.env.TRANSCRIPTS_DIR = ROOT; +process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); + +import type { SiteSettings } from "./settings"; +import type { AppInstanceConfig } from "./transcriptionApps"; +import type { Worker } from "./workers"; +import type { AutoQueueSettings } from "./autoQueueTypes"; +import type { ChannelPriority } from "./channelPriority"; +import type { CookieMode } from "./cookiePolicy"; +import type { DownloadFormatPreset } from "../ytdlp/downloadFormat"; +import type { StorageSettings } from "./storageLocations"; +import type { + AttributionSettings, + BackfillSettings, + BuildPipelineSettings, + DiarizationSettings, + DigestSettings, + ReportDebouncePreset, + SavedVideoBackupSettings, + SocialLink, + SyncSchedulerSettings, +} from "./settingsSchema"; + +const S = await import("./settings"); +const { siteSettingsSchema, defaultSiteSettings, defaults, getSettings } = S; + +after(() => rm(ROOT, { recursive: true, force: true })); + +// ── THE SHAPE, PINNED ──────────────────────────────────────────────────────── +// +// `SiteSettings` is `z.infer<typeof siteSettingsSchema>` now. This is the type +// as it was hand-written before slice 4a, spelled out literally, so a schema +// edit that changes what 200 importers see is a tsc error here — not a surprise +// somewhere else. +// +// ONE DELIBERATE DIFFERENCE: `archiveStorage` was `archiveStorage?: {…}`. It was +// never absent at runtime (defaults(), getSettings and writeSettings all +// emitted it), and zod 4 cannot express "optional key that is always emitted", +// so it is required now. Nothing in the repo relied on the `?`. +type PreSchemaSiteSettings = { + adminTitle: string; + maxTranscriptPageBytes: number; + transcriptionApp: string; + transcriptionApps: Record<string, AppInstanceConfig>; + workers: Worker[]; + cookiesFromBrowser: string; + cookieMode: CookieMode; + sleepBetweenDownloadsSeconds: number; + downloadFormat: DownloadFormatPreset; + minFreeDiskGB: number; + resumeMarginGB: number; + parallelTranscriptions: number; + inlineTranscribeOnFallback: boolean; + skipLiveDownloads: boolean; + verifyAvailabilityBeforeClean: boolean; + buildArchives: boolean; + archiveStorage: { bucket: string; publicBaseUrl: string }; + reportDebouncePreset: ReportDebouncePreset; + autoRefreshIntervalSeconds: number; + syncScheduler: SyncSchedulerSettings; + autoQueue: AutoQueueSettings; + channelPriority: ChannelPriority; + socialLinks: SocialLink[]; + homepageUrl: string; + savedVideoBackup: SavedVideoBackupSettings; + storage: StorageSettings; + buildPipeline: BuildPipelineSettings; + digest: DigestSettings; + diarization: DiarizationSettings; + backfill: BackfillSettings; + attribution: AttributionSettings; +}; + +// Bracketed so the conditional does not distribute (see commit 8c43231). +type Same<A, B> = [A] extends [B] ? ([B] extends [A] ? true : false) : false; +const shapeUnchanged: Same<SiteSettings, PreSchemaSiteSettings> = true; + +test("SiteSettings keeps its 31 fields, in file order", () => { + assert.equal(shapeUnchanged, true); + assert.deepEqual(Object.keys(siteSettingsSchema.shape), [ + "adminTitle", + "maxTranscriptPageBytes", + "transcriptionApp", + "transcriptionApps", + "workers", + "cookiesFromBrowser", + "cookieMode", + "sleepBetweenDownloadsSeconds", + "downloadFormat", + "minFreeDiskGB", + "resumeMarginGB", + "parallelTranscriptions", + "inlineTranscribeOnFallback", + "skipLiveDownloads", + "verifyAvailabilityBeforeClean", + "buildArchives", + "archiveStorage", + "reportDebouncePreset", + "autoRefreshIntervalSeconds", + "syncScheduler", + "autoQueue", + "channelPriority", + "socialLinks", + "homepageUrl", + "savedVideoBackup", + "storage", + "buildPipeline", + "digest", + "diarization", + "backfill", + "attribution", + ]); + // A parsed object carries every key, in that order — writeSettings writes + // exactly this, so the order is the on-disk order. + assert.deepEqual( + Object.keys(defaults()), + Object.keys(siteSettingsSchema.shape), + ); +}); + +test("every field has a description (SETTINGS.md is generated from them)", () => { + for (const [key, field] of Object.entries(siteSettingsSchema.shape)) { + assert.ok( + typeof field.description === "string" && field.description.length > 20, + `${key} has no .describe()`, + ); + } +}); + +test("defaults() and defaultSiteSettings() are the schema's answer for an empty file", () => { + assert.deepEqual(defaultSiteSettings(), siteSettingsSchema.parse({})); + assert.deepEqual(defaults(), defaultSiteSettings()); + const d = defaults(); + assert.equal(d.adminTitle, S.DEFAULT_ADMIN_TITLE); + assert.equal(d.maxTranscriptPageBytes, S.TRANSCRIPT_PAGE_DEFAULT_BYTES); + assert.deepEqual(d.workers, []); + assert.deepEqual(d.archiveStorage, { bucket: "", publicBaseUrl: "" }); + assert.equal(d.skipLiveDownloads, true); + assert.equal(d.inlineTranscribeOnFallback, false); +}); + +// ── THE CLAMP BOUNDARIES ───────────────────────────────────────────────────── + +type Case = [input: unknown, expected: number]; + +function checkField( + key: keyof SiteSettings, + cases: Case[], +): void { + for (const [input, expected] of cases) { + const got = siteSettingsSchema.parse({ [key]: input })[key]; + assert.equal(got, expected, `${key}: ${JSON.stringify(input)} -> ${got}`); + } +} + +// min / max / NaN / string / null / absent, for each clamped top-level field. +test("maxTranscriptPageBytes clamps into [256 KiB, 20 MiB]", () => { + checkField("maxTranscriptPageBytes", [ + [0, S.TRANSCRIPT_PAGE_MIN_BYTES], + [S.TRANSCRIPT_PAGE_MIN_BYTES - 1, S.TRANSCRIPT_PAGE_MIN_BYTES], + [S.TRANSCRIPT_PAGE_MIN_BYTES, S.TRANSCRIPT_PAGE_MIN_BYTES], + [S.TRANSCRIPT_PAGE_HARD_CAP_BYTES, S.TRANSCRIPT_PAGE_HARD_CAP_BYTES], + [S.TRANSCRIPT_PAGE_HARD_CAP_BYTES + 1, S.TRANSCRIPT_PAGE_HARD_CAP_BYTES], + [1_000_000.7, 1_000_000], + [NaN, S.TRANSCRIPT_PAGE_DEFAULT_BYTES], + ["9000000", S.TRANSCRIPT_PAGE_DEFAULT_BYTES], + [null, S.TRANSCRIPT_PAGE_DEFAULT_BYTES], + [undefined, S.TRANSCRIPT_PAGE_DEFAULT_BYTES], + ]); +}); + +test("sleepBetweenDownloadsSeconds: 0 is off, capped at the max", () => { + checkField("sleepBetweenDownloadsSeconds", [ + [-1, 0], + [0, 0], + [S.SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS, S.SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS], + [S.SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS + 1, S.SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS], + [2.9, 2], + [NaN, S.SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS], + ["5", S.SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS], + [null, S.SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS], + ]); +}); + +test("minFreeDiskGB: 0 disables the gate, capped at the max", () => { + checkField("minFreeDiskGB", [ + [-5, 0], + [0, 0], + [S.MIN_FREE_DISK_GB_MAX, S.MIN_FREE_DISK_GB_MAX], + [S.MIN_FREE_DISK_GB_MAX + 1, S.MIN_FREE_DISK_GB_MAX], + [NaN, S.MIN_FREE_DISK_GB_DEFAULT], + ["0", S.MIN_FREE_DISK_GB_DEFAULT], + [null, S.MIN_FREE_DISK_GB_DEFAULT], + ]); +}); + +test("resumeMarginGB: 0 disables the hysteresis, capped at the max", () => { + checkField("resumeMarginGB", [ + [-1, 0], + [0, 0], + [S.RESUME_MARGIN_GB_MAX, S.RESUME_MARGIN_GB_MAX], + [S.RESUME_MARGIN_GB_MAX + 1, S.RESUME_MARGIN_GB_MAX], + [NaN, S.RESUME_MARGIN_GB_DEFAULT], + ["2", S.RESUME_MARGIN_GB_DEFAULT], + [null, S.RESUME_MARGIN_GB_DEFAULT], + ]); +}); + +test("parallelTranscriptions: floored at 1, capped at the max", () => { + checkField("parallelTranscriptions", [ + [0, 1], + [-3, 1], + [1, 1], + [S.PARALLEL_TRANSCRIPTIONS_MAX, S.PARALLEL_TRANSCRIPTIONS_MAX], + [S.PARALLEL_TRANSCRIPTIONS_MAX + 1, S.PARALLEL_TRANSCRIPTIONS_MAX], + [NaN, S.PARALLEL_TRANSCRIPTIONS_DEFAULT], + ["4", S.PARALLEL_TRANSCRIPTIONS_DEFAULT], + [null, S.PARALLEL_TRANSCRIPTIONS_DEFAULT], + ]); +}); + +test("autoRefreshIntervalSeconds: 0 survives as off, the rest clamps", () => { + checkField("autoRefreshIntervalSeconds", [ + [0, 0], + [-10, 0], + [0.5, 0], + [S.AUTO_REFRESH_INTERVAL_MIN_SECONDS, S.AUTO_REFRESH_INTERVAL_MIN_SECONDS], + [S.AUTO_REFRESH_INTERVAL_MAX_SECONDS, S.AUTO_REFRESH_INTERVAL_MAX_SECONDS], + [S.AUTO_REFRESH_INTERVAL_MAX_SECONDS + 1, S.AUTO_REFRESH_INTERVAL_MAX_SECONDS], + [NaN, S.AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS], + ["5", S.AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS], + [null, S.AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS], + ]); +}); + +test("syncScheduler.heartbeatSeconds: 0 is off, positive clamps into [MIN, MAX]", () => { + const hb = (v: unknown) => + siteSettingsSchema.parse({ syncScheduler: { heartbeatSeconds: v } }) + .syncScheduler.heartbeatSeconds; + assert.equal(hb(0), 0); + assert.equal(hb(-1), 0); + assert.equal(hb(1), S.SYNC_HEARTBEAT_MIN_SECONDS); + assert.equal(hb(S.SYNC_HEARTBEAT_MAX_SECONDS + 1), S.SYNC_HEARTBEAT_MAX_SECONDS); + assert.equal(hb(NaN), S.SYNC_HEARTBEAT_DEFAULT_SECONDS); + assert.equal(hb("60"), S.SYNC_HEARTBEAT_DEFAULT_SECONDS); + assert.equal(hb(null), S.SYNC_HEARTBEAT_DEFAULT_SECONDS); +}); + +test("sync cadences that allow zero keep it; the ones that do not floor at 1", () => { + const s = siteSettingsSchema.parse({ + syncScheduler: { + fullSweepIntervalMinutes: 0, + fullSweepConfirmMaxSuspects: 0, + fullSweepShrinkGuardPercent: 0, + defaultIntervalMinutes: 0, + maxConcurrentSyncs: 0, + quietHoursStart: 24, + quietHoursEnd: 6, + }, + }).syncScheduler; + assert.equal(s.fullSweepIntervalMinutes, 0); + assert.equal(s.fullSweepConfirmMaxSuspects, 0); + assert.equal(s.fullSweepShrinkGuardPercent, 0); + assert.equal(s.defaultIntervalMinutes, 1); + assert.equal(s.maxConcurrentSyncs, 1); + // An out-of-range hour clears BOTH ends of the window. + assert.equal(s.quietHoursStart, null); + assert.equal(s.quietHoursEnd, null); +}); + +test("enum fields fall back to their default on junk", () => { + const s = siteSettingsSchema.parse({ + cookieMode: "sometimes", + downloadFormat: "best", + reportDebouncePreset: "instant", + transcriptionApp: "no-such-app", + }); + const d = defaults(); + assert.equal(s.cookieMode, d.cookieMode); + assert.equal(s.downloadFormat, "auto"); + assert.equal(s.reportDebouncePreset, d.reportDebouncePreset); + assert.equal(s.transcriptionApp, d.transcriptionApp); +}); + +test("booleans: only an explicit value moves them off their default", () => { + const s = siteSettingsSchema.parse({ + inlineTranscribeOnFallback: "yes", + skipLiveDownloads: "no", + verifyAvailabilityBeforeClean: 0, + buildArchives: false, + }); + assert.equal(s.inlineTranscribeOnFallback, false); + assert.equal(s.skipLiveDownloads, true); + assert.equal(s.verifyAvailabilityBeforeClean, true); + assert.equal(s.buildArchives, false); +}); + +test("an unknown key is stripped, at the top level and inside a block", () => { + const s = siteSettingsSchema.parse({ + siteTitle: "pre-multi-site", + transcriptionsPaused: true, + backfill: { enabled: true, concurrency: 2 }, + }) as Record<string, unknown>; + assert.equal("siteTitle" in s, false); + assert.equal("transcriptionsPaused" in s, false); + assert.deepEqual(s.backfill, { concurrency: 2, allowRedownload: false }); +}); + +// ── THE LANE GATE AND THE ZERO HOLD ────────────────────────────────────────── + +test("held defaults to [false, false, false, true] — backfill ships held", () => { + const aq = defaults().autoQueue; + assert.deepEqual( + [aq.transcription.held, aq.download.held, aq.digest.held, aq.backfill.held], + [false, false, false, true], + ); + // And through a file that names the lanes but no gate. + const bare = siteSettingsSchema.parse({ + autoQueue: { transcription: {}, download: {}, digest: {}, backfill: {} }, + }).autoQueue; + assert.deepEqual( + [bare.transcription.held, bare.download.held, bare.digest.held, bare.backfill.held], + [false, false, false, true], + ); +}); + +// ── getSettings: THE RAW MIGRATIONS AND THE NEVER-THROW ────────────────────── + +function withFile(contents: string): SiteSettings { + writeFileSync(process.env.SETTINGS_FILE!, contents); + return getSettings(); +} + +test("getSettings never throws on a file that is not a settings object", () => { + // THE EMPTY FILE is the reference, not bare defaults(): an empty file still + // runs the absence-keyed migrations (workers synthesized, the two sweep lanes + // built by migrateSweepsToLanes), and a non-object file must read as exactly + // that. Before slice 4a, a file containing `null` threw. + const empty = withFile("{}"); + assert.equal(empty.workers.length, S.PARALLEL_TRANSCRIPTIONS_DEFAULT); + for (const contents of ["{", "null", "[]", "3", "", '"text"']) { + let got: SiteSettings | undefined; + assert.doesNotThrow(() => { + got = withFile(contents); + }, `contents ${JSON.stringify(contents)}`); + assert.deepEqual(got, empty, `contents ${JSON.stringify(contents)}`); + } +}); + +test("workers are synthesized ONLY when the key is absent", () => { + const absent = withFile(JSON.stringify({ parallelTranscriptions: 3 })); + assert.equal(absent.workers.length, 3); + assert.ok(absent.workers.every((w) => w.enabled && w.kind === "local")); + const empty = withFile(JSON.stringify({ workers: [] })); + assert.deepEqual(empty.workers, []); +}); + +test("legacy transcribe* fields migrate ONLY when transcriptionApp is absent", () => { + const legacy = { + transcribeBin: "/opt/chough", + transcribeModel: "/models/x.bin", + }; + const migrated = withFile(JSON.stringify(legacy)); + assert.equal(migrated.transcriptionApp, "chough"); + assert.equal(migrated.transcriptionApps.chough?.bin, "/opt/chough"); + assert.equal(migrated.workers[0].appId, "chough"); + + const named = withFile( + JSON.stringify({ ...legacy, transcriptionApp: "whisper-cpp" }), + ); + assert.equal(named.transcriptionApp, "whisper-cpp"); + assert.equal(named.transcriptionApps.chough, undefined); +}); + +test("storage.mediaRoot migrates ONLY when locations is absent", () => { + const migrated = withFile( + JSON.stringify({ storage: { mediaRoot: "/mnt/cold" } }), + ); + assert.deepEqual(migrated.storage.locations.map((l) => l.root), ["/mnt/cold"]); + assert.equal(migrated.storage.defaultLocationId, "default"); + + const listed = withFile( + JSON.stringify({ storage: { mediaRoot: "/mnt/cold", locations: [] } }), + ); + assert.deepEqual(listed.storage.locations, []); + + const none = withFile(JSON.stringify({})); + assert.deepEqual(none.storage, { locations: [], defaultLocationId: "" }); +}); + +test("the retired sweep fields migrate onto a lane ONLY when the lane is absent", () => { + const migrated = withFile( + JSON.stringify({ digest: { sweepEnabled: true, sweepChannels: ["a"] } }), + ); + assert.equal(migrated.autoQueue.digest.enabled, true); + // Absence was the trigger, not the default: a file that spells the lane + // keeps its own answer even though the sweep flag says otherwise. + const spelled = withFile( + JSON.stringify({ + digest: { sweepEnabled: true }, + autoQueue: { digest: { enabled: false } }, + }), + ); + assert.equal(spelled.autoQueue.digest.enabled, false); + // No sweep flag, no lane: the migration never enables anything. + assert.equal(withFile("{}").autoQueue.digest.enabled, false); + assert.equal(withFile("{}").autoQueue.backfill.enabled, false); +}); diff --git a/common/lib/settingsSchema.ts b/common/lib/settingsSchema.ts @@ -0,0 +1,1557 @@ +// THE SETTINGS SCHEMA — one definition of settings.json, used by the reader, +// the writer, the example file and the key table. +// +// one-core phase 3 slice 4a. What used to be three copies of the same list — +// the `SiteSettings` type, the `defaults()` literal, and the two field-by-field +// sanitizing literals in getSettings and writeSettings — is `siteSettingsSchema` +// below. `SiteSettings` is inferred from it; `defaults()` is `parse({})`; +// getSettings and writeSettings (lib/settings.ts) both parse through it; and +// `common/bin/settings-example.ts` generates settings.json.example and SETTINGS.md +// from it, including each field's `.describe()` text — which is where the +// comments that used to sit on the `SiteSettings` type now live, so they have one +// home and cannot drift from the key they describe. +// +// ZOD SUPPLIES THE PLUMBING, NOT THE ARITHMETIC. Every field is +// `settingsField(coerce)` — `z.unknown().catch(undefined).transform(coerce)` — +// and every `coerce` is the clamp or sanitizer that already existed, reused, so +// no boundary moved. Each is total over `unknown`, and that — not zod, whose +// `.catch` cannot fire on `z.unknown()` — is what makes a read never throw. +// The nested blocks' own fields are documented in the `*_FIELD_DOCS` record +// beside each block type (here and in workers.ts, channelPriority.ts, +// autoQueueTypes.ts, storageLocations.ts, transcriptionApps.ts, digest.ts), +// type-checked complete. Unknown keys are dropped by zod's default strip (never +// `.passthrough()`), which is what retires a field: a key the schema does not +// name cannot survive a read or a save. +// +// SERVER-ONLY IN PRACTICE: it imports node:path (sanitizeStorage) and is reached +// through lib/settings.ts, which imports node:fs at module scope. A `"use +// client"` form that needs a constant imports the constant — never this schema. +// +// NOT HERE: the three migrations keyed on a field's ABSENCE in the raw file +// (sweeps → lanes, mediaRoot → locations, legacy transcribe* → app registry and +// synthesized workers), because a parsed object cannot tell "absent" from +// "default". They run in getSettings around the parse. See lib/settings.ts. + +import path from "node:path"; +import { z } from "zod"; +// The auto-queue, workers and channel-priority blocks keep their own +// sanitizers in their own homes; these are their zod seams. +import { + autoQueueSchema, + channelPrioritySchema, + settingsField, + workersSchema, +} from "./settingsFieldSchemas"; +import { SYNC_INTERVAL_MAX_MINUTES } from "./channelConfig"; +import { + isDownloadFormatPreset, + type DownloadFormatPreset, +} from "../ytdlp/downloadFormat"; +import { + type AppInstanceConfig, + DEFAULT_TRANSCRIPTION_APP_ID, + TRANSCRIPTION_APPS, +} from "./transcriptionApps"; +import { sanitizeWorkerConfig } from "./workers"; +import { + INTERNAL_LOCATION_ID, + type StorageLocation, + type StorageSettings, + type StorageVolume, +} from "./storageLocations"; +import { + DEFAULT_DIARIZATION_ENGINE, + DEFAULT_DIARIZATION_THRESHOLD, + DIARIZATION_BACKENDS, + DIARIZATION_ENGINE_IDS, + type DiarizationBackend, + type DiarizationEngineId, +} from "./diarization"; +import { ATTRIBUTION_PROMPT_VERSION } from "./attribution"; +import { + DEFAULT_COOKIE_MODE, + isCookieMode, + type CookieMode, +} from "./cookiePolicy"; +// From the CLIENT-SAFE digest module, deliberately — digestApps.ts imports execa, +// and settings.ts must stay reachable from anywhere. +import { + CLAUDE_DIGEST_APP_ID, + DEFAULT_DIGEST_APP_ID, + DEFAULT_DIGEST_TIMESTAMP_MODE, + DIGEST_SECTION_KINDS, + DIGEST_TIMESTAMP_MODES, + isDigestSectionKind, + isDigestTimestampMode, + type DigestAppConfig, + type DigestSectionKind, + type DigestTimestampMode, +} from "./digest"; +import type { FieldDocs } from "./fieldDocs"; + +export type { Worker } from "./workers"; +export type { AutoQueueSettings } from "./autoQueueTypes"; +export type { ChannelPriority } from "./channelPriority"; + +// Transcribe placeholder/arg helpers now live with the whisper-cpp app in +// transcriptionApps.ts. Re-exported here so existing import sites keep working. +export { + type AppInstanceConfig, + TRANSCRIBE_PLACEHOLDER_AUDIO, + TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE, + TRANSCRIBE_PLACEHOLDER_MODEL, + TRANSCRIBE_KNOWN_PLACEHOLDERS, + DEFAULT_TRANSCRIBE_ARGS, + validateTranscribeArgs, +} from "./transcriptionApps"; + +// Configuration for speaker attribution — putting names to the speaker turns. +// +// OFF by default, and that default is doing real work rather than being +// cautious. The text-only lane costs roughly one model call per transcript +// CHUNK, which on this corpus is ~194,000 calls, the same order as the digest +// sweep — and the digest sweep has completed 0.17% of its own. Arming both at +// once puts two multi-week lanes on the same 8 GB card with nothing to arbitrate +// between them (the backfill lane's yield deliberately watches only the +// transcription lane). Nothing here arms anything; a pilot decides whether the +// corpus-wide text-only pass is worth 25-55 GPU-days at all. +// Each field is documented in ATTRIBUTION_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). +export type AttributionSettings = { + enabled: boolean; + appId: string; + model: string; + diarizedEnabled: boolean; + textOnlyEnabled: boolean; + promptVersion: number; +}; + +export const ATTRIBUTION_SETTINGS_FIELD_DOCS: FieldDocs<AttributionSettings> = { + enabled: + "Master switch. Off means the backfill registry reports no attribution " + + "work at all — the feature gate every Operation has.", + appId: + "Which digest app runs the naming. Attribution IS a digest-app workload" + + " — constrained JSON decoding over transcript text — so it reuses that " + + "registry and that per-app config (settings.digest.apps[appId]) rather " + + "than growing a second copy of the ollama URL, context size and " + + "timeout.", + model: + "Model override. Empty = the app's configured model, then its default. " + + "It is separate from the digest's because the two workloads may want " + + "different sizes, and because it is part of the freshness identity: " + + "sharing the digest's model field would make a digest bake-off " + + "invalidate every attribution record on disk as a side effect.", + diarizedEnabled: + "The lanes, separately. Both default OFF even when `enabled` is on, so " + + "turning the feature on to look at it cannot start a corpus sweep.\n\n" + + "They are not a fallback pair. `diarized` is one call per video and " + + "grounded in acoustic clustering; `textOnly` is ~30 calls and guesses " + + "at identity across chunk seams. An operator may reasonably want the " + + "first forever and the second never.", + textOnlyEnabled: + "The text-only attribution lane: names speakers from the transcript alone (~30 model calls per video, guessing identity across chunk seams). Default OFF even when `enabled` is on. See `diarizedEnabled` — the two are separate lanes, not a fallback pair.", + promptVersion: + "The prompt generation a record must match to count as fresh.\n\n" + + "Defaults to (and is floored at) ATTRIBUTION_PROMPT_VERSION, the " + + "shipped constant. Raising it forces a corpus-wide regeneration without" + + " a code change, which is the honest way to redo everything after a " + + "prompt tweak. It cannot be set BELOW the shipped constant, and that " + + "floor is the lesson from digestPrompt.ts's version 1 -> 2 note: " + + "pinning freshness to an older generation freezes output from a " + + "superseded prompt into the corpus, looking identical to output from " + + "the current one.", +}; + +// Configuration for the backfill lane — the generic answer to "a derived-data +// feature landed and 77,000 existing videos do not have it". +// +// WHAT THE LANE GETS is its own queueKey (concurrency with transcription) plus a +// limit() that returns 0 to stand aside — the same mechanism the digest yield +// uses, which fails OPEN so a bad read costs contention rather than a deadlock. +// There is no priority system to join: the registry submits every named queue at +// concurrency 1 and SchedulerTier only orders work within a single key. +// +// THE SHARE IS `concurrency` AND THE LANE'S `autoQueue.backfill.maxWorkers`; the +// yield is the operation's declared `contendsFor`. Slice 1.3 retired the +// `weight` scalar that used to mean both — see backfillLimit(). +// Each field is documented in BACKFILL_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). +export type BackfillSettings = { + concurrency: number; + allowRedownload: boolean; +}; + +export const BACKFILL_SETTINGS_FIELD_DOCS: FieldDocs<BackfillSettings> = { + concurrency: + "Slots the lane may use when it is not standing aside. Kept at 1 by " + + "default for the same reason diarization.concurrency is: this is CPU-" + + "bound work competing with GPU feeding and the digest sweep for the " + + "same 8 threads.", + allowRedownload: + "Re-acquire media for videos whose input is GONE (audio deleted after " + + "transcription). OFF by default and deliberately so: measured on this " + + "corpus, 836 videos still have media and ~76,270 would need a re-" + + "download — 91x the reachable work, against 45 GB free at 97% full. " + + "When on, each re-fetched file is removed in a `finally` as soon as the" + + " backfill has used it, unless the video is marked do-not-clean, or " + + "unless the auto-transcribe policy would replace its auto-captions " + + "(`replaceAutoSubs`, or a leaf on `downloadedAutoSubsOnly`), in which " + + "case the audio is kept for that runner.\n\n" + + "WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: " + + "\"youtube\"` channel — which normally only fetches subtitles — the re-" + + "acquire applies a PER-VIDEO transcribe override so yt-dlp lands audio " + + "a diarizer can read; the channel's stored config is not changed. " + + "Without that override the fetch re-downloads the captions the video " + + "already has and lands nothing, which is what happened to ~16,000 " + + "videos on eight channels in 2026-08.", +}; + +// Configuration for the speaker-diarization capture lane. +// +// This exists because AUDIO IS THE ONE PERISHABLE INPUT in the pipeline: +// cleanAudioFromTranscribed deletes it once a video is transcribed, so +// diarization has to happen while the audio is still there or not at all. The +// capture half is deliberately all that ships here — attribution, LLM speaker +// naming, viewer badges and quote filtering can all be redone later from the +// saved JSON, whereas the audio cannot. +// Each field is documented in DIARIZATION_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). +export type DiarizationSettings = { + enabled: boolean; + inlineAfterTranscribe: boolean; + threshold: number; + threads: number; + engine: DiarizationEngineId; + backend: DiarizationBackend; + python: string; + segModel: string; + embModel: string; + sortformerBin: string; + sortformerModel: string; + concurrency: number; + maxAudioHours: number; +}; + +export const DIARIZATION_SETTINGS_FIELD_DOCS: FieldDocs<DiarizationSettings> = { + enabled: + "Master switch. OFF by default so a transcription batch can start " + + "before this lands, with diarization backfilled over the retained audio" + + " afterwards.\n\n" + + "Turning it ON also arms the cleanup guard: the Clean-audio sweep stops" + + " deleting audio for a transcribed video that has no diarization.json " + + "yet. That is the point — it is what keeps the perishable input alive " + + "long enough to be captured — but it means enabling this holds disk.", + inlineAfterTranscribe: + "Run diarization inline in the post-transcribe hook.\n\n" + + "OFF by default, and that default is a MEASURED decision, not caution. " + + "Measured on this box: GPU transcription runs at 221 s/audio-hour " + + "(16.3x realtime, over 3,602 real videos), CPU diarization at ~500-680 " + + "s/audio-hour. Diarization is therefore ~2-3x SLOWER than the " + + "transcription it follows, so running it inline drops whole-pipeline " + + "throughput by roughly 3-4x and leaves the GPU idle while the CPU " + + "catches up.\n\n" + + "The intended sequence for a large batch is the opposite: leave this " + + "off, let the batch transcribe at full GPU speed with `enabled` holding" + + " the audio, and diarize afterwards with the backfill pass. Turn it on " + + "for steady state, once the arrival rate is a few videos a day rather " + + "than a corpus.", + threshold: + "Clustering threshold — the single most consequential knob, since it " + + "decides how many speakers come out. Larger merges more aggressively.\n\n" + + "The default is 0.9, NOT sherpa-onnx's own 0.5, and that is measured on" + + " this corpus. On a 6-minute excerpt of a two-person interview (known " + + "ground truth: 2 speakers), sherpa's default produced 22 clusters; 0.9 " + + "produced 6, with the top two at 40%/40% of talk time — recognizably " + + "the two hosts. Sweep on the same clip: 0.4→23, 0.5→22, 0.6→17, 0.7→12," + + " 0.8→10, 0.9→6.\n\n" + + "It still over-splits, which is why this is a capture lane and not an " + + "answer: the turns are recorded with the threshold that produced them, " + + "so a later attribution pass can re-cluster or re-run without needing " + + "the audio back.", + threads: + "Engine threads per diarize run.", + engine: + "Which engine runs. \"sherpa-onnx\" is the shipped default and what every" + + " sidecar on disk was produced by; \"sortformer\" is the ggml engine " + + "built by scripts/build-sortformer.sh.\n\n" + + "CHANGING THIS RESTATES THE FRESHNESS IDENTITY (see diarizationTarget)," + + " so every sidecar written by the other engine becomes stale and the " + + "backfill lane offers to redo it. That is intended — the two disagree " + + "about how many speakers exist, and a corpus half-diarized by each is " + + "not one corpus — but on the retained audio it is weeks of work, not a " + + "toggle.\n\n" + + "Why anyone would: on the same file, sherpa at its tuned threshold " + + "returns 13 speakers and sortformer returns 4, agreeing on the dominant" + + " speaker's share to within half a point (73.1% vs 73.5%). On the " + + "corpus's worst case sherpa returns 35 and sortformer 4. Over-splitting" + + " is the failure mode this lane has always had, and sortformer is end-" + + "to-end rather than clustered, so it does not have it. The cost is a " + + "hard ceiling of 4 speakers and ~1.8x the wall clock.", + backend: + "Compute device for the sortformer engine; ignored by sherpa-onnx, " + + "which has no Vulkan compute path on Linux.\n\n" + + "\"vulkan\" is 1.5x faster than a thread-tuned CPU run (894 vs 1305 " + + "s/audio-hour, measured on this box) and holds 558 MB resident instead " + + "of 4.84 GB by keeping weights and activations in VRAM. It also takes " + + "~4.4 GB of an 8 GB card, which is why the lane YIELDS to transcription" + + " rather than sharing — see controller/digestYield.ts.", + python: + "Python interpreter for the default sherpa-onnx engine. sherpa-onnx " + + "ships wheels only up to cp313, and this box's system python is 3.14 — " + + "so this usually points at a dedicated venv rather than `python3`.", + segModel: + "ONNX model paths for the default engine. Empty = the lane cannot run, " + + "which is reported as a skip rather than a failure.", + embModel: + "ONNX speaker-embedding model path for the sherpa-onnx engine. Empty = the lane cannot run, reported as a `not-configured` skip rather than a failure (same as `segModel`).", + sortformerBin: + "Binary and model for the sortformer engine, both produced by " + + "scripts/build-sortformer.sh. Empty = that engine cannot run, reported " + + "as the same \"not-configured\" skip as an unset segModel/embModel.", + sortformerModel: + "Model for the sortformer engine, produced by scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the same `not-configured` skip as an unset `sortformerBin`.", + concurrency: + "How many diarize runs may execute at once in the backfill pass. Kept " + + "low by default: diarization is CPU-bound and competes with GPU feeding" + + " and the digest sweep for the same 8 threads.", + maxAudioHours: + "Videos longer than this are DEFERRED rather than diarized: reported as" + + " a third number that is never summed into reachable work, so a capped " + + "corpus can never read as finished.\n\n" + + "THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering " + + "holds a pairwise distance matrix over speech-segment embeddings — " + + "O(n^2) in SEGMENT count — and speaker-turn density varies 40x across " + + "this corpus (33-1364 turns/hour), so duration does not actually " + + "predict the blowup: a sparse 7h42m video completed while a dense 6h12m" + + " one was OOM-killed. Duration is merely the only predictor available " + + "for free, from metadata already on disk, BEFORE spending 45 minutes to" + + " find out. n^2 at 30k segments is 6.7 GiB and at 40k is 11.9 GiB, " + + "which brackets the 10.6 GB and 9.6 GB peaks measured on this 16 GB " + + "box.\n\n" + + "0 disables the cap. That is where this goes once windowed diarization " + + "lands: windowing divides per-window n by the window count, so the " + + "matrix falls by its square, and the cap stops being needed rather than" + + " being tuned.", +}; + +// Configuration for the derived-corpus digest layer. Local-first by decision: +// `remoteEnabled` gates the metered lane and defaults to false, so nothing here +// can spend money until it is explicitly turned on. +// Each field is documented in DIGEST_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). +export type DigestSettings = { + remoteEnabled: boolean; + longTailSeconds: number; + localAppId: string; + remoteAppId: string; + apps: Record<string, DigestAppConfig>; + yieldToTranscription: boolean; + yieldToCpuWorkers: boolean; + spendCapUsd: number; + sections: DigestSectionKind[]; + timestampMode: DigestTimestampMode; + promptVariant: string; +}; + +export const DIGEST_SETTINGS_FIELD_DOCS: FieldDocs<DigestSettings> = { + remoteEnabled: + "Master switch for the metered (remote-api) lane. OFF by default — an " + + "opt-in overflow for the long tail or a channel where local quality is " + + "poor, never the default path.", + longTailSeconds: + "Videos longer than this are \"long tail\": 8.2% of the corpus by count, " + + "46% of all transcript tokens. The batch's duration-aware ordering and " + + "the optional remote overflow both key off it.", + localAppId: + "The engine each lane uses (ids from common/lib/digestApps.ts).", + remoteAppId: + "The engine the metered (remote-api) lane uses — an id from common/lib/digestApps.ts. Unknown ids degrade to the default app rather than failing.", + apps: + "Per-app config, keyed by app id — the same id-keyed sub-record shape " + + "as transcriptionApps.", + yieldToTranscription: + "Yield the GPU to the transcription lane: while transcription is " + + "working, the digest batch's limit() returns 0 and the pool idle-waits." + + " ON by default, because `digest:local` is deliberately on a different " + + "queue from TRANSCRIPTION_QUEUE and so would otherwise run ollama and " + + "the transcription engine on the same 8 GB card. See " + + "controller/digestYield.ts.", + yieldToCpuWorkers: + "Whether a busy worker pinned to `device: \"cpu\"` counts as GPU " + + "contention.\n\n" + + "OFF by default, which is the FIX for a real bug: the yield originally " + + "tested only `kind === \"local\"`, so on a box with one GPU worker and " + + "two CPU-pinned ones (this box, at parallelTranscriptions 2) the digest" + + " lane stopped dead for transcription that competes for zero GPU " + + "shaders.\n\n" + + "Only an EXPLICIT \"cpu\" is treated as non-contending. A worker with no " + + "device set is using the engine binary's own default, which may be the " + + "GPU, so it still triggers the yield — the unknown case fails safe.\n\n" + + "Composes with `yieldToTranscription`: that is the master switch, this " + + "only narrows which workers it reacts to.", + spendCapUsd: + "Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. " + + "Only ever consulted for a metered app.", + sections: + "Which sections a sweep generates.\n\n" + + "Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, " + + "and that is not a contradiction: a tag call sends the same transcript " + + "as the chapter call before it, so it hits the engine's cached prefix " + + "and pays essentially no prefill (+0.4 s across 4 extra calls, against " + + "22.4 s for the first 4). All it pays is decode, and a tag list is ~30 " + + "output tokens where a chapter list is ~200–290.\n\n" + + "The corollary matters more than the number: run them in the SAME pass." + + " Tags generated later, on their own, pay full prefill again — measured" + + " at 44% of a whole chapters pass, i.e. 3–9× the marginal cost of just " + + "including them now.", + timestampMode: + "How each chunk's transcript markers are numbered — see " + + "DigestTimestampMode. Was a scored variable in the bake-off rather than" + + " a pre-applied fix; the measurement is in and \"chunk-local\" is now the" + + " shipped default.", + promptVariant: + "A free-text label for a non-default prompt shape, folded into the " + + "recorded provenance by digestPromptVariant(). Setting it invalidates " + + "every digest generated under a different label, which is exactly what " + + "makes a bake-off round re-run its sample instead of skipping it as " + + "fresh. Empty = default.", +}; + +// "basic" — `pnpm run build` in export/, serialized on the build queue (shared +// output tree → no safe parallelism). +// "docker" — isolated per-site container builds (follow-up); enables real +// parallel multi-site builds capped by maxParallelBuilds. +export type BuildMode = "basic" | "docker"; + +// Each field is documented in BUILD_PIPELINE_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). +export type BuildPipelineSettings = { + mode: BuildMode; + maxParallelBuilds: number; + dockerImage: string; + dockerfile: string; +}; + +export const BUILD_PIPELINE_SETTINGS_FIELD_DOCS: FieldDocs<BuildPipelineSettings> = { + mode: + "\"basic\" — `pnpm run build` in export/, serialized on the build queue (shared output tree, no safe parallelism). \"docker\" — isolated per-site container builds, parallel up to `maxParallelBuilds`.", + maxParallelBuilds: + "Cap on concurrent per-site container builds in docker mode. Ignored in" + + " basic mode (which is always serial). Clamped to [1, " + + "BUILD_MAX_PARALLEL_MAX].", + dockerImage: + "Tag of the reusable build image (built once, reused for every site).", + dockerfile: + "Dockerfile path relative to the monorepo root, used to (re)build the " + + "image.", +}; + +// Each field is documented in SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). +export type SavedVideoBackupSettings = { + enabled: boolean; + dest: string; + intervalMinutes: number; +}; + +export const SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS: FieldDocs<SavedVideoBackupSettings> = { + enabled: + "Master switch for the scheduled backup. A backup can still be run " + + "manually when this is false, as long as a destination is set.", + dest: + "Destination root the store is mirrored into (a local path or any rsync" + + " target). Empty disables both scheduled and manual backups.", + intervalMinutes: + "Cadence (minutes) for the scheduled backup when enabled. Clamped into " + + "the sync-interval window; default daily.", +}; + +// Each field is documented in SYNC_SCHEDULER_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). +export type SyncSchedulerSettings = { + enabled: boolean; + defaultIntervalMinutes: number; + maxConcurrentSyncs: number; + quietHoursStart: number | null; + quietHoursEnd: number | null; + backoffBaseMinutes: number; + backoffMaxMinutes: number; + heartbeatSeconds: number; + keepLatestCheckIntervalMinutes: number; + fullSweepIntervalMinutes: number; + fullSweepConfirmMaxSuspects: number; + fullSweepShrinkGuardPercent: number; +}; + +export const SYNC_SCHEDULER_SETTINGS_FIELD_DOCS: FieldDocs<SyncSchedulerSettings> = { + enabled: + "Master switch. When false, a tick selects nothing (manual sync still " + + "works).", + defaultIntervalMinutes: + "Fallback cadence (minutes) for channels with no per-channel override.", + maxConcurrentSyncs: + "Cap on sync jobs running/queued at once. A tick queues at most (cap - " + + "currently-active) channels; the rest roll to the next tick. This is " + + "also the stagger mechanism that keeps a big due-batch from hitting the" + + " source all at once.", + quietHoursStart: + "Optional local-clock quiet window during which auto-sync is " + + "suppressed. Both null = always allowed. The window may wrap past " + + "midnight (e.g. start=22, end=6). Hours are [0,23]; the window is " + + "[start, end).", + quietHoursEnd: + "End hour of the quiet window, [0,23], exclusive. See `quietHoursStart`: both must be valid hours or the window is cleared (null = always allowed).", + backoffBaseMinutes: + "Failure backoff bounds. After N consecutive failed scheduled syncs a " + + "channel waits min(base * 2^(N-1), max) minutes before it's eligible " + + "again.", + backoffMaxMinutes: + "Ceiling on the failure backoff (see `backoffBaseMinutes`): a channel waits min(base * 2^(N-1), max) minutes after N consecutive failures. Never below the base.", + heartbeatSeconds: + "Cadence (seconds) for the editor's in-process heartbeat — the internal" + + " timer armed by the instrumentation hook (editor/instrumentation.ts) " + + "that calls the scheduler tick directly, so no external cron is needed." + + " 0 = off: rely on the external `pnpm sync:tick` heartbeat instead. Any" + + " positive value is clamped to [SYNC_HEARTBEAT_MIN_SECONDS, " + + "SYNC_HEARTBEAT_MAX_SECONDS]. The env var SYNC_HEARTBEAT_SECONDS " + + "overrides this at runtime. See SCHEDULED_SYNC.md.", + keepLatestCheckIntervalMinutes: + "Cadence (minutes) for the scheduled keep-latest deletion check. For " + + "each channel with ChannelConfig.keepLatest > 0, the tick re-probes the" + + " kept window for source deletion (checkKeptDeletedAction) at most this" + + " often and pins any gone videos as do-not-clean. Clamped into the " + + "sync-interval window; default daily. The check shares the same " + + "concurrency cap and quiet-hours window as scheduled syncs. See " + + "editor/app/scheduler/runTick.ts.", + fullSweepIntervalMinutes: + "Default cadence (minutes) for the sync FULL SWEEP — the deep pass that" + + " re-enumerates a channel's whole listing in one yt-dlp spawn, " + + "refreshes the stored `playlist` file, and flags videos that have left " + + "the listing into maybe-missing.json. Ordinary syncs stay on the cheap " + + "newest-first paged walk; a sync only upgrades itself to a sweep when " + + "this interval has elapsed since the channel's lastFullSweepAt. Per-" + + "channel override: ChannelConfig.fullSweepIntervalMinutes. 0 = never " + + "sweep. Default daily. See common/jobs/deepSync.ts.", + fullSweepConfirmMaxSuspects: + "Upper bound on how many maybe-missing suspects a full sweep will " + + "resolve in-line with the per-video availability probe (deleted vs " + + "private vs unlisted). At or under the cap the sweep runs the targeted " + + "check itself, so \"Sync all\" surfaces upstream deletions with no extra " + + "clicks; over it, the suspects are flagged and left for a manual check " + + "rather than firing hundreds of probes inside a sync. 0 = never auto-" + + "confirm.", + fullSweepShrinkGuardPercent: + "Shrink guard: how far a fresh listing may fall below the stored one " + + "before it is treated as suspect rather than acted on. Expressed as a " + + "percentage of the previous count, floored at SHRINK_ABS_FLOOR entries " + + "so ordinary churn on a small channel doesn't trip it. A suspect " + + "listing does not rewrite `playlist` or maybe-missing.json and does not" + + " count as a sweep — but a SECOND enumeration reporting a similar count" + + " confirms it and is accepted, so a genuine mass deletion costs at most" + + " one cadence period. 0 = off (the empty-listing rejection still " + + "applies). See controller/acceptListing.ts.", +}; + +// Each field is documented in SOCIAL_LINK_FIELD_DOCS below (rendered into SETTINGS.md). +export type SocialLink = { + label: string; + url: string; + svg: string; +}; + +export const SOCIAL_LINK_FIELD_DOCS: FieldDocs<SocialLink> = { + label: + "Visible name, also the accessible label of the icon.", + url: + "Link target: http(s), mailto: or a site-relative path.", + svg: + "Inline SVG markup. Normalized on save (width/height stripped, " + + "fill=\"currentColor\", aria-hidden) and rejected when unsafe (script, " + + "foreignObject, event handlers, javascript: URLs) or when it has no " + + "viewBox.", +}; + +// Each field is documented in ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). +export type ArchiveStorageSettings = { + bucket: string; + publicBaseUrl: string; +}; + +export const ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS: FieldDocs<ArchiveStorageSettings> = { + bucket: + "Cloudflare R2 bucket an oversize archive zip is uploaded to on deploy " + + "(`wrangler r2 object put`, keyed `<siteId>/archives/<file>`). Blank = " + + "no overflow.", + publicBaseUrl: + "Public base URL of that bucket; the Downloads page links " + + "`<publicBaseUrl>/<key>`. Both fields must be set for overflow to " + + "happen.", +}; + +export const SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS = 600; +export const SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS = 10; + +export const MIN_FREE_DISK_GB_DEFAULT = 5; +export const MIN_FREE_DISK_GB_MAX = 100000; + +// Hysteresis margin for the low-disk gate. 2 GB is deliberately larger than any +// single scratch file the pipeline writes, so cleaning one up cannot by itself +// reopen the gate. +export const RESUME_MARGIN_GB_DEFAULT = 2; +export const RESUME_MARGIN_GB_MAX = 1000; + +export const PARALLEL_TRANSCRIPTIONS_MAX = 16; +export const PARALLEL_TRANSCRIPTIONS_DEFAULT = 2; + +// Passive UI auto-refresh cadence. 0 is a sentinel meaning "disabled"; any other +// value is clamped into [MIN, MAX] seconds. +export const AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS = 5; +export const AUTO_REFRESH_INTERVAL_MIN_SECONDS = 1; +export const AUTO_REFRESH_INTERVAL_MAX_SECONDS = 600; + +// Global snapshot-scheduler debounce presets. `debounceMs` is the quiet-period +// window after the last report-changing action; `maxWaitMs` caps the total +// delay under continuous activity (null = no cap, fire purely on the quiet +// period). Consumed by common/jobs/snapshotScheduler.ts and surfaced in the +// Settings form. +export type ReportDebouncePreset = "fast" | "balanced" | "lazy"; + +export const REPORT_DEBOUNCE_PRESETS: Record< + ReportDebouncePreset, + { debounceMs: number; maxWaitMs: number | null } +> = { + fast: { debounceMs: 1000, maxWaitMs: null }, + balanced: { debounceMs: 3000, maxWaitMs: 30000 }, + lazy: { debounceMs: 10000, maxWaitMs: 60000 }, +}; + +export const DEFAULT_REPORT_DEBOUNCE_PRESET: ReportDebouncePreset = "fast"; + +export function isReportDebouncePreset(v: unknown): v is ReportDebouncePreset { + return v === "fast" || v === "balanced" || v === "lazy"; +} + +export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024; +export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024; +export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024; + +export const DEFAULT_ADMIN_TITLE = "Transcript Browser Admin"; + +// Sync-scheduler bounds + defaults. Default cadence is daily; concurrency is +// conservative so a tick doesn't fan out into the source provider all at once. +export const SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES = 1440; +export const SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT = 2; +export const SYNC_SCHEDULER_MAX_CONCURRENT_MAX = 16; +export const SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES = 30; +export const SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES = 1440; +export const KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES = 1440; +// Full-sweep defaults. Daily: a sweep is one full enumeration of the channel, +// far more expensive than the 50-entry page an ordinary sync fetches. The +// confirm cap keeps an unattended sweep from fanning out into hundreds of +// per-video probes when a channel's listing changes wholesale. +export const FULL_SWEEP_DEFAULT_INTERVAL_MINUTES = 1440; +export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT = 25; +export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX = 10000; +// Shrink-guard default: a listing that has lost more than a tenth of its +// entries (and more than SHRINK_ABS_FLOOR of them) needs a second opinion. +export const FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT = 10; +export const FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX = 100; +export const SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES = 1440; + +// Internal-heartbeat cadence bounds. 0 means "off" (use an external cron +// heartbeat); any other value is clamped into [MIN, MAX] seconds. The floor +// keeps the in-process timer from busy-looping; the ceiling is one hour. +export const SYNC_HEARTBEAT_DEFAULT_SECONDS = 0; +export const SYNC_HEARTBEAT_MIN_SECONDS = 15; +export const SYNC_HEARTBEAT_MAX_SECONDS = 3600; + +export function defaultSyncScheduler(): SyncSchedulerSettings { + return { + enabled: false, + defaultIntervalMinutes: SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES, + maxConcurrentSyncs: SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT, + quietHoursStart: null, + quietHoursEnd: null, + backoffBaseMinutes: SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES, + backoffMaxMinutes: SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES, + heartbeatSeconds: SYNC_HEARTBEAT_DEFAULT_SECONDS, + keepLatestCheckIntervalMinutes: KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES, + fullSweepIntervalMinutes: FULL_SWEEP_DEFAULT_INTERVAL_MINUTES, + fullSweepConfirmMaxSuspects: FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT, + fullSweepShrinkGuardPercent: FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT, + }; +} + +// Clamp an internal-heartbeat cadence: 0 (off) passes through; any positive +// value is clamped up into [MIN, MAX]; junk falls back to the default. +export function clampHeartbeatSeconds(value: unknown): number { + if (typeof value !== "number" || !Number.isFinite(value)) { + return SYNC_HEARTBEAT_DEFAULT_SECONDS; + } + const n = Math.floor(value); + if (n <= 0) return 0; + if (n < SYNC_HEARTBEAT_MIN_SECONDS) return SYNC_HEARTBEAT_MIN_SECONDS; + if (n > SYNC_HEARTBEAT_MAX_SECONDS) return SYNC_HEARTBEAT_MAX_SECONDS; + return n; +} + +function clampHourOrNull(value: unknown): number | null { + if (typeof value !== "number" || !Number.isFinite(value)) return null; + const n = Math.floor(value); + if (n < 0 || n > 23) return null; + return n; +} + +// Like clampPositiveInt, but 0 survives as a sentinel ("off"/"never"). Used by +// the cadences whose disabled state is expressed as a zero rather than a +// separate boolean. +function clampIntAllowZero(value: unknown, fallback: number, max: number): number { + const n = + typeof value === "number" && Number.isFinite(value) + ? Math.floor(value) + : fallback; + if (n <= 0) return 0; + if (n > max) return max; + return n; +} + +function clampPositiveInt(value: unknown, fallback: number, max: number): number { + const n = + typeof value === "number" && Number.isFinite(value) + ? Math.floor(value) + : fallback; + if (n < 1) return 1; + if (n > max) return max; + return n; +} + +// Coerce a raw settings.syncScheduler value into a clean SyncSchedulerSettings, +// falling back to defaults for missing/ill-typed fields. Quiet hours are only +// honored when BOTH endpoints are valid hours; otherwise the window is cleared. +export function sanitizeSyncScheduler(value: unknown): SyncSchedulerSettings { + const d = defaultSyncScheduler(); + if (!value || typeof value !== "object") return d; + const r = value as Record<string, unknown>; + const start = clampHourOrNull(r.quietHoursStart); + const end = clampHourOrNull(r.quietHoursEnd); + const backoffBase = clampPositiveInt( + r.backoffBaseMinutes, + d.backoffBaseMinutes, + SYNC_INTERVAL_MAX_MINUTES, + ); + return { + enabled: r.enabled === true, + defaultIntervalMinutes: clampPositiveInt( + r.defaultIntervalMinutes, + d.defaultIntervalMinutes, + SYNC_INTERVAL_MAX_MINUTES, + ), + maxConcurrentSyncs: clampPositiveInt( + r.maxConcurrentSyncs, + d.maxConcurrentSyncs, + SYNC_SCHEDULER_MAX_CONCURRENT_MAX, + ), + quietHoursStart: start !== null && end !== null ? start : null, + quietHoursEnd: start !== null && end !== null ? end : null, + backoffBaseMinutes: backoffBase, + // Cap can't sit below the base, or backoff would never grow. + backoffMaxMinutes: Math.max( + backoffBase, + clampPositiveInt( + r.backoffMaxMinutes, + d.backoffMaxMinutes, + SYNC_INTERVAL_MAX_MINUTES, + ), + ), + heartbeatSeconds: clampHeartbeatSeconds(r.heartbeatSeconds), + keepLatestCheckIntervalMinutes: clampPositiveInt( + r.keepLatestCheckIntervalMinutes, + d.keepLatestCheckIntervalMinutes, + SYNC_INTERVAL_MAX_MINUTES, + ), + fullSweepIntervalMinutes: clampIntAllowZero( + r.fullSweepIntervalMinutes, + d.fullSweepIntervalMinutes, + SYNC_INTERVAL_MAX_MINUTES, + ), + fullSweepConfirmMaxSuspects: clampIntAllowZero( + r.fullSweepConfirmMaxSuspects, + d.fullSweepConfirmMaxSuspects, + FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX, + ), + fullSweepShrinkGuardPercent: clampIntAllowZero( + r.fullSweepShrinkGuardPercent, + d.fullSweepShrinkGuardPercent, + FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX, + ), + }; +} + +export function defaultSavedVideoBackup(): SavedVideoBackupSettings { + return { + enabled: false, + dest: "", + intervalMinutes: SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES, + }; +} + +// Coerce a raw settings.savedVideoBackup value into a clean +// SavedVideoBackupSettings. A missing destination forces enabled off, since a +// backup with nowhere to go is meaningless. +export function sanitizeSavedVideoBackup( + value: unknown, +): SavedVideoBackupSettings { + const d = defaultSavedVideoBackup(); + if (!value || typeof value !== "object") return d; + const r = value as Record<string, unknown>; + const dest = typeof r.dest === "string" ? r.dest.trim() : ""; + return { + enabled: dest !== "" && r.enabled === true, + dest, + intervalMinutes: clampPositiveInt( + r.intervalMinutes, + d.intervalMinutes, + SYNC_INTERVAL_MAX_MINUTES, + ), + }; +} + +// Where relocated channel media goes: the named locations. +// +// This used to be ONE FIELD, `mediaRoot` — a single absolute string, the cold +// drive, typed once. It grew into a list of entities because a root alone +// cannot answer the two questions the operator actually has: is that disk here, +// and if it came up somewhere else, how do I point the channels at it without +// ssh and hand edits? A location carries an id, a label, the root, an opt-in +// `autoRepoint`, and the volume identity learned at its last probe. +// +// Still NOT a policy: a channel on a location is not thereby deprioritized, and +// nothing auto-relocates anything because a location exists. +// +// AVAILABILITY IS NEVER STORED HERE. A refresh that wrote "available" would +// rewrite settings.json — and so bump the pulse revision — every few seconds. +// The probe (common/lib/storageVolumes.ts) is computed per request; only the +// `volume` identity is ever written back, and only when it changed. +// +// The types live in lib/storageLocations.ts, which is pure: a `"use client"` +// file may import them, and must not reach storageVolumes.ts (execa). +export type { StorageLocation, StorageVolume, StorageSettings }; + +export function defaultStorage(): StorageSettings { + return { locations: [], defaultLocationId: "" }; +} + +const LOCATION_ID_RE = /^[a-z0-9][a-z0-9-]{0,63}$/; + +// "internal" IS TAKEN. It is the synthetic /storage row for the corpus volume +// (INTERNAL_LOCATION_ID), and the regex above admits it — so a hand-edited +// settings.json, or an operator typing the obvious word into the New location +// form, could store a real location under the one id the page assembles for +// itself. The row would then be built twice, the rollup would count channels +// into whichever assembled last, and `locationOfDataDir` would start matching +// unrelocated channels against it. +function isReservedLocationId(id: string): boolean { + return id === INTERNAL_LOCATION_ID; +} + +function sanitizeVolume(value: unknown): StorageVolume | undefined { + if (!value || typeof value !== "object") return undefined; + const v = value as Record<string, unknown>; + const uuid = typeof v.uuid === "string" ? v.uuid.trim() : ""; + const mountpoint = + typeof v.mountpoint === "string" ? v.mountpoint.trim() : ""; + // No uuid is no identity, and no mountpoint means `root === join(mountpoint, + // relPath)` cannot hold — either way the record is not usable for finding the + // volume again, so it is dropped rather than half-kept. + if (!uuid || !mountpoint) return undefined; + const relPath = typeof v.relPath === "string" ? v.relPath.trim() : ""; + const fstype = typeof v.fstype === "string" ? v.fstype.trim() : ""; + const label = typeof v.label === "string" ? v.label.trim() : ""; + return { + uuid, + ...(fstype ? { fstype } : {}), + ...(label ? { label } : {}), + mountpoint, + relPath, + }; +} + +// Coerce a raw settings.storage value into a clean StorageSettings. +// +// EXISTENCE IS NOT CHECKED, on purpose: the whole point of a cold location is +// that it is a drive that may not be mounted when settings are read, and a +// sanitizer that dropped the root on an unmounted platter would silently erase +// the operator's choice on the next save. +// +// ABSOLUTENESS *IS* checked, and a location with a relative root is DROPPED +// rather than resolved. Resolving it would anchor the location to whatever cwd +// the reader booted in — a different directory under docker, under a worktree, +// and under `pnpm dev` — so the same settings.json would name three different +// drives. The location form rejects a relative path with a message before it +// ever gets here; this is the last line, not the only one. +// +// NESTED ROOTS ARE ALLOWED. "/mnt/platter" and "/mnt/platter/archive" may both +// be locations; `locationOfDataDir` resolves a channel to the LONGEST matching +// root. Nothing here rejects the nesting, because the operator who arranges a +// disk that way means it. +// +// A STALE `mediaRoot` SITTING BESIDE `locations` IS IGNORED — it is not merged +// back in as an extra location. `migrateMediaRootToLocations` reads it exactly +// once, when `locations` is absent; after that the list is the whole truth, and +// resurrecting a root the operator deleted would be a bug, not a kindness. +// +// ROLLBACK: an older binary sanitizes this block to `{ mediaRoot: "" }` — the +// locations are dropped and the single cold root comes back blank. One string +// lost, nothing on disk moved. `cp settings.json settings.json.pre-storage- +// locations` before the upgrade and a downgrade is a file copy. +export function sanitizeStorage(value: unknown): StorageSettings { + const d = defaultStorage(); + if (!value || typeof value !== "object") return d; + const r = value as Record<string, unknown>; + const rawList = Array.isArray(r.locations) ? r.locations : []; + const locations: StorageLocation[] = []; + const seen = new Set<string>(); + for (const entry of rawList) { + if (!entry || typeof entry !== "object") continue; + const e = entry as Record<string, unknown>; + const id = typeof e.id === "string" ? e.id.trim() : ""; + if (!LOCATION_ID_RE.test(id) || isReservedLocationId(id) || seen.has(id)) { + continue; + } + const rawRoot = typeof e.root === "string" ? e.root.trim() : ""; + if (!path.isAbsolute(rawRoot)) continue; + // "/mnt/platter/" and "/mnt/platter" are one root; "/" stays "/". + const stripped = rawRoot.replace(/\/+$/, ""); + const root = stripped === "" ? "/" : stripped; + const label = typeof e.label === "string" ? e.label.trim() : ""; + const volume = sanitizeVolume(e.volume); + seen.add(id); + locations.push({ + id, + label: label || id, + root, + autoRepoint: e.autoRepoint === true, + ...(volume ? { volume } : {}), + }); + } + const wanted = + typeof r.defaultLocationId === "string" ? r.defaultLocationId.trim() : ""; + // A default naming a location that is gone falls back to the first one, not + // to "": with a location configured, "no default" is never the answer the + // operator wanted, and a blank default silently disables every prefill. + const defaultLocationId = locations.some((l) => l.id === wanted) + ? wanted + : (locations[0]?.id ?? ""); + // THE SAVED-VIDEO STORE'S LOCATION IS NOT FALLEN BACK, and the asymmetry + // with `defaultLocationId` above is deliberate. That one is a PREFERENCE, so + // picking another location when the named one is gone is helpful. This one is + // a RECORD OF WHERE BYTES ARE: pointing it at a different location because + // the recorded one was deleted would claim the store had moved when nothing + // had. A dangling id sanitizes to "" — "in place" — which is what the disk + // says as soon as anybody looks, and the symlink (if any) keeps working + // regardless, because the store is reached through it and not through this. + const savedWanted = + typeof r.savedVideosLocationId === "string" + ? r.savedVideosLocationId.trim() + : ""; + const savedVideosLocationId = locations.some((l) => l.id === savedWanted) + ? savedWanted + : ""; + return { + locations, + defaultLocationId, + ...(savedVideosLocationId ? { savedVideosLocationId } : {}), + }; +} + +// 4 hours. Measured: videos over this are 8.2% of the corpus by count but hold +// 46% of all transcript tokens, so they are where a sweep's wall-clock actually +// goes and where chunk-seam bugs live. +export const DIGEST_LONG_TAIL_DEFAULT_SECONDS = 4 * 3600; +export const DIGEST_LONG_TAIL_MAX_SECONDS = 24 * 3600; + +export function defaultDigest(): DigestSettings { + return { + // OFF. The metered lane is built but never the default — see PLAN.md. + remoteEnabled: false, + longTailSeconds: DIGEST_LONG_TAIL_DEFAULT_SECONDS, + localAppId: DEFAULT_DIGEST_APP_ID, + remoteAppId: CLAUDE_DIGEST_APP_ID, + // Empty on purpose: every per-app knob falls through to its own default + // constant (resolveNumCtx -> DEFAULT_DIGEST_NUM_CTX, now 8192, and + // maxCuesForContext sizes the chunk to it). Seeding a copy of those values + // here would give the same number two homes and let them drift. + apps: {}, + // ON. Real GPU contention with the transcription engine is a genuine cost + // (re-priced: 11.2 s/chunk idle against 24.9 s/chunk on a contended box), so + // the safe default is to step aside; turning it off is the deliberate choice. + yieldToTranscription: true, + // OFF. A CPU-pinned worker is not GPU contention, and treating it as such + // stalled the digest lane for nothing. See DigestSettings.yieldToCpuWorkers. + yieldToCpuWorkers: false, + spendCapUsd: 0, + sections: ["chapters"], + timestampMode: DEFAULT_DIGEST_TIMESTAMP_MODE, + promptVariant: "", + }; +} + +// Coerce a raw settings.digest.apps value into a clean keyed map of +// DigestAppConfig. Mirrors sanitizeTranscriptionApps — INCLUDING its +// Array.isArray guard, without which a JSON array would pass the typeof check and +// produce numeric-keyed garbage. +export function sanitizeDigestApps( + value: unknown, +): Record<string, DigestAppConfig> { + if (!value || typeof value !== "object" || Array.isArray(value)) return {}; + const out: Record<string, DigestAppConfig> = {}; + for (const [id, raw] of Object.entries(value as Record<string, unknown>)) { + if (!raw || typeof raw !== "object") continue; + const r = raw as Record<string, unknown>; + const cfg: DigestAppConfig = {}; + if (typeof r.bin === "string" && r.bin.trim()) cfg.bin = r.bin.trim(); + if (typeof r.baseUrl === "string" && r.baseUrl.trim()) { + cfg.baseUrl = r.baseUrl.trim(); + } + if (typeof r.model === "string" && r.model.trim()) cfg.model = r.model.trim(); + if (typeof r.numCtx === "number" && r.numCtx > 0) { + cfg.numCtx = Math.floor(r.numCtx); + } + if (typeof r.temperature === "number" && r.temperature >= 0) { + cfg.temperature = r.temperature; + } + if (typeof r.timeoutMs === "number" && r.timeoutMs > 0) { + cfg.timeoutMs = Math.floor(r.timeoutMs); + } + // Only carried when explicitly set — see DigestAppConfig.think. + if (typeof r.think === "boolean") cfg.think = r.think; + out[id] = cfg; + } + return out; +} + +export function sanitizeDigest(value: unknown): DigestSettings { + const d = defaultDigest(); + if (!value || typeof value !== "object") return d; + const r = value as Record<string, unknown>; + const sections = Array.isArray(r.sections) + ? (r.sections.filter(isDigestSectionKind) as DigestSectionKind[]) + : []; + return { + remoteEnabled: r.remoteEnabled === true, + longTailSeconds: clampPositiveInt( + r.longTailSeconds, + d.longTailSeconds, + DIGEST_LONG_TAIL_MAX_SECONDS, + ), + // Unknown app ids are not rejected here: getDigestApp() is total and falls + // back to the local default, so a stale id degrades rather than breaking. + localAppId: + typeof r.localAppId === "string" && r.localAppId.trim() + ? r.localAppId.trim() + : d.localAppId, + remoteAppId: + typeof r.remoteAppId === "string" && r.remoteAppId.trim() + ? r.remoteAppId.trim() + : d.remoteAppId, + apps: sanitizeDigestApps(r.apps), + // Defaults to ON when absent — `=== false` rather than `!== true`, so a + // settings file written before this field existed keeps the GPU-safe + // behaviour instead of silently opting into contention. + yieldToTranscription: r.yieldToTranscription !== false, + // The OPPOSITE idiom, and deliberately so: `=== true`, so absence falls to + // OFF. The field's absence means a settings file written before the CPU-worker + // bug was found, and for those files OFF is the FIXED behaviour, not a silent + // change of intent — nobody ever asked to stall the digest lane for a CPU + // transcription. `yieldToTranscription` still gates the whole thing, so the + // GPU-safe default is untouched. + yieldToCpuWorkers: r.yieldToCpuWorkers === true, + spendCapUsd: + typeof r.spendCapUsd === "number" && r.spendCapUsd > 0 + ? Math.round(r.spendCapUsd * 100) / 100 + : 0, + // An empty/garbage list would silently generate nothing, so fall back to the + // default rather than honoring it. + sections: sections.length > 0 ? sections : d.sections, + timestampMode: isDigestTimestampMode(r.timestampMode) + ? r.timestampMode + : d.timestampMode, + // Trimmed and length-capped: it goes into provenance on every record, and a + // runaway value would bloat 119k sidecars. + promptVariant: + typeof r.promptVariant === "string" + ? r.promptVariant.trim().slice(0, 40) + : d.promptVariant, + }; +} + +// Every known section kind, for the settings UI's checkbox list. +export const DIGEST_SECTION_OPTIONS = DIGEST_SECTION_KINDS; +export const DIGEST_TIMESTAMP_MODE_OPTIONS = DIGEST_TIMESTAMP_MODES; + +export const BUILD_MAX_PARALLEL_DEFAULT = 2; +export const BUILD_MAX_PARALLEL_MAX = 16; +export const DEFAULT_BUILD_IMAGE = "yt-dlp-transcript-browser-build"; +export const DEFAULT_BUILD_DOCKERFILE = "Dockerfile.build"; + +export function isBuildMode(v: unknown): v is BuildMode { + return v === "basic" || v === "docker"; +} + +export function defaultBuildPipeline(): BuildPipelineSettings { + return { + mode: "basic", + maxParallelBuilds: BUILD_MAX_PARALLEL_DEFAULT, + dockerImage: DEFAULT_BUILD_IMAGE, + dockerfile: DEFAULT_BUILD_DOCKERFILE, + }; +} + +// Coerce a raw settings.buildPipeline value into a clean BuildPipelineSettings, +// falling back to defaults for missing/ill-typed fields. +export function sanitizeBuildPipeline(value: unknown): BuildPipelineSettings { + const d = defaultBuildPipeline(); + if (!value || typeof value !== "object") return d; + const r = value as Record<string, unknown>; + const dockerImage = + typeof r.dockerImage === "string" && r.dockerImage.trim() + ? r.dockerImage.trim() + : d.dockerImage; + const dockerfile = + typeof r.dockerfile === "string" && r.dockerfile.trim() + ? r.dockerfile.trim() + : d.dockerfile; + return { + mode: isBuildMode(r.mode) ? r.mode : d.mode, + maxParallelBuilds: clampPositiveInt( + r.maxParallelBuilds, + d.maxParallelBuilds, + BUILD_MAX_PARALLEL_MAX, + ), + dockerImage, + dockerfile, + }; +} + +export function defaultBackfill(): BackfillSettings { + return { + concurrency: 1, + // See BackfillSettings.allowRedownload — this one holds disk. + allowRedownload: false, + }; +} + +export function sanitizeBackfill(value: unknown): BackfillSettings { + const d = defaultBackfill(); + if (!value || typeof value !== "object") return d; + const r = value as Record<string, unknown>; + return { + // Clamped rather than rejected: a hand-edited 5 means "as much as possible", + // and reading it as 0 would be the opposite of the intent. + concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16), + allowRedownload: r.allowRedownload === true, + }; +} + +export function defaultAttribution(): AttributionSettings { + return { + // OFF, and both lanes OFF under it. See AttributionSettings. + enabled: false, + appId: DEFAULT_DIGEST_APP_ID, + model: "", + diarizedEnabled: false, + textOnlyEnabled: false, + promptVersion: ATTRIBUTION_PROMPT_VERSION, + }; +} + +export function sanitizeAttribution(value: unknown): AttributionSettings { + const d = defaultAttribution(); + if (!value || typeof value !== "object") return d; + const r = value as Record<string, unknown>; + const str = (v: unknown, fallback: string) => + typeof v === "string" && v.trim() ? v.trim() : fallback; + return { + enabled: r.enabled === true, + appId: str(r.appId, d.appId), + // NOT `str(...)` with a fallback: empty is a meaningful value here ("use the + // app's own model"), so an empty string must survive rather than reverting + // to a default that is also empty by coincidence. + model: typeof r.model === "string" ? r.model.trim() : d.model, + diarizedEnabled: r.diarizedEnabled === true, + textOnlyEnabled: r.textOnlyEnabled === true, + // FLOORED at the shipped constant, never merely defaulted. A hand-edited + // value below it would pin freshness to a superseded prompt generation and + // freeze its output into the corpus — see AttributionSettings.promptVersion. + promptVersion: + typeof r.promptVersion === "number" && Number.isFinite(r.promptVersion) + ? Math.max(ATTRIBUTION_PROMPT_VERSION, Math.floor(r.promptVersion)) + : d.promptVersion, + }; +} + +export function defaultDiarization(): DiarizationSettings { + return { + // OFF. Capture is opt-in: turning it on makes the cleanup sweep start + // refusing to delete audio for transcribed-but-undiarized videos, which is + // correct but is a disk-pressure decision an operator should make. + enabled: false, + // OFF — see DiarizationSettings.inlineAfterTranscribe. Diarization is slower + // than the transcription it would follow, so inline is the exception. + inlineAfterTranscribe: false, + // Measured, not sherpa-onnx's 0.5 — see DiarizationSettings.threshold. The + // constant lives in lib/diarization.ts because isDiarizationFresh needs it + // to normalize an absent recorded threshold; importing it keeps the default + // and the comparator from drifting apart. + threshold: DEFAULT_DIARIZATION_THRESHOLD, + threads: 4, + // The engine every sidecar on disk was produced by. Switching is an explicit + // decision that restates the freshness identity — see DiarizationSettings. + engine: DEFAULT_DIARIZATION_ENGINE, + // Only consulted when engine is "sortformer". Defaulting to the GPU is safe + // because the lane yields the card to transcription rather than sharing it. + backend: "vulkan", + python: "python3", + segModel: "", + embModel: "", + sortformerBin: "", + sortformerModel: "", + concurrency: 1, + // OFF, because windowing made it unnecessary — which is what it was always + // for. It shipped at 4 hours as a stopgap while long recordings were being + // OOM-killed; the engine now processes them in windows and the 6h12m file + // that was being killed at 10.5 GB completes at ~2 GB. The knob stays, and + // stays honest about what it does, for a machine smaller than this one or a + // recording longer than anything measured here. + maxAudioHours: 0, + }; +} + +export function sanitizeDiarization(value: unknown): DiarizationSettings { + const d = defaultDiarization(); + if (!value || typeof value !== "object") return d; + const r = value as Record<string, unknown>; + const str = (v: unknown, fallback: string) => + typeof v === "string" && v.trim() ? v.trim() : fallback; + return { + enabled: r.enabled === true, + inlineAfterTranscribe: r.inlineAfterTranscribe === true, + threshold: + typeof r.threshold === "number" && + Number.isFinite(r.threshold) && + r.threshold > 0 + ? r.threshold + : d.threshold, + threads: clampPositiveInt(r.threads, d.threads, 64), + // An unknown engine falls back to the default rather than disabling the lane: + // a typo in settings.json must not silently stop diarization, and the default + // is the one every existing sidecar already matches. + engine: DIARIZATION_ENGINE_IDS.includes(r.engine as DiarizationEngineId) + ? (r.engine as DiarizationEngineId) + : d.engine, + backend: DIARIZATION_BACKENDS.includes(r.backend as DiarizationBackend) + ? (r.backend as DiarizationBackend) + : d.backend, + python: str(r.python, d.python), + segModel: str(r.segModel, d.segModel), + embModel: str(r.embModel, d.embModel), + sortformerBin: str(r.sortformerBin, d.sortformerBin), + sortformerModel: str(r.sortformerModel, d.sortformerModel), + concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16), + // 0 is meaningful here (cap off), so this cannot use clampPositiveInt. + // Fractional hours are allowed — the knob is a duration, not a count. + maxAudioHours: + typeof r.maxAudioHours === "number" && + Number.isFinite(r.maxAudioHours) && + r.maxAudioHours >= 0 + ? r.maxAudioHours + : d.maxAudioHours, + }; +} + +const SOCIAL_URL_RE = /^(https?:\/\/|mailto:|\/)/i; + +// Normalize the family hub URL into a trailing-slash-free absolute http(s) URL. +// Returns "" for anything that isn't a usable absolute URL (the "no hub" state). +// Kept local to settings.ts to avoid a settings<->homepage import cycle; mirrors +// parseHomepageUrl() in homepage.ts. +export function normalizeHomepageUrl(input: unknown): string { + if (typeof input !== "string") return ""; + const trimmed = input.trim().replace(/\/+$/, ""); + return /^https?:\/\/\S+/i.test(trimmed) ? trimmed : ""; +} + +export function parseSocialLinks(input: unknown): SocialLink[] { + if (!Array.isArray(input)) return []; + const out: SocialLink[] = []; + for (const raw of input) { + if (!raw || typeof raw !== "object") continue; + const r = raw as Record<string, unknown>; + const label = typeof r.label === "string" ? r.label.trim() : ""; + const url = typeof r.url === "string" ? r.url.trim() : ""; + const svg = typeof r.svg === "string" ? r.svg : ""; + if (!label || !url || !svg) continue; + if (!SOCIAL_URL_RE.test(url)) continue; + out.push({ label, url, svg }); + } + return out; +} + +// Normalize an admin-provided SVG snippet for inline use in the export +// footer. Returns null on anything that looks unsafe or unrenderable. +// Steps: trim, allowlist-check, strip width/height, force fill="currentColor" +// + aria-hidden on the root <svg>. Requires a viewBox so the icon scales. +export function normalizeSocialSvg(raw: string): string | null { + if (typeof raw !== "string") return null; + const trimmed = raw.trim(); + if (!trimmed.startsWith("<svg") || !trimmed.endsWith("</svg>")) return null; + if (/<script\b/i.test(trimmed)) return null; + if (/<foreignObject\b/i.test(trimmed)) return null; + if (/<iframe\b/i.test(trimmed)) return null; + if (/javascript:/i.test(trimmed)) return null; + if (/\son[a-z]+\s*=/i.test(trimmed)) return null; + if (/<\?|<!ENTITY/i.test(trimmed)) return null; + + const openEnd = trimmed.indexOf(">"); + if (openEnd < 0) return null; + let opening = trimmed.slice(0, openEnd); + const rest = trimmed.slice(openEnd); + + if (!/\sviewBox\s*=\s*"/i.test(opening)) return null; + + opening = opening.replace(/\s(width|height)\s*=\s*"[^"]*"/gi, ""); + opening = opening.replace(/\s(width|height)\s*=\s*'[^']*'/gi, ""); + + if (!/\sfill\s*=/i.test(opening)) { + opening = opening.replace(/^<svg/i, '<svg fill="currentColor"'); + } + if (!/\saria-hidden\s*=/i.test(opening)) { + opening = opening.replace(/^<svg/i, '<svg aria-hidden="true"'); + } + return opening + rest; +} + +// normalizeSocialSvg() deliberately STRIPS width/height so the icon scales to its +// wrapper. The cost is that a viewBox-only <svg> has no intrinsic size, so before +// the stylesheet loads on a static host it paints at the replaced-element default +// (huge) — the "flash of giant social icons" FOUC. sizeSocialSvg() re-injects an +// intrinsic pixel size at RENDER time (existing site.json files already have the +// attributes stripped, so this must run on read, not just on write). The size is +// an *attribute*, not inline style, so a wrapper's `w-*`/`h-*` utilities still win +// once CSS loads — it only governs the pre-CSS first paint. +export function sizeSocialSvg(svg: string, px = 20): string { + if (typeof svg !== "string") return svg; + if (/^<svg[^>]*\swidth\s*=/i.test(svg)) return svg; // already sized + return svg.replace(/^<svg\b/i, `<svg width="${px}" height="${px}"`); +} + +export function clampSleepBetweenDownloadsSeconds(value: unknown): number { + const n = + typeof value === "number" && Number.isFinite(value) + ? Math.floor(value) + : SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS; + if (n < 0) return 0; + if (n > SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS) { + return SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS; + } + return n; +} + +export function clampMinFreeDiskGB(value: unknown): number { + const n = + typeof value === "number" && Number.isFinite(value) + ? Math.floor(value) + : MIN_FREE_DISK_GB_DEFAULT; + if (n < 0) return 0; + if (n > MIN_FREE_DISK_GB_MAX) return MIN_FREE_DISK_GB_MAX; + return n; +} + +export function clampResumeMarginGB(value: unknown): number { + const n = + typeof value === "number" && Number.isFinite(value) + ? Math.floor(value) + : RESUME_MARGIN_GB_DEFAULT; + if (n < 0) return 0; + if (n > RESUME_MARGIN_GB_MAX) return RESUME_MARGIN_GB_MAX; + return n; +} + +export function clampParallelTranscriptions(value: unknown): number { + const n = + typeof value === "number" && Number.isFinite(value) + ? Math.floor(value) + : PARALLEL_TRANSCRIPTIONS_DEFAULT; + if (n < 1) return 1; + if (n > PARALLEL_TRANSCRIPTIONS_MAX) return PARALLEL_TRANSCRIPTIONS_MAX; + return n; +} + +// 0 means "disabled" and is preserved as-is. Anything else is clamped into the +// [MIN, MAX] window; a non-finite value falls back to the default cadence. +export function clampAutoRefreshIntervalSeconds(value: unknown): number { + if (typeof value !== "number" || !Number.isFinite(value)) { + return AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS; + } + const n = Math.floor(value); + if (n <= 0) return 0; + if (n < AUTO_REFRESH_INTERVAL_MIN_SECONDS) { + return AUTO_REFRESH_INTERVAL_MIN_SECONDS; + } + if (n > AUTO_REFRESH_INTERVAL_MAX_SECONDS) { + return AUTO_REFRESH_INTERVAL_MAX_SECONDS; + } + return n; +} + +export function clampPageBytes(value: unknown): number { + const n = + typeof value === "number" && Number.isFinite(value) + ? value + : TRANSCRIPT_PAGE_DEFAULT_BYTES; + if (n < TRANSCRIPT_PAGE_MIN_BYTES) return TRANSCRIPT_PAGE_MIN_BYTES; + if (n > TRANSCRIPT_PAGE_HARD_CAP_BYTES) return TRANSCRIPT_PAGE_HARD_CAP_BYTES; + return Math.floor(n); +} + + +// Coerce a raw settings.transcriptionApps value into a clean keyed map of +// AppInstanceConfig, dropping unknown/ill-typed fields. +export function sanitizeTranscriptionApps( + value: unknown, +): Record<string, AppInstanceConfig> { + if (!value || typeof value !== "object" || Array.isArray(value)) return {}; + const out: Record<string, AppInstanceConfig> = {}; + for (const [id, raw] of Object.entries(value as Record<string, unknown>)) { + if (!raw || typeof raw !== "object") continue; + out[id] = sanitizeWorkerConfig(raw); + } + return out; +} + + +// --- The schema ------------------------------------------------------------- + +// Global, OPERATIONAL settings shared across every site this editor powers. +// Per-site presentation (branding, social links, channel groups, membership) +// lives in sites/<siteId>/site.json — see common/lib/site.ts. +// +// FIELD ORDER IS FILE ORDER: zod emits keys in the order they are declared, and +// writeSettings writes what the schema emits, so reordering these reorders every +// settings.json on its next save. +export const siteSettingsSchema = z.object({ + adminTitle: settingsField((v): string => + typeof v === "string" && v.trim() ? v.trim() : DEFAULT_ADMIN_TITLE).describe( + "Title for the EDITOR admin shell only (the editor manages all sites and so is not tied to any one site's branding). Public sites get their own titles from site.json.", + ), + maxTranscriptPageBytes: settingsField((v): number => clampPageBytes(v)).describe( + "Target size (bytes) of one exported transcript page shard — the unit the export site fetches. Clamped into [TRANSCRIPT_PAGE_MIN_BYTES, TRANSCRIPT_PAGE_HARD_CAP_BYTES] (256 KiB – 20 MiB); default 8 MiB.", + ), + transcriptionApp: settingsField((v): string => + typeof v === "string" && TRANSCRIPTION_APPS[v] ? v : DEFAULT_TRANSCRIPTION_APP_ID).describe( + "Active transcription app id (key into TRANSCRIPTION_APPS, e.g. \"whisper-cpp\" or \"chough\"). Selected globally; see common/lib/transcriptionApps.ts.", + ), + transcriptionApps: settingsField((v): Record<string, AppInstanceConfig> => sanitizeTranscriptionApps(v)).describe( + "Per-app configuration, keyed by app id. Each app reads only its own block; a missing block means \"use the app's defaults\". DEPRECATED in favor of `workers` (each local worker carries its own config); kept one release to drive migration and allow rollback. See common/lib/workers.ts.", + ), + workers: workersSchema.describe( + "Configured transcription workers (named processing slots). The scheduler distributes each video to the highest-priority free worker. A settings.json predating this field is migrated to a single enabled worker from the active app (see defaultWorkersFromApps). See common/lib/workers.ts.", + ), + cookiesFromBrowser: settingsField((v): string => (typeof v === "string" ? v.trim() : "")).describe( + "Browser spec (e.g. \"firefox\", \"chrome:Default\") passed to `yt-dlp --cookies-from-browser`. WHEN it is passed is governed by `cookieMode` below. Empty string = no cookies configured. Per-channel override available (ChannelConfig.cookiesFromBrowser).", + ), + cookieMode: settingsField((v): CookieMode => (isCookieMode(v) ? v : DEFAULT_COOKIE_MODE)).describe( + "How yt-dlp invocations use the configured cookies (see common/lib/cookiePolicy.ts): \"always\" passes them on every invocation, \"when-required\" (default; the historical behavior) only to retry an auth/age failure, \"defer\" never in normal runs — auth-gated videos are excluded from batches and collected into the per-channel \"Needs cookies\" bucket for a manual cookie run. Per-channel override available (ChannelConfig.cookieMode).", + ), + sleepBetweenDownloadsSeconds: settingsField((v): number => clampSleepBetweenDownloadsSeconds(v)).describe( + "Pause (seconds) inserted between per-video yt-dlp invocations in managed batch downloads. yt-dlp's own `-t sleep` only paces requests within one invocation, so without this the managed loop hammers the source IP back-to-back. 0 disables. Per-channel override available.", + ), + downloadFormat: settingsField((v): DownloadFormatPreset => + isDownloadFormatPreset(v) ? v : "auto").describe( + "Default yt-dlp `-f` download format for every channel that doesn't set its own (ChannelConfig.downloadFormat). \"auto\" picks per-source: `original` for Odysee (whose HLS rungs are CDN-truncated), `bestaudio/worst` elsewhere. See common/ytdlp/downloadFormat.ts.", + ), + minFreeDiskGB: settingsField((v): number => clampMinFreeDiskGB(v)).describe( + "Minimum free disk space (GB) required on the transcripts data directory for downloads to run. When free space is below this floor, a download job is prevented from starting and a running batch stops launching new videos (the in-flight one finishes). 0 disables the gate. See common/lib/diskSpace.ts.", + ), + resumeMarginGB: settingsField((v): number => clampResumeMarginGB(v)).describe( + "Extra headroom (GB) above minFreeDiskGB that a stopped pipeline must see before it resumes. Resuming at the same number we stopped at flaps — the first restarted download pushes free space back under the floor. This is the hysteresis margin, so \"resumed\" means the operator actually freed something rather than a scratch file being cleaned up. 0 disables the hysteresis (resume at the floor). See diskGate() in common/lib/diskSpace.ts.", + ), + parallelTranscriptions: settingsField((v): number => clampParallelTranscriptions(v)).describe( + "Default number of videos transcribed in parallel when a \"Transcribe missing\" / bucket run doesn't specify its own concurrency. The per-run Concurrency input in the channel UI overrides this for a single run.", + ), + inlineTranscribeOnFallback: settingsField((v): boolean => v === true).describe( + "When true, the no-subs fallback in the managed downloader runs whisper inline immediately after the audio download succeeds. When false (default), audio is left for the next \"Transcribe missing\" pass so a batch download finishes faster and whisper can parallelize.", + ), + skipLiveDownloads: settingsField((v): boolean => v !== false).describe( + "When true (default), managed downloads skip videos that are currently live or scheduled/upcoming, decided from a metadata-only prefetch pass. Finished livestream VODs (was_live) are NOT skipped and download normally. A skip is recorded but not archived, so the next sync/download-missing retries the video once the stream ends. Per-channel override available (ChannelConfig.skipLiveDownloads).", + ), + verifyAvailabilityBeforeClean: settingsField((v): boolean => v !== false).describe( + "Whether the transcribed-audio cleanup sweep checks each candidate is still available upstream before deleting its audio, pinning (do-not-clean) any video found permanently gone. The delete is irreversible and a gone video's audio is irreplaceable, so this defaults to true. Turn it off for an offline or URL-less setup, where the check can never resolve and cleanup would otherwise never delete anything. See verifyBeforeClean.ts.", + ), + buildArchives: settingsField((v): boolean => v !== false).describe( + "Whether site builds generate downloadable transcript/live-chat archive zips (into public/archives, linked on the Downloads page). Global default; a site can opt out via site.json `archives: false`, and a single build can skip via the \"Skip archive zips\" build control. Opt-out: default true.", + ), + archiveStorage: settingsField((v): ArchiveStorageSettings => { + const r = (v && typeof v === "object" ? v : {}) as Record<string, unknown>; + return { + bucket: typeof r.bucket === "string" ? r.bucket.trim() : "", + publicBaseUrl: + typeof r.publicBaseUrl === "string" ? r.publicBaseUrl.trim() : "", + }; + }).describe( + "Overflow object storage (Cloudflare R2) for archive zips that exceed the Pages per-file size cap (see Site.archiveMaxBytes). When both fields are set, an oversize archive is uploaded here on deploy — via `wrangler r2 object put`, keyed `<siteId>/archives/<file>` — instead of being dropped, and the Downloads page links to `<publicBaseUrl>/<key>`. Blank/absent → no overflow, so oversize archives stay unavailable (\"Too large to host\").", + ), + reportDebouncePreset: settingsField((v): ReportDebouncePreset => + isReportDebouncePreset(v) ? v : DEFAULT_REPORT_DEBOUNCE_PRESET).describe( + "Debounce preset for the global snapshot scheduler: how long it waits after the last report-changing action before regenerating affected channel reports. See REPORT_DEBOUNCE_PRESETS. Default \"fast\" (~1s, no cap).", + ), + autoRefreshIntervalSeconds: settingsField((v): number => clampAutoRefreshIntervalSeconds(v)).describe( + "How often (seconds) the editor UI passively re-fetches the current page's server-rendered data via router.refresh(), so sidebar badges and reports stay live without a manual reload. Mounted globally; pauses while the tab is hidden. 0 disables passive refresh entirely. See AUTO_REFRESH_INTERVAL_*.", + ), + syncScheduler: settingsField((v): SyncSchedulerSettings => sanitizeSyncScheduler(v)).describe( + "Global configuration for the scheduled (cron-driven) channel sync system. The per-channel cadence lives on ChannelConfig.syncIntervalMinutes; this block holds the defaults and guard rails the scheduler applies across all channels. See common/jobs/syncScheduler.ts.", + ), + autoQueue: autoQueueSchema.describe( + "Configuration for the automatic priority-queue runners (auto-transcribe / auto-download). Each holds a tree policy that decides which channel's video to process next, cross-channel, by priority/round-robin/weighted-fair rules. Independent of syncScheduler (which decides staleness, not work order). See common/jobs/autoQueuePolicy.ts.", + ), + channelPriority: channelPrioritySchema.describe( + "THE OPERATOR-FACING PRIORITY MODEL: one tier per channel plus one corpus-wide focus selector. It is the SOURCE the four `autoQueue[lane].root` trees are compiled from (common/lib/channelPriority.ts), not a second mechanism beside them — and its `paused` tier is the one part that is not a tree shape, filtering the runner's channel list instead. An empty document (the default) is today's behaviour exactly: no focus, every channel normal, the stored trees stand.", + ), + socialLinks: settingsField((v): SocialLink[] => parseSocialLinks(v)).describe( + "Default social links applied to every site that doesn't define its own. A site inherits these unless its site.json carries an explicit `socialLinks` array — see Site.socialLinks / resolveSocialLinks in common/lib/site.ts. The one presentation field that lives globally so a shared footer doesn't have to be repeated per site.", + ), + homepageUrl: settingsField((v): string => normalizeHomepageUrl(v)).describe( + "Absolute public URL of the family hub/homepage (e.g. \"https://archilyzer.pages.dev\"). Every export site links back to it (\"the family\" backlink) when set. Empty = no hub link rendered. Normalized to a trailing-slash-free http(s) URL.", + ), + savedVideoBackup: settingsField((v): SavedVideoBackupSettings => sanitizeSavedVideoBackup(v)).describe( + "Backup configuration for the saved-video store (Phase 4 of the video-persistence feature). When enabled with a destination, the store is mirrored there (additively, no deletes) with a per-backup manifest, and the sync scheduler runs the backup on the configured cadence. See common/controller/backupSavedVideos.ts.", + ), + storage: settingsField((v): StorageSettings => sanitizeStorage(v)).describe( + "Where a channel's downloaded media goes when it is relocated off the corpus disk. A DEFAULT ONLY: the relocate controller never reads it and always takes an explicit root, so this is the value the per-channel Storage panel prefills and the /channels bulk move falls back to. Blank = no default. See StorageSettings.", + ), + buildPipeline: settingsField((v): BuildPipelineSettings => sanitizeBuildPipeline(v)).describe( + "How the static export is built: \"basic\" reuses the single export/ tree and serializes builds on one queue (the long-standing behavior); \"docker\" runs each site's build in an isolated container for safe parallelism. The Docker pipeline itself is a follow-up; this block persists the chosen mode plus the container/concurrency knobs the deploy page and the future orchestrator read.", + ), + digest: settingsField((v): DigestSettings => sanitizeDigest(v)).describe( + "AI digest generation (chapters + topic tags over the existing transcripts). Local-first: the metered lane is off by default. See DigestSettings.", + ), + diarization: settingsField((v): DiarizationSettings => sanitizeDiarization(v)).describe( + "Speaker diarization captured right after transcription, while the audio is still on disk. OFF by default. See DiarizationSettings.", + ), + backfill: settingsField((v): BackfillSettings => sanitizeBackfill(v)).describe( + "The generic catch-up lane for derived data the existing corpus predates. OFF by default, and idle-only when on. See BackfillSettings.", + ), + attribution: settingsField((v): AttributionSettings => sanitizeAttribution(v)).describe( + "Naming the speakers diarization found (or reconstructing them from the transcript when it found none). OFF by default. See AttributionSettings.", + ), +}); + +export type SiteSettings = z.infer<typeof siteSettingsSchema>; + +// The whole default settings object, without touching disk: the schema's answer +// for an empty file. Not a second literal — a default that lived anywhere but in +// the field's own coercion would be a second place to change it. +export function defaults(): SiteSettings { + return siteSettingsSchema.parse({}); +} + +// Exported under this name too, because tests and e2e helpers already say it: +// a caller that needs a settings-SHAPED value rather than the operator's actual +// configuration builds one here without a settings.json. +export function defaultSiteSettings(): SiteSettings { + return defaults(); +} diff --git a/common/lib/storageLocations.ts b/common/lib/storageLocations.ts @@ -1,4 +1,5 @@ import path from "node:path"; +import type { FieldDocs } from "./fieldDocs"; // STORAGE LOCATIONS — the named places a channel's media may live. // @@ -31,57 +32,88 @@ import path from "node:path"; export const INTERNAL_LOCATION_ID = "internal"; export const INTERNAL_LOCATION_LABEL = "Internal (in place)"; +// Each field is documented in STORAGE_VOLUME_FIELD_DOCS below (rendered into SETTINGS.md). export type StorageVolume = { - // Filesystem UUID, the one stable name a disk has across mountpoints. This is - // what makes "the platter came up somewhere else" a recoverable situation. uuid: string; fstype?: string; label?: string; - // Where the volume was mounted at the last successful probe, and the path of - // the location's root RELATIVE to that mountpoint. Invariant: - // `root === join(mountpoint, relPath)`. Keeping the two halves is what lets a - // probe compute a candidate root when the volume reappears elsewhere. mountpoint: string; relPath: string; }; +export const STORAGE_VOLUME_FIELD_DOCS: FieldDocs<StorageVolume> = { + uuid: + "Filesystem UUID, the one stable name a disk has across mountpoints. " + + "This is what makes \"the platter came up somewhere else\" a recoverable " + + "situation.", + fstype: + "Filesystem type reported by the probe (e.g. \"ext4\"). Informational; omitted when unknown.", + label: + "Filesystem label reported by the probe. Informational; omitted when unknown.", + mountpoint: + "Where the volume was mounted at the last successful probe, and the " + + "path of the location's root RELATIVE to that mountpoint. Invariant: " + + "`root === join(mountpoint, relPath)`. Keeping the two halves is what " + + "lets a probe compute a candidate root when the volume reappears " + + "elsewhere.", + relPath: + "The location root's path RELATIVE to `mountpoint` (see there). Invariant: `root === join(mountpoint, relPath)`.", +}; + +// Each field is documented in STORAGE_LOCATION_FIELD_DOCS below (rendered into SETTINGS.md). export type StorageLocation = { - // /^[a-z0-9][a-z0-9-]{0,63}$/, unique within the list. Stable: it is what - // `defaultLocationId` and every form and action refer to. id: string; - // Human name. Blank sanitizes to the id. label: string; - // Absolute directory, trailing "/" stripped. NEVER existence-checked on read - // — the whole point of a cold location is a drive that may not be mounted - // when settings are parsed. root: string; - // Opt-in: when the volume is found mounted somewhere else, re-point without - // asking (if the preflight passes). Off by default — re-point rewrites every - // channel symlink on the location, and that is not something to do silently - // unless the operator asked for it. autoRepoint: boolean; - // Identity learned at the last successful probe. Optional because a location - // may never have been probed, and because in a container block devices are - // invisible and identity is permanently unknown. volume?: StorageVolume; }; +export const STORAGE_LOCATION_FIELD_DOCS: FieldDocs<StorageLocation> = { + id: + "/^[a-z0-9][a-z0-9-]{0,63}$/, unique within the list. Stable: it is " + + "what `defaultLocationId` and every form and action refer to.", + label: + "Human name. Blank sanitizes to the id.", + root: + "Absolute directory, trailing \"/\" stripped. NEVER existence-checked on " + + "read — the whole point of a cold location is a drive that may not be " + + "mounted when settings are parsed.", + autoRepoint: + "Opt-in: when the volume is found mounted somewhere else, re-point " + + "without asking (if the preflight passes). Off by default — re-point " + + "rewrites every channel symlink on the location, and that is not " + + "something to do silently unless the operator asked for it.", + volume: + "Identity learned at the last successful probe. Optional because a " + + "location may never have been probed, and because in a container block " + + "devices are invisible and identity is permanently unknown.", +}; + +// Each field is documented in STORAGE_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). export type StorageSettings = { locations: StorageLocation[]; - // The location prefilled as the destination of a move. "" = no default. defaultLocationId: string; - // WHERE THE SAVED-VIDEO STORE IS, by location id. "" = in place, under the - // corpus at `paths.savedVideosDir`. - // - // A RECORD OF WHAT IS ON DISK, never an intention — the same contract as a - // channel's `config.dataDir`. It is written by the move, on success, after - // the copy has verified and the symlink is in place; nothing else writes it, - // and a reader that disagrees with the disk trusts the disk. Optional so an - // older settings.json parses (and an older binary that drops it leaves a - // store that still works, because the symlink is what every reader follows). savedVideosLocationId?: string; }; +export const STORAGE_SETTINGS_FIELD_DOCS: FieldDocs<StorageSettings> = { + locations: + "The named storage locations a channel's media may be relocated to — one entry per root, each with an id, label, root, `autoRepoint` and the learned volume identity. Order is display order. Managed on /storage.", + defaultLocationId: + "The location prefilled as the destination of a move. \"\" = no default.", + savedVideosLocationId: + "WHERE THE SAVED-VIDEO STORE IS, by location id. \"\" = in place, under " + + "the corpus at `paths.savedVideosDir`.\n\n" + + "A RECORD OF WHAT IS ON DISK, never an intention — the same contract as" + + " a channel's `config.dataDir`. It is written by the move, on success, " + + "after the copy has verified and the symlink is in place; nothing else " + + "writes it, and a reader that disagrees with the disk trusts the disk. " + + "Optional so an older settings.json parses (and an older binary that " + + "drops it leaves a store that still works, because the symlink is what " + + "every reader follows).", +}; + // Strip trailing slashes so "/mnt/platter/" and "/mnt/platter" are one root. // The sanitizer does this on write too; this is here so a hand-edited // settings.json still compares correctly. diff --git a/common/lib/transcriptionApps.ts b/common/lib/transcriptionApps.ts @@ -17,29 +17,45 @@ import { createChoughProgressParser, createParakeetProgressParser, } from "../jobs/progressParsers"; +import type { FieldDocs } from "./fieldDocs"; export type TranscriptOutputFormat = "whisper-json" | "chough-json" | "vtt"; // Per-app configuration persisted under settings.transcriptionApps[id]. Every // field is optional; an app falls back to its own defaults (defaultBin, env). +// Each field is documented in APP_INSTANCE_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md). export type AppInstanceConfig = { - // Binary path/name override. Empty/undefined falls back to app.defaultBin(). bin?: string; - // whisper.cpp model path (substituted for {model}); for chough this is the - // optional CHOUGH_MODEL env (chough auto-downloads a model when unset). model?: string; - // chough remote server URL (CHOUGH_URL). Empty/undefined = local transcription. remoteUrl?: string; - // chough chunk size in seconds (-c). Undefined = chough's own default. chunkSize?: number; - // whisper.cpp custom argv template using the {audioFile}/{outputBase}/{model} - // placeholders. Undefined = DEFAULT_TRANSCRIBE_ARGS. customArgs?: string[]; - // parakeet compute device passed to parakeet-cli (--device / PARAKEET_DEVICE), - // e.g. "cuda:0", "cpu". Undefined = parakeet-cli's default device. device?: string; }; +export const APP_INSTANCE_CONFIG_FIELD_DOCS: FieldDocs<AppInstanceConfig> = { + bin: + "Binary path/name override. Empty/undefined falls back to " + + "app.defaultBin().", + model: + "whisper.cpp model path (substituted for {model}); for chough this is " + + "the optional CHOUGH_MODEL env (chough auto-downloads a model when " + + "unset).", + remoteUrl: + "chough remote server URL (CHOUGH_URL). Empty/undefined = local " + + "transcription.", + chunkSize: + "chough chunk size in seconds (-c). Undefined = chough's own default.", + customArgs: + "whisper.cpp custom argv template using the " + + "{audioFile}/{outputBase}/{model} placeholders. Undefined = " + + "DEFAULT_TRANSCRIBE_ARGS.", + device: + "parakeet compute device passed to parakeet-cli (--device / " + + "PARAKEET_DEVICE), e.g. \"cuda:0\", \"cpu\". Undefined = parakeet-cli's " + + "default device.", +}; + export type TranscribeBuild = { // Args passed after the binary. argv: string[]; diff --git a/common/lib/workers.ts b/common/lib/workers.ts @@ -19,6 +19,7 @@ import { getTranscriptionApp, validateTranscribeArgs, } from "./transcriptionApps"; +import type { FieldDocs } from "./fieldDocs"; export type WorkerKind = "local" | "remote" | "llm"; @@ -36,32 +37,50 @@ export type WorkerKind = "local" | "remote" | "llm"; // "close enough": its output would be permanently-stale. The dispatcher // verifies the tag against /api/tags before first use and degrades the worker // when it is missing (controller/llmWorkers.ts). +// Each field is documented in LLM_WORKER_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md). export type LlmWorkerConfig = { - baseUrl: string; // e.g. http://macbook.lan:11434 - // Concurrent generations to allow this endpoint. Defaults to 1 — one model - // instance, one generation — unless the operator knows better. + baseUrl: string; slots?: number; }; +export const LLM_WORKER_CONFIG_FIELD_DOCS: FieldDocs<LlmWorkerConfig> = { + baseUrl: + "e.g. http://macbook.lan:11434", + slots: + "Concurrent generations to allow this endpoint. Defaults to 1 — one " + + "model instance, one generation — unless the operator knows better.", +}; + // Where a remote worker delegates. The remote runs its OWN worker pool and picks // among ITS local workers — so the primary stores only how to reach it, not which // engine to use. +// Each field is documented in REMOTE_WORKER_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md). export type RemoteWorkerConfig = { - baseUrl: string; // e.g. http://gpu-box.lan:3011 - // Outbound bearer token sent with every /api/worker request to this remote. - // The accepting side validates against its own WORKER_TOKEN env, never this. + baseUrl: string; token?: string; - // When true the remote shares the transcripts mount, so we send - // {channelSlug, videoId} instead of uploading the audio bytes. sharedFs?: boolean; - // How many units this remote takes in parallel. The pool expands one remote - // config into this many independently-schedulable slot entries at - // reconfigure time (the defaultWorkersFromApps trick, applied live). Absent = - // probed from the remote's /api/worker/health (its enabled worker count) — - // see controller/remoteCapacity.ts; 1 until the probe answers. slots?: number; }; +export const REMOTE_WORKER_CONFIG_FIELD_DOCS: FieldDocs<RemoteWorkerConfig> = { + baseUrl: + "e.g. http://gpu-box.lan:3011", + token: + "Outbound bearer token sent with every /api/worker request to this " + + "remote. The accepting side validates against its own WORKER_TOKEN env," + + " never this.", + sharedFs: + "When true the remote shares the transcripts mount, so we send " + + "{channelSlug, videoId} instead of uploading the audio bytes.", + slots: + "How many units this remote takes in parallel. The pool expands one " + + "remote config into this many independently-schedulable slot entries at" + + " reconfigure time (the defaultWorkersFromApps trick, applied live). " + + "Absent = probed from the remote's /api/worker/health (its enabled " + + "worker count) — see controller/remoteCapacity.ts; 1 until the probe " + + "answers.", +}; + // A worker is ONE processing slot — one transcription at a time. To run N in // parallel on the same engine, define N workers (the settings editor's "Copy" // button duplicates one). This makes each slot independently togglable on the @@ -71,34 +90,53 @@ export type RemoteWorkerConfig = { // probed capacity): the pool expands it into N slot entries itself, because // asking the operator to hand-copy a remote once per slot of a machine whose // slot count the machine already reports would be busywork. +// Each field is documented in WORKER_FIELD_DOCS below (rendered into SETTINGS.md). export type Worker = { - // Stable slug; used in settings, task ids, and logs. id: string; - // Human label shown in the UI. name: string; kind: WorkerKind; enabled: boolean; - // Lower = preferred. Ties broken by array order in the scheduler. priority: number; - // Capability routing. A tag is an OPERATION id from the backfill catalog - // ("diarization", "attribution-text", …) or a contended RESOURCE - // (WORKER_RESOURCE_TAGS). The scheduler consults them through workerMatches - // below: an untagged worker takes anything, a tagged worker takes only work - // whose requirement intersects its tags. Unknown tags are tolerated (they - // match nothing and warn in the settings UI), never fatal. tags?: string[]; - // LOCAL: an instance of a TRANSCRIPTION_APPS entry + its per-worker config. appId?: string; config?: AppInstanceConfig; - // REMOTE: how to reach the delegate instance. remote?: RemoteWorkerConfig; - // LLM: how to reach the bare model endpoint. llm?: LlmWorkerConfig; }; +export const WORKER_FIELD_DOCS: FieldDocs<Worker> = { + id: + "Stable slug; used in settings, task ids, and logs.", + name: + "Human label shown in the UI.", + kind: + "\"local\" runs an app from TRANSCRIPTION_APPS on this machine (`appId` + `config`); \"remote\" delegates to another instance on the LAN (`remote`); \"llm\" is a bare ollama endpoint serving digest/attribution calls only (`llm`).", + enabled: + "Whether the scheduler may hand this slot work. Each worker is one slot, so parallelism is toggled per slot on the Workers page. Anything but an explicit `false` reads as enabled.", + priority: + "Lower = preferred. Ties broken by array order in the scheduler.", + tags: + "Capability routing. A tag is an OPERATION id from the backfill catalog" + + " (\"diarization\", \"attribution-text\", …) or a contended RESOURCE " + + "(WORKER_RESOURCE_TAGS). The scheduler consults them through " + + "workerMatches below: an untagged worker takes anything, a tagged " + + "worker takes only work whose requirement intersects its tags. Unknown " + + "tags are tolerated (they match nothing and warn in the settings UI), " + + "never fatal.", + appId: + "LOCAL: an instance of a TRANSCRIPTION_APPS entry + its per-worker " + + "config.", + config: + "LOCAL: the per-worker engine config (binary, model, device, …) — an AppInstanceConfig, see `transcriptionApps.<appId>`.", + remote: + "REMOTE: how to reach the delegate instance.", + llm: + "LLM: how to reach the bare model endpoint.", +}; + // The contended-resource half of the tag vocabulary — Lane.contendsFor's // three values, restated here because this module must stay client-safe and the // lane type lives in operations.ts, whose import graph reaches controllers. diff --git a/common/package.json b/common/package.json @@ -65,7 +65,8 @@ "recharts": "2.15.4", "sonner": "^2.0.7", "tailwind-merge": "^3.6.0", - "tw-animate-css": "^1.4.0" + "tw-animate-css": "^1.4.0", + "zod": "^4.3.6" }, "peerDependencies": { "next": "16.2.3", diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **`settings.json` has one schema and one writer, and its key table is generated.** Every key, its default, its clamp and its documentation is now one zod schema (`common/lib/settingsSchema.ts`); `getSettings`/`writeSettings` both parse through it, and every settings form saves through one helper (`editor/app/settings/saveSettings.ts`) that merges only what the form changed. **`SETTINGS.md`** (new, repo root) lists every key with its default and what it does, and `settings.json.example` is now the full default object — both generated by `common/bin/settings-example.ts` and checked by a test, so neither can drift. Nothing an operator has configured reads differently. **Fixed:** adding or editing a storage location on `/storage` no longer erases the record of which location the saved-video store is on (`storage.savedVideosLocationId`). - **Every live panel now polls one endpoint, `/api/view/<name>`, and the eight old addresses still answer.** The change token, the job head, workers, the operations board, the sync console and the widget's three strips were eight separate API routes that each did the same thing; they are one route serving eight named views (`pulse`, `activeJobs`, `workers`, `autoQueueStatus`, `schedulerStatus`, `widgetSync`, `widgetActionable`, `cleanable`), and the editor's own pages poll it there. `/api/pulse`, `/api/jobs/active`, `/api/workers`, `/api/auto-queue/status`, `/api/scheduler/status` and `/api/widget/{sync,actionable,cleanable}` are kept as **rewrites**, not redirects — same method, status, body and query string (`?rev=` included) — so a monitor widget pinned in a browser, or any script polling the old path, keeps working untouched. `/api/widget/presets` is unchanged. An unknown view name is a 404. **One behaviour change you might notice:** the operations board (every 3 s) and the sync console (every 5 s) now send their next poll only after the previous one answers, and abandon a poll that takes longer than 10–15 s — so a slow editor no longer piles requests up behind itself, and a hung request no longer stops the page updating. - **A lane's pause is one key on the lane, and the four old pause fields are gone from `settings.json`.** Holding a lane has been `autoQueue.<lane>.held` since the runner work landed; until now the file also still carried the four flags that used to mean it — `transcriptionsPaused`, `downloadsPaused`, `digest.digestsPaused` and the backwards `backfill.enabled` (where *enabled* meant *not held*) — which were read only when a lane had no `held` yet, to carry an older file's pause across. Every lane now carries its own key, so those four are **deleted**: nothing reads them, no form writes them, and the next settings save drops them from the file. A settings.json that still spells one of them holds nothing with it, so a hand-edited file (or a very old backup restored over a newer one) can no longer resurrect a pause you had lifted, or lift one you had set. "Run the backfill lane" on the diarization page and the Hold/Pause buttons write the one key, as they already did. **UPGRADING: boot once on the release that writes `held` before taking this one.** That release is the one that moved the gate onto the lane and carried the old fields across on read; a single boot of it (any settings save, or just starting the editor and pausing/resuming anything) puts `autoQueue.<lane>.held` in your settings.json, after which **nothing you can see changes here** — the same buttons, the same labels, the same pauses. An install that jumps straight from an older release to this one has no `held` keys at all and **loses its pauses**: transcription, downloads and digests come up running, and the backfill lane comes up held. Re-set them from the dashboard, or add the keys by hand before starting. - **A relocate job says how far it has got.** `rsync` has been printing its progress the whole time (`--info=progress2`) and every frame of it went into the job log as a carriage-return redraw of one line — so a 131 GB move and a 3 MB one looked identical from `/jobs`: a spinner. Now each frame is parsed into the **task bar** every other long job on that page already draws, reading `12.3 GB of 45.6 GB · 27 % · 110.50MB/s · ETA 5:32`, and the log gets **one line per 10 %** instead of several thousand frames of one. The percentage is against the tree the job already measured for its space check, not rsync's own — under incremental recursion that one is a percentage of what it has enumerated so far and walks backwards. diff --git a/editor/app/channels/[slug]/incompleteTranscriptActions.ts b/editor/app/channels/[slug]/incompleteTranscriptActions.ts @@ -7,7 +7,8 @@ import { resolveQueueKey, } from "yt-dlp-transcript-common/lib/queueKeys"; import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"; -import { getSettings, writeSettings } from "yt-dlp-transcript-common/lib/settings"; +import { getSettings } from "yt-dlp-transcript-common/lib/settings"; +import { saveSettings } from "../../settings/saveSettings"; import { startAutoRunner } from "yt-dlp-transcript-common/controller/autoRunner"; import { runManagedFunction, @@ -40,8 +41,7 @@ export async function enableAutoRunners(): Promise<void> { !current.autoQueue.transcription.enabled || !current.autoQueue.download.enabled ) { - await writeSettings({ - ...current, + await saveSettings({ autoQueue: { ...current.autoQueue, transcription: { ...current.autoQueue.transcription, enabled: true }, diff --git a/editor/app/channels/actions.ts b/editor/app/channels/actions.ts @@ -35,10 +35,8 @@ import { siteChannelIndex, type Site, } from "yt-dlp-transcript-common/lib/site"; -import { - getSettings, - writeSettings, -} from "yt-dlp-transcript-common/lib/settings"; +import { getSettings } from "yt-dlp-transcript-common/lib/settings"; +import { saveSettings } from "../settings/saveSettings"; import { LANES, type AutoQueueKind, @@ -944,7 +942,10 @@ export async function saveChannelPriorityAction( autoQueue[lane] = { ...autoQueue[lane], enabled: true }; } try { - await writeSettings({ ...settings, channelPriority: next, autoQueue }); + // BOTH BLOCKS IN ONE WRITE — the reason saveSettings takes a whole-settings + // patch rather than one block. `channelPriority` is a full document (its + // `channels` map replaces), and `autoQueue` carries all four lanes. + await saveSettings({ channelPriority: next, autoQueue }); } catch (e) { return { error: (e as Error).message }; } diff --git a/editor/app/jobs/actions.ts b/editor/app/jobs/actions.ts @@ -2,10 +2,8 @@ import { revalidatePath } from "next/cache"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; -import { - getSettings, - writeSettings, -} from "yt-dlp-transcript-common/lib/settings"; +import { getSettings } from "yt-dlp-transcript-common/lib/settings"; +import { saveSettings } from "../settings/saveSettings"; import { laneRootFromScope } from "yt-dlp-transcript-common/lib/laneMigration"; import { isDefaultChannelPriority } from "yt-dlp-transcript-common/lib/channelPriority"; import { operationsForLane } from "yt-dlp-transcript-common/lib/operations"; @@ -216,8 +214,7 @@ export async function armLaneAction( }; } const policy = settings.autoQueue[lane]; - await writeSettings({ - ...settings, + await saveSettings({ autoQueue: { ...settings.autoQueue, [lane]: { @@ -276,8 +273,7 @@ export async function disarmLaneAction( ): Promise<ArmLaneResult> { try { const settings = getSettings(); - await writeSettings({ - ...settings, + await saveSettings({ autoQueue: { ...settings.autoQueue, [lane]: { ...settings.autoQueue[lane], enabled: false }, diff --git a/editor/app/operations/actions.ts b/editor/app/operations/actions.ts @@ -3,9 +3,9 @@ import { revalidatePath } from "next/cache"; import { getSettings, - writeSettings, type SiteSettings, } from "yt-dlp-transcript-common/lib/settings"; +import { saveSettings } from "../settings/saveSettings"; import { isGateHeld, withGateHeld, @@ -94,8 +94,7 @@ export async function saveAutoQueueAction( // in slice 1.4, is exactly what "forgotten here" would now mean. `undefined` // is carried as undefined on purpose: that is what keeps a lane that has // never been written falling back to its retired field. - const next: SiteSettings = { - ...current, + const next: Partial<SiteSettings> = { autoQueue: { ...current.autoQueue, [kind]: { @@ -110,7 +109,7 @@ export async function saveAutoQueueAction( }, }; try { - await writeSettings(next); + await saveSettings(next); } catch (e) { return { ok: false, error: (e as Error).message }; } @@ -153,15 +152,14 @@ export async function snoozeAutoQueueAction( untilMs: number | null, ): Promise<SaveResult> { const current = getSettings(); - const next: SiteSettings = { - ...current, + const next: Partial<SiteSettings> = { autoQueue: { ...current.autoQueue, [kind]: { ...current.autoQueue[kind], snoozeUntil: untilMs }, }, }; try { - await writeSettings(next); + await saveSettings(next); } catch (e) { return { ok: false, error: (e as Error).message }; } @@ -199,7 +197,7 @@ async function setLaneHeld( try { const cur = getSettings(); if (isGateHeld(cur, lane) !== held) { - await writeSettings(withGateHeld(cur, lane, held)); + await saveSettings({ autoQueue: withGateHeld(cur, lane, held).autoQueue }); } } catch (e) { // REPORTED, not swallowed. The workers page's old best-effort persist diff --git a/editor/app/operations/settingsActions.ts b/editor/app/operations/settingsActions.ts @@ -6,7 +6,9 @@ // `*FormPresent` marker, because unchecked checkboxes are simply ABSENT from a // FormData and a submit from a form lacking the block would read every switch // as off. One form per block makes the marker unnecessary — the `...current` -// spread at the top of each block is now the whole isolation story. +// spread at the top of each block is now the whole isolation story, and +// `saveSettings` (../settings/saveSettings.ts) merges each patch onto the +// rest of the file. // // Every field name is the one the old single form used, and the persisted keys // are unchanged: this is a move, not a redesign. @@ -14,9 +16,9 @@ import { revalidatePath } from "next/cache"; import { getSettings, - writeSettings, type SiteSettings, } from "yt-dlp-transcript-common/lib/settings"; +import { saveSettings } from "../settings/saveSettings"; import { isDigestSectionKind, isDigestTimestampMode, @@ -62,7 +64,7 @@ export async function saveDigestSettingsAction( return { ok: false, error: "Digest app config payload is malformed" }; } } - // Filtered to KNOWN kinds here rather than leaning on writeSettings' sanitizer. + // Filtered to KNOWN kinds here rather than leaning on the settings schema's sanitizer. // sanitizeDigest would drop an unknown value anyway, but typing it honestly is // what lets the digest block below be checked against DigestSettings instead of // cast — and the cast is what hid the dropped-fields bug. @@ -73,8 +75,7 @@ export async function saveDigestSettingsAction( const digestTimestampModeRaw = String( formData.get("digestTimestampMode") ?? "", ).trim(); - const next: SiteSettings = { - ...current, + const next: Partial<SiteSettings> = { digest: { // Spread the CURRENT block first. Every field this form does not render // must survive a save untouched, and before this spread they did not: @@ -119,7 +120,7 @@ export async function saveDigestSettingsAction( }, }; try { - await writeSettings(next); + await saveSettings(next); } catch (e) { return { ok: false, error: (e as Error).message }; } @@ -139,8 +140,7 @@ export async function saveDiarizationSettingsAction( ): Promise<SaveResult> { const current = getSettings(); const dDiar = current.diarization; - const next: SiteSettings = { - ...current, + const next: Partial<SiteSettings> = { diarization: { ...dDiar, enabled: formData.get("diarizationEnabled") === "on", @@ -179,7 +179,7 @@ export async function saveDiarizationSettingsAction( }, }; try { - await writeSettings(next); + await saveSettings(next); } catch (e) { return { ok: false, error: (e as Error).message }; } @@ -210,20 +210,20 @@ export async function saveBackfillLaneSettingsAction( ): Promise<SaveResult> { const current = getSettings(); const dBack = current.backfill; - const next: SiteSettings = withGateHeld( - { - ...current, - backfill: { - ...dBack, - concurrency: num(formData, "backfillConcurrency", dBack.concurrency), - allowRedownload: formData.get("backfillAllowRedownload") === "on", - }, + const next: Partial<SiteSettings> = { + backfill: { + ...dBack, + concurrency: num(formData, "backfillConcurrency", dBack.concurrency), + allowRedownload: formData.get("backfillAllowRedownload") === "on", }, - "backfill", - formData.get("backfillEnabled") !== "on", - ); + autoQueue: withGateHeld( + current, + "backfill", + formData.get("backfillEnabled") !== "on", + ).autoQueue, + }; try { - await writeSettings(next); + await saveSettings(next); } catch (e) { return { ok: false, error: (e as Error).message }; } @@ -245,8 +245,7 @@ export async function saveAttributionSettingsAction( ): Promise<SaveResult> { const current = getSettings(); const dAttr = current.attribution; - const next: SiteSettings = { - ...current, + const next: Partial<SiteSettings> = { attribution: { ...dAttr, enabled: formData.get("attributionEnabled") === "on", @@ -265,7 +264,7 @@ export async function saveAttributionSettingsAction( }, }; try { - await writeSettings(next); + await saveSettings(next); } catch (e) { return { ok: false, error: (e as Error).message }; } diff --git a/editor/app/saved-videos/backupActions.ts b/editor/app/saved-videos/backupActions.ts @@ -2,7 +2,8 @@ import { revalidatePath } from "next/cache"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; -import { getSettings, writeSettings } from "yt-dlp-transcript-common/lib/settings"; +import { getSettings } from "yt-dlp-transcript-common/lib/settings"; +import { saveSettings } from "../settings/saveSettings"; import { SYNC_INTERVAL_MAX_MINUTES } from "yt-dlp-transcript-common/lib/channelConfig"; import { backupSavedVideos, @@ -101,8 +102,7 @@ export async function saveSavedVideoBackupAction( } const settings = getSettings(); try { - await writeSettings({ - ...settings, + await saveSettings({ savedVideoBackup: { enabled, dest, diff --git a/editor/app/scheduler/actions.ts b/editor/app/scheduler/actions.ts @@ -13,9 +13,9 @@ import { } from "yt-dlp-transcript-common/lib/channelConfig"; import { getSettings, - writeSettings, type SiteSettings, } from "yt-dlp-transcript-common/lib/settings"; +import { saveSettings } from "../settings/saveSettings"; import { DURATION_KEEP } from "yt-dlp-transcript-common/lib/duration"; export type SaveResult = { ok: true } | { ok: false; error: string }; @@ -114,7 +114,7 @@ export async function setChannelCadencesAction( // block is the sync operation's settings form now, on the sync operation's // page, and this is what saves it. // -// Values are clamped/sanitized by sanitizeSyncScheduler inside writeSettings, +// Values are clamped/sanitized by sanitizeSyncScheduler when saveSettings writes, // so we only coerce here. export async function saveSchedulerSettingsAction( _prev: SaveResult | undefined, @@ -143,8 +143,7 @@ export async function saveSchedulerSettingsAction( const n = Number.parseInt(raw, 10); return Number.isFinite(n) ? n : null; }; - const next: SiteSettings = { - ...current, + const next: Partial<SiteSettings> = { syncScheduler: { ...current.syncScheduler, enabled: formData.get("syncSchedulerEnabled") === "on", @@ -189,7 +188,7 @@ export async function saveSchedulerSettingsAction( }, }; try { - await writeSettings(next); + await saveSettings(next); } catch (e) { return { ok: false, error: (e as Error).message }; } diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts @@ -8,7 +8,6 @@ import { defaultBuildPipeline, isBuildMode, isReportDebouncePreset, - getSettings, MIN_FREE_DISK_GB_MAX, RESUME_MARGIN_GB_DEFAULT, RESUME_MARGIN_GB_MAX, @@ -18,11 +17,10 @@ import { SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS, TRANSCRIPT_PAGE_HARD_CAP_BYTES, TRANSCRIPT_PAGE_MIN_BYTES, - writeSettings, type SiteSettings, type SocialLink, } from "yt-dlp-transcript-common/lib/settings"; -import { DEFAULT_TRANSCRIPTION_APP_ID } from "yt-dlp-transcript-common/lib/transcriptionApps"; +import { saveSettings } from "./saveSettings"; import { DEFAULT_COOKIE_MODE, isCookieMode, @@ -182,7 +180,7 @@ export async function saveSettingsAction( } // Build pipeline. Values are clamped/coerced by sanitizeBuildPipeline inside - // writeSettings, so we only read the form here (NaN/blank → default). The + // the settings schema on save, so we only read the form here (NaN/blank → default). The // deploy-page toggle also writes `mode`; whichever saves last wins. const dB = defaultBuildPipeline(); const buildModeRaw = String(formData.get("buildMode") ?? "").trim(); @@ -198,22 +196,25 @@ export async function saveSettingsAction( String(formData.get("dockerfile") ?? "").trim() || dB.dockerfile, }; - const next: SiteSettings = { + // ONLY WHAT THIS FORM EDITS. saveSettings merges the patch over the stored + // file, so every block edited elsewhere — workers (/workers), syncScheduler + // (/operations/sync), autoQueue and channelPriority (the operation pages and + // /channels), savedVideoBackup (/saved-videos), storage (/storage) and the + // four operation blocks (/operations/<id>) — survives a save here without + // being read and re-listed. Before slice 4a this literal named all 31 fields + // and carried each foreign block through `getSettings()`, which is how an + // unrelated save once disarmed a sweep. + const next: Partial<SiteSettings> = { adminTitle, maxTranscriptPageBytes: parsed, - // Workers are the source of truth; writeSettings derives the deprecated - // transcriptionApp/transcriptionApps shadow from them. parallelTranscriptions - // is vestigial (kept only for rollback) so pass the default. - transcriptionApp: DEFAULT_TRANSCRIPTION_APP_ID, - transcriptionApps: {}, - // workers: edited on /workers, beside the live list, by saveWorkersAction. - workers: getSettings().workers, cookiesFromBrowser, cookieMode, sleepBetweenDownloadsSeconds: sleepParsed, downloadFormat, minFreeDiskGB: minFreeDiskParsed, resumeMarginGB: resumeMarginParsed, + // Vestigial (the worker list is the parallelism; kept only for rollback), + // and this form has always reset it to the default. parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT, inlineTranscribeOnFallback, skipLiveDownloads, @@ -222,43 +223,12 @@ export async function saveSettingsAction( archiveStorage, reportDebouncePreset, autoRefreshIntervalSeconds: autoRefreshParsed, - // syncScheduler: edited on /operations/sync, below the schedule, by - // saveSchedulerSettingsAction. - syncScheduler: getSettings().syncScheduler, - // Preserve the existing auto-queue policy on an unrelated settings save - // (this form doesn't edit it; the Auto-queue page does). writeSettings - // re-sanitizes it regardless. - autoQueue: getSettings().autoQueue, - // Preserved for the same reason, and for one more: it is the SOURCE the - // four roots above are compiled from, so rebuilding it here would silently - // undo a focus. Edited on /channels by saveChannelPriorityAction, which is - // its one writer. - channelPriority: getSettings().channelPriority, socialLinks, homepageUrl, - // Preserve the saved-video backup config on an unrelated settings save (the - // Saved Videos page edits it). writeSettings re-sanitizes it regardless. - savedVideoBackup: getSettings().savedVideoBackup, - // NO LONGER THIS FORM'S. The storage block became a list of named - // locations edited on /storage; preserved here the way autoQueue and - // channelPriority are, so an unrelated settings save cannot erase it. - // writeSettings re-sanitizes it regardless. - storage: getSettings().storage, buildPipeline, - // Each edited on its own operation page; preserved here. Slice 3 moved - // these four fieldsets to /operations/<id>, and with them the hidden - // `*FormPresent` markers that used to gate them — a marker only existed - // because one form saved everything, and reading an absent checkbox as - // `false` could disarm a sweep, drop the diarization capture lane, flip - // `allowRedownload` on, or arm ~194,000 model calls. Reading the CURRENT - // block is now the whole protection, exactly as it is for autoQueue. - digest: getSettings().digest, - diarization: getSettings().diarization, - backfill: getSettings().backfill, - attribution: getSettings().attribution, }; try { - await writeSettings(next); + await saveSettings(next); } catch (e) { return { ok: false, error: (e as Error).message }; } diff --git a/editor/app/settings/saveSettings.test.ts b/editor/app/settings/saveSettings.test.ts @@ -0,0 +1,78 @@ +import { mkdtempSync, readFileSync, writeFileSync } from "node:fs"; +import { rm } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { test, after } from "node:test"; +import assert from "node:assert/strict"; + +// Run with: pnpm -C editor exec tsx --test "app/**/*.test.ts" +// +// THE MERGE RULE of the editor's one settings writer: ONE LEVEL DEEP. An object +// block merges over the stored block's keys; arrays and scalars replace; an +// object nested INSIDE a block replaces. Same rule as editor/e2e/helpers.ts. +// +// SETTINGS SEAM as in common/lib/settingsWrite.test.ts: getPaths() memoizes, so +// SETTINGS_FILE is set before the module under test is imported. + +const ROOT = mkdtempSync(path.join(os.tmpdir(), "save-settings-")); +process.env.TRANSCRIPTS_DIR = ROOT; +process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); + +const { mergeSettingsPatch, saveSettings } = await import("./saveSettings"); +const { defaultSiteSettings, getSettings } = await import( + "yt-dlp-transcript-common/lib/settings" +); + +after(() => rm(ROOT, { recursive: true, force: true })); + +test("an object block merges over the stored block, one level", () => { + const base = defaultSiteSettings(); + base.syncScheduler.maxConcurrentSyncs = 7; + const out = mergeSettingsPatch(base, { + syncScheduler: { enabled: true } as typeof base.syncScheduler, + }); + assert.equal(out.syncScheduler.enabled, true); + assert.equal(out.syncScheduler.maxConcurrentSyncs, 7); + // The input is not mutated. + assert.equal(base.syncScheduler.enabled, false); +}); + +test("arrays and scalars replace", () => { + const base = defaultSiteSettings(); + base.socialLinks = [{ label: "a", url: "/a", svg: "<svg></svg>" }]; + const out = mergeSettingsPatch(base, { socialLinks: [], adminTitle: "X" }); + assert.deepEqual(out.socialLinks, []); + assert.equal(out.adminTitle, "X"); + assert.equal(out.minFreeDiskGB, base.minFreeDiskGB); +}); + +test("an object nested inside a block replaces, it is not merged", () => { + const base = defaultSiteSettings(); + base.digest.apps = { a: { model: "m1", numCtx: 4096 } }; + base.channelPriority.channels = { keep: { tier: "low" } }; + const out = mergeSettingsPatch(base, { + digest: { apps: { b: { model: "m2" } } } as unknown as typeof base.digest, + channelPriority: { + focus: { kind: "none" }, + channels: { other: { tier: "paused" } }, + }, + }); + assert.deepEqual(out.digest.apps, { b: { model: "m2" } }); + assert.equal(out.digest.localAppId, base.digest.localAppId); + assert.deepEqual(out.channelPriority.channels, { other: { tier: "paused" } }); +}); + +test("saveSettings writes the merged result and touches nothing else", async () => { + writeFileSync( + process.env.SETTINGS_FILE!, + JSON.stringify({ adminTitle: "Kept", minFreeDiskGB: 9 }), + ); + await saveSettings({ minFreeDiskGB: 3, backfill: { concurrency: 4 } as never }); + const s = getSettings(); + assert.equal(s.adminTitle, "Kept"); + assert.equal(s.minFreeDiskGB, 3); + assert.equal(s.backfill.concurrency, 4); + assert.equal(s.backfill.allowRedownload, false); + const onDisk = JSON.parse(readFileSync(process.env.SETTINGS_FILE!, "utf8")); + assert.equal(onDisk.adminTitle, "Kept"); +}); diff --git a/editor/app/settings/saveSettings.ts b/editor/app/settings/saveSettings.ts @@ -0,0 +1,58 @@ +import { + getSettings, + writeSettings, + type SiteSettings, +} from "yt-dlp-transcript-common/lib/settings"; + +// THE EDITOR'S ONE SETTINGS WRITER (one-core phase 3 slice 4a). +// +// Every server action that changes settings.json calls `saveSettings(patch)` +// with only what it changed; this reads the current settings, merges the +// patch, and hands the result to `writeSettings` — which is imported by this +// file and by no other file under editor/app. +// +// THE MERGE IS ONE LEVEL DEEP, and no deeper — the same rule e2e's +// `writeSettings` helper (editor/e2e/helpers.ts) applies to the fixture: +// +// - a patch key whose value is a PLAIN OBJECT merges over the current block's +// keys: `{ syncScheduler: { enabled: true } }` changes one flag and keeps the +// rest of the scheduler; +// - ARRAYS and SCALARS replace: `workers: []` means no workers; +// - an object NESTED INSIDE a block replaces: `autoQueue: { digest: {…} }` +// replaces the digest lane whole (a half-merged policy tree would be a +// worse surprise than a replaced one), and `channelPriority: { channels }` +// replaces the whole channel map. +// +// WHY A PATCH OF THE WHOLE SETTINGS OBJECT, NOT `saveSettingsBlock(block, +// patch)` as plans/one-core.md sketched: the channel-priority action writes +// `channelPriority` AND the four `autoQueue` roots compiled from it in ONE +// write. A one-block signature would split that into two, and a reader between +// them would see a priority model and trees that disagree. +// +// NOT a "use server" module: it is a helper the action files call, never an +// action a client can invoke with an arbitrary patch. + +type Patch = Partial<SiteSettings>; + +function isPlainObject(v: unknown): v is Record<string, unknown> { + return typeof v === "object" && v !== null && !Array.isArray(v); +} + +// Pure, so the merge rule is testable without a settings file. +export function mergeSettingsPatch( + current: SiteSettings, + patch: Patch, +): SiteSettings { + const merged: Record<string, unknown> = { ...current }; + for (const [key, value] of Object.entries(patch)) { + if (value === undefined) continue; + const base = (current as Record<string, unknown>)[key]; + merged[key] = + isPlainObject(value) && isPlainObject(base) ? { ...base, ...value } : value; + } + return merged as SiteSettings; +} + +export async function saveSettings(patch: Patch): Promise<void> { + await writeSettings(mergeSettingsPatch(getSettings(), patch)); +} diff --git a/editor/app/sites/lib/buildModeAction.ts b/editor/app/sites/lib/buildModeAction.ts @@ -4,9 +4,9 @@ import { revalidatePath } from "next/cache"; import { getSettings, isBuildMode, - writeSettings, type BuildMode, } from "yt-dlp-transcript-common/lib/settings"; +import { saveSettings } from "../../settings/saveSettings"; // Persist the build-mode choice (Basic vs Docker) from the family page (/sites) // so it becomes the default for every subsequent build. Only the mode is touched @@ -17,10 +17,8 @@ export async function setBuildModeAction( ): Promise<{ ok: boolean }> { if (!isBuildMode(mode)) return { ok: false }; const settings = getSettings(); - await writeSettings({ - ...settings, - buildPipeline: { ...settings.buildPipeline, mode }, - }); + // One key of one block: saveSettings merges it over the stored pipeline. + await saveSettings({ buildPipeline: { ...settings.buildPipeline, mode } }); revalidatePath("/sites"); return { ok: true }; } diff --git a/editor/app/storage/actions.ts b/editor/app/storage/actions.ts @@ -3,10 +3,8 @@ import path from "node:path"; import { revalidatePath } from "next/cache"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; -import { - getSettings, - writeSettings, -} from "yt-dlp-transcript-common/lib/settings"; +import { getSettings } from "yt-dlp-transcript-common/lib/settings"; +import { saveSettings } from "../settings/saveSettings"; import type { StorageLocation } from "yt-dlp-transcript-common/lib/storageLocations"; import { mountByUuid, @@ -105,8 +103,11 @@ export async function addStorageLocationAction( autoRepoint: draft.autoRepoint, }; const locations = [...settings.storage.locations, location]; - await writeSettings({ - ...settings, + // A PATCH of the storage block: saveSettings merges these keys over the + // stored block, so `savedVideosLocationId` — which this action does not + // edit — survives the save. (Before slice 4a it was rebuilt from these two + // keys alone, and adding or editing a location erased the store's record.) + await saveSettings({ storage: { locations, // The FIRST location is the default whether or not the box was ticked: @@ -152,8 +153,8 @@ export async function editStorageLocationAction( }; if (!keepVolume) delete (updated as { volume?: unknown }).volume; - await writeSettings({ - ...settings, + // A patch of the storage block — `savedVideosLocationId` survives (see above). + await saveSettings({ storage: { locations: settings.storage.locations.map((l) => (l.id === id ? updated : l)), defaultLocationId: draft.makeDefault @@ -214,8 +215,8 @@ export async function deleteStorageLocationAction( }; } const locations = settings.storage.locations.filter((l) => l.id !== id); - await writeSettings({ - ...settings, + // A patch of the storage block — `savedVideosLocationId` survives (see above). + await saveSettings({ storage: { locations, defaultLocationId: diff --git a/editor/app/workers/actions.ts b/editor/app/workers/actions.ts @@ -3,7 +3,7 @@ import { revalidatePath } from "next/cache"; import { getWorkerPool } from "yt-dlp-transcript-common/jobs/workerPool"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; -import { getSettings, writeSettings } from "yt-dlp-transcript-common/lib/settings"; +import { saveSettings } from "../settings/saveSettings"; import { sanitizeWorkers, validateWorkers, @@ -38,7 +38,7 @@ export async function saveWorkersAction( formData: FormData, ): Promise<SaveWorkersResult> { // Workers are submitted as a JSON array by the WorkersField client component. - // Sanitize + validate here for a friendly error; writeSettings re-validates. + // Sanitize + validate here for a friendly error; writeSettings re-validates on save. let workersInput: unknown; try { workersInput = JSON.parse(String(formData.get("workersJson") ?? "[]")); @@ -50,7 +50,7 @@ export async function saveWorkersAction( if (workersErr) return { ok: false, error: workersErr }; try { - await writeSettings({ ...getSettings(), workers }); + await saveSettings({ workers }); } catch (e) { return { ok: false, error: (e as Error).message }; } diff --git a/plans/one-core-phase-3.md b/plans/one-core-phase-3.md @@ -577,3 +577,110 @@ Deviations: Not done here, by design: no batch route, no auth on views, `operations/status.ts`, `requestCache.ts` and `liveInputs.ts` untouched. The full editor suite was not run; only the 20-spec subset above. + +## Slice 4a, as shipped — one settings schema (2026-09-23) + +Branch `one-core/phase-3-s4a` off `main` `54cf1b31`, unmerged. Commits (shas after the +trailer rewrite; this record is `plans:` commit 6 on top): + +| sha | what | +|---|---| +| `50215ab5` | zod `^4.3.6` joins `common` (lockfile +3 lines, no new resolution); the auto-queue defaults/clamps/tree normalisation/lane gate move from `jobs/autoQueuePolicy.ts` to `lib/autoQueueSchema.ts` (picker stays, re-exports every moved name); allow-list entry `lib/settings.ts -> jobs/autoQueuePolicy` burned, **10 → 9**; `autoQueuePolicy.test.ts` repointed (imports only), 59/59; `plans/tools/phase3-settings-numbers.ts` added | +| `e120a2a5` | `workersSchema`, `channelPrioritySchema`, `autoQueueSchema` — zod seams over the existing sanitizers, in `lib/settingsFieldSchemas.ts` | +| `f1abe024` | `lib/settingsSchema.ts`: `siteSettingsSchema` (31 fields, `.describe()` on each), `SiteSettings = z.infer`, `defaults()` = `defaultSiteSettings()` = `parse({})`; `lib/settings.ts` reduced to I/O (1,783 → 250 lines) and `export *`s the schema module; `settingsSchema.test.ts` | +| `85478527` | `editor/app/settings/saveSettings.ts` + unit test; 19 call sites in 11 files converted to patches; `writeSettings` is imported by one editor file | +| `0316e988` | `common/bin/settings-example.ts` (+`--check`), `lib/settingsDocs.ts`, generated `settings.json.example` + `SETTINGS.md`, `settingsDocs.test.ts`; SETUP.md points at SETTINGS.md | + +**What moved.** Every type, constant, clamp and block sanitizer that was in `lib/settings.ts` +is in `lib/settingsSchema.ts` and re-exported, so no importer changed. `getSettings` = +`finishRawMigrations(siteSettingsSchema.parse(premigrateRaw(raw)), raw)`: sweeps→lanes and +mediaRoot→locations rewrite the raw input (keyed on the raw file's absence); legacy +`transcribe*`→app registry and worker synthesis run after the parse, keyed on the raw +object's `transcriptionApp` / `workers` absence. `writeSettings` = +`parse({...deriveWorkerShadow(next), socialLinks: validatedSocialLinks(...)})` + tmp/rename; +the two validators still throw. Every field is `z.unknown().catch(undefined).transform(coerce)` +over the pre-existing clamp or sanitizer; no `.passthrough()`, no `.default()`. + +**Deviations from the spec, and why.** +1. *The zod seams are not beside their sanitizers.* `workers.ts`, `channelPriority.ts` and + (through `jobs/autoQueuePolicy`) `autoQueueSchema.ts` are value-imported by `"use client"` + forms (WorkersConfigForm, ChannelTierSelect, LadderRung, …); a zod import there would ship + zod to the browser, contradicting "zod cannot reach a client bundle". The three schemas + live in `lib/settingsFieldSchemas.ts` (server-only importers); the sanitizers keep their + homes, names and signatures. Verified: no `ZodError`/`_zod` in `editor/.next/static` or + `export/.next/static` after both builds. +2. *`archiveStorage` lost its `?`.* zod 4 cannot express an optional key that is always + emitted (`.optional()` omits it when absent; a transform returning `T | undefined` is a + required key). It was always emitted at runtime, so the shape test pins it as required; + tsc across the workspace needed no change. +3. *`saveSettings(patch)` not `saveSettingsBlock(block, patch)`* — as instructed: + `channels/actions.ts` writes `channelPriority` and all four `autoQueue` roots in one write. +4. *The example omits `workers`.* `defaultSiteSettings().workers` is `[]`; a copied template + spelling `workers: []` would mean zero transcription slots, where an absent key + synthesizes one. Stated in SETTINGS.md. +5. *Only the 31 top-level comments moved into `.describe()`.* The nested block types + (`DigestSettings`, `DiarizationSettings`, …) keep their per-field comments on the + hand-written types, because each block is one sanitizer-backed field, not a zod object. + +**Behaviour changes (all intended, all small).** +- A settings.json containing `null` threw in `getSettings` (`parsed[key]` on null); it now + reads as the empty file, like `[]`, `3` and `{`. +- `adminTitle`, `cookiesFromBrowser`, `archiveStorage.*` are trimmed on read as they always + were on write (one schema). The live file has no untrimmed values. +- `storage/actions.ts`: adding/editing a location used to rebuild the storage block from two + keys and so **erased `storage.savedVideosLocationId`**; the one-level merge keeps it. +- `/settings` form (`editor/app/settings/actions.ts:207`): it used to pass + `transcriptionApp: DEFAULT` + `transcriptionApps: {}` with the stored workers. When the + stored list was `[]`, main's worker shadow therefore synthesized a default whisper-cpp + worker with NO config; the branch patches only the form's own fields, so the shadow + synthesizes from the STORED app and its stored config (better). And nothing now prunes + stale per-app entries from the `transcriptionApps` shadow — the old `{}` did — since the + shadow is rollback-only and still rewritten from workers on every save (accepted). + +**Dead example keys removed**: `transcribeBin`, `transcribeModel`, `transcribeArgs` (the +pre-multi-app spelling, migrated on read). + +**Numbers** (`plans/tools/phase3-settings-numbers.ts`, `getSettings()` sorted-key JSON). +The live settings.json was re-saved at 19:24 mid-slice — the operator's Gate B change, applied +through the backfill lane form (lane held, `allowRedownload` off; that save also stripped the +retired `backfill.enabled`, as every save does), not a stray write — so the +comparison runs both builds over the SAME frozen inputs: the live file as of 19:24, the +pre-slice example, the e2e fixture, and both `docker/entrypoint.sh` seeds (parakeet, +whisper). Main `54cf1b31` vs branch tip: **empty diff, 3,846 lines**. After review the tool +also prints what `writeSettings(getSettings())` puts on disk — written to a scratch copy +under `os.tmpdir()`, never the measured file (the frozen inputs' md5s were re-checked +after the run). Re-run over the same frozen inputs: **read AND write both diff-empty, 7,691 +lines**, no write threw. The expected +`backfill.enabled` line never appeared: `sanitizeBackfill` already dropped it at main, so it +was not in `getSettings()` output before or after. The regenerated example of course parses +differently from the old one (no legacy `whisper-cli`/`firefox` keys) — by design. + +**Entrypoint seed.** Both seeds (`workers[0]` enabled local parakeet / whisper-cpp, +`parallelTranscriptions: 1`) parse to byte-identical settings through main and through the +schema (included in the numbers above). The entrypoint does not read the example. + +**Review fix (one commit after commit 6).** SETTINGS.md now documents every +NESTED key, not only the 31 top-level ones: each block type carries a +`<TYPE>_FIELD_DOCS: FieldDocs<Type>` record beside it (`lib/fieldDocs.ts`; the mapped type +requires one entry per key, optional keys and every union member's keys included, so an +undocumented new field is a tsc error). The per-field comments moved out of the types into +those records — 24 records across `settingsSchema.ts` (the seven blocks + `SocialLink` + +the newly named `ArchiveStorageSettings`), `storageLocations.ts` (settings, location, +volume), `workers.ts` (worker, remote, llm), `transcriptionApps.ts` (`AppInstanceConfig`), +`digest.ts` (`DigestAppConfig`), `autoQueueTypes.ts` (policy, tree node, match) and +`channelPriority.ts` (document, focus, entry, and `autoPaused`, now the named type +`ChannelAutoPause`). `settingsDocs.ts` renders each as a key · default · description table +under its block (lane-policy defaults per lane, so `held`'s `[false,false,false,true]` is +visible). Also: the `settingsField` comment no longer implies zod guards a throwing +sanitizer (`z.unknown().catch` cannot fire — totality is each coercion's); the docs say +`workers: []` means no transcription only until the next save; SETTINGS.md warns that a +copied example pins every default, `held` included. + +**Gates.** tsc (`pnpm -r --workspace-concurrency=1 exec tsc --noEmit`; the parallel `-r` form +was OOM-killed, exit 137) clean after every commit. common **1625 → 1648** (+20 schema, +3 +docs); `test:scripts` 156 pass + 1 skip of 157 (unchanged); mcp 219/219; editor unit **59 → 63**; +`next build` editor and export clean. +**e2e** (from the worktree root, detached, ports 3311/3310): auto-queue, backfill, digest, +diarization, attribution, scheduler, cadence-ui, storage-locations, channel-storage, workers, +worker-remote, parakeet, parakeet-partial, chough, transcription-app-migration, disk-space, +channel-priority, settings — **157/157 passed, exit 0, 9.9 min**, first run, nothing re-run. diff --git a/plans/tools/phase3-settings-numbers.ts b/plans/tools/phase3-settings-numbers.ts @@ -0,0 +1,167 @@ +#!/usr/bin/env tsx +// The one-core Phase 3 slice 4a measurement: what `getSettings()` ANSWERS, for +// every settings.json this repo can point at, printed deterministically so two +// runs can be diffed. +// +// WHY THIS IS A SCRIPT AND NOT A TEST, same as phase1-numbers.ts next door. +// Slice 4a replaces ten hand-written sanitizers with one zod schema. The claim +// it makes is "nothing an operator has configured reads differently", and the +// only way to check that is to parse the REAL files — the live corpus's +// settings.json, the shipped example, the e2e fixture — before and after, and +// diff two files. A test would have to carry the operator's configuration. +// +// NEVER WRITES A MEASURED FILE. Each target is copied to a scratch directory +// under os.tmpdir(); the read and the write-back both happen on the copy, which +// is deleted afterwards. (It measures the WRITE side too since slice 4a's +// review: what `writeSettings(getSettings())` puts on disk.) +// +// NEVER BOOT AN EDITOR FOR THIS. `getSettings` is called in-process, offline; +// instrumentation.ts is not loaded, so no runner, sweep or scheduler is armed. +// +// ONE PROCESS PER FILE, because `getPaths()` memoises its answer at module +// scope: `SETTINGS_FILE` has to be set before `lib/settings.ts` is imported, so +// a second file needs a second process. The parent below spawns itself once per +// target; the child prints one block. +// +// Usage, from the repo root: +// node_modules/.bin/tsx plans/tools/phase3-settings-numbers.ts +// node_modules/.bin/tsx plans/tools/phase3-settings-numbers.ts live=/abs/settings.json +// +// Each argv entry is `label=path`. Given any, they REPLACE the default list; +// the label is what the output names, so a file copied elsewhere (an archived +// "before" example, say) can still be diffed against its original line for line. + +import { spawnSync } from "node:child_process"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const REPO = path.resolve(HERE, "..", ".."); + +type Target = { label: string; file: string }; + +// THE LIVE CORPUS'S settings.json, not this checkout's. A worktree carries its +// own copy, and the file that matters is the one the editor actually runs on. +// Overridable so the script is not pinned to one machine's layout. +function liveSettingsFile(): string { + return ( + process.env.LIVE_SETTINGS_FILE ?? + path.join( + path.dirname(REPO), + "yt-dlp-transcript-browser", + "settings.json", + ) + ); +} + +function defaultTargets(): Target[] { + const out: Target[] = [ + { label: "live", file: liveSettingsFile() }, + { label: "example", file: path.join(REPO, "settings.json.example") }, + ]; + const fixtures = path.join(REPO, "editor", "e2e", "fixtures"); + for (const name of fs.readdirSync(fixtures).sort()) { + if (!/settings.*\.json$/i.test(name)) continue; + out.push({ label: `fixture:${name}`, file: path.join(fixtures, name) }); + } + return out; +} + +// Sort every object's keys so the output is diffable regardless of the order a +// sanitizer (or a schema) happens to build its result in. Arrays keep their +// order — in settings.json order IS data (workers are priority-ordered, and an +// auto-queue tree's children compete in the order they are listed). +function sortedKeys(_key: string, value: unknown): unknown { + if (!value || typeof value !== "object" || Array.isArray(value)) return value; + const src = value as Record<string, unknown>; + const out: Record<string, unknown> = {}; + for (const k of Object.keys(src).sort()) out[k] = src[k]; + return out; +} + +// READ, THEN WRITE — BOTH AGAINST A SCRATCH COPY. The target is copied into a +// fresh directory under os.tmpdir() and SETTINGS_FILE points at the copy, so +// `writeSettings` — which writes `getPaths().settingsFile` — can never touch +// the file being measured (the live settings.json included). The read is of +// identical bytes; the write side is what a save of that reading puts on disk. +async function child(file: string): Promise<void> { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), "phase3-settings-")); + const scratch = path.join(dir, "settings.json"); + fs.copyFileSync(file, scratch); + process.env.SETTINGS_FILE = scratch; + process.env.TRANSCRIPTS_DIR = dir; + try { + const { getSettings, writeSettings } = await import( + "../../common/lib/settings" + ); + const read = getSettings(); + console.log(JSON.stringify(read, sortedKeys, 2)); + console.log("### written by writeSettings(getSettings())"); + try { + await writeSettings(read); + const written = JSON.parse(fs.readFileSync(scratch, "utf8")); + console.log(JSON.stringify(written, sortedKeys, 2)); + } catch (e) { + console.log(`WRITE THREW: ${(e as Error).message}`); + } + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } +} + +function parent(targets: Target[]): void { + console.log("# one-core phase 3 slice 4a — getSettings() and writeSettings() over every settings file"); + console.log(""); + for (const { label, file } of targets) { + console.log(`## ${label}`); + if (!fs.existsSync(file)) { + console.log("MISSING"); + console.log(""); + continue; + } + const res = spawnSync( + process.execPath, + [ + path.join(REPO, "node_modules", "tsx", "dist", "cli.mjs"), + fileURLToPath(import.meta.url), + ], + { + cwd: REPO, + encoding: "utf8", + env: { + ...process.env, + PHASE3_SETTINGS_TARGET: file, + // The snooze sanitizer compares against the clock, and a storage + // probe would shell out. Neither is settings data; neither is read + // here. (`sanitizeSnooze` self-clears a lapsed snooze, which is + // stable as long as nothing is snoozed — noted, not worked around.) + TZ: "UTC", + }, + }, + ); + if (res.status !== 0) { + console.log(`FAILED status=${res.status}`); + console.log(res.stderr.trim()); + } else { + console.log(res.stdout.trimEnd()); + } + console.log(""); + } +} + +const target = process.env.PHASE3_SETTINGS_TARGET; +if (target) { + await child(target); +} else { + const args = process.argv.slice(2); + const targets: Target[] = args.length + ? args.map((a) => { + const eq = a.indexOf("="); + if (eq < 0) return { label: path.basename(a), file: path.resolve(a) }; + return { label: a.slice(0, eq), file: path.resolve(a.slice(eq + 1)) }; + }) + : defaultTargets(); + parent(targets); +} diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml @@ -86,6 +86,9 @@ importers: tw-animate-css: specifier: ^1.4.0 version: 1.4.0 + zod: + specifier: ^4.3.6 + version: 4.3.6 devDependencies: '@types/d3-scale': specifier: ^4.0.9 diff --git a/settings.json.example b/settings.json.example @@ -1,19 +1,182 @@ { "adminTitle": "Transcript Browser Admin", "maxTranscriptPageBytes": 8388608, - "transcribeBin": "whisper-cli", - "transcribeModel": "~/whispercpp/whisper.cpp/models/ggml-base.en.bin", - "transcribeArgs": [ - "-ojf", - "-l", - "en", - "-m", - "{model}", - "-of", - "{outputBase}", - "{audioFile}" - ], - "cookiesFromBrowser": "firefox", + "transcriptionApp": "whisper-cpp", + "transcriptionApps": {}, + "cookiesFromBrowser": "", + "cookieMode": "when-required", "sleepBetweenDownloadsSeconds": 10, - "inlineTranscribeOnFallback": false + "downloadFormat": "auto", + "minFreeDiskGB": 5, + "resumeMarginGB": 2, + "parallelTranscriptions": 2, + "inlineTranscribeOnFallback": false, + "skipLiveDownloads": true, + "verifyAvailabilityBeforeClean": true, + "buildArchives": true, + "archiveStorage": { + "bucket": "", + "publicBaseUrl": "" + }, + "reportDebouncePreset": "fast", + "autoRefreshIntervalSeconds": 5, + "syncScheduler": { + "enabled": false, + "defaultIntervalMinutes": 1440, + "maxConcurrentSyncs": 2, + "quietHoursStart": null, + "quietHoursEnd": null, + "backoffBaseMinutes": 30, + "backoffMaxMinutes": 1440, + "heartbeatSeconds": 0, + "keepLatestCheckIntervalMinutes": 1440, + "fullSweepIntervalMinutes": 1440, + "fullSweepConfirmMaxSuspects": 25, + "fullSweepShrinkGuardPercent": 10 + }, + "autoQueue": { + "transcription": { + "enabled": false, + "maxWorkers": null, + "replaceAutoSubs": false, + "order": "listed", + "snoozeUntil": null, + "held": false, + "root": { + "id": "root", + "mode": "strict", + "weight": 1, + "maxWorkers": null, + "children": [] + } + }, + "download": { + "enabled": false, + "maxWorkers": null, + "replaceAutoSubs": false, + "order": "listed", + "snoozeUntil": null, + "held": false, + "root": { + "id": "root", + "mode": "strict", + "weight": 1, + "maxWorkers": null, + "children": [] + } + }, + "digest": { + "enabled": false, + "maxWorkers": null, + "replaceAutoSubs": false, + "order": "cheapest", + "snoozeUntil": null, + "held": false, + "root": { + "id": "root", + "mode": "strict", + "weight": 1, + "maxWorkers": null, + "children": [ + { + "id": "all", + "match": { + "type": "all" + }, + "weight": 1, + "maxWorkers": null + } + ] + } + }, + "backfill": { + "enabled": false, + "maxWorkers": null, + "replaceAutoSubs": false, + "order": "listed", + "snoozeUntil": null, + "held": true, + "root": { + "id": "root", + "mode": "strict", + "weight": 1, + "maxWorkers": null, + "children": [ + { + "id": "all", + "match": { + "type": "all" + }, + "weight": 1, + "maxWorkers": null + } + ] + } + } + }, + "channelPriority": { + "focus": { + "kind": "none" + }, + "channels": {} + }, + "socialLinks": [], + "homepageUrl": "", + "savedVideoBackup": { + "enabled": false, + "dest": "", + "intervalMinutes": 1440 + }, + "storage": { + "locations": [], + "defaultLocationId": "" + }, + "buildPipeline": { + "mode": "basic", + "maxParallelBuilds": 2, + "dockerImage": "yt-dlp-transcript-browser-build", + "dockerfile": "Dockerfile.build" + }, + "digest": { + "remoteEnabled": false, + "longTailSeconds": 14400, + "localAppId": "ollama-direct", + "remoteAppId": "claude-code", + "apps": {}, + "yieldToTranscription": true, + "yieldToCpuWorkers": false, + "spendCapUsd": 0, + "sections": [ + "chapters" + ], + "timestampMode": "chunk-local", + "promptVariant": "" + }, + "diarization": { + "enabled": false, + "inlineAfterTranscribe": false, + "threshold": 0.9, + "threads": 4, + "engine": "sherpa-onnx", + "backend": "vulkan", + "python": "python3", + "segModel": "", + "embModel": "", + "sortformerBin": "", + "sortformerModel": "", + "concurrency": 1, + "maxAudioHours": 0 + }, + "backfill": { + "concurrency": 1, + "allowRedownload": false + }, + "attribution": { + "enabled": false, + "appId": "ollama-direct", + "model": "", + "diarizedEnabled": false, + "textOnlyEnabled": false, + "promptVersion": 1 + } }