commit 8b5aa2cd6315d4fe1fa829cb66249cce08f7c183
parent af7c34c3c944660cd902009b395e9ab6d560b05c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 23 Sep 2026 19:30:00 -0400
settings: one schema — lib/settingsSchema.ts; getSettings/writeSettings parse through it
one-core phase 3 slice 4a, commit 3. `siteSettingsSchema` (zod) is the one
definition of settings.json: 31 fields, each `settingsField(coerce)` over
the clamp or sanitizer that already existed, each carrying the comment that
used to sit on the hand-written type as its `.describe()`. `SiteSettings`
is `z.infer` of it, re-exported from lib/settings.ts under the same name;
`defaults()` / `defaultSiteSettings()` are `parse({})`.
lib/settings.ts keeps only I/O: getSettings = finishRawMigrations(
parse(premigrateRaw(raw)), raw) — the three absence-keyed migrations still
key on the RAW file — and writeSettings = parse(deriveWorkerShadow(next))
plus the two throwing validators and the tmp+rename write. Everything else
moved to settingsSchema.ts and is re-exported (`export *`).
Behaviour: identical on the live settings.json, the example and the e2e
fixture (plans/tools/phase3-settings-numbers.ts, frozen inputs, empty diff).
Two deliberate changes: a file containing `null` no longer throws on read;
adminTitle / cookiesFromBrowser / archiveStorage are trimmed on read as they
always were on write. `archiveStorage` loses its vestigial `?` (always
emitted; zod 4 cannot express an optional key that is always present).
settingsSchema.test.ts: shape pinned against the pre-schema type, clamp
boundaries, strip, held defaults, each migration fires only on absence,
getSettings never throws on "{", "null", "[]", "3". common 1625 -> 1645.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
3 files changed, 1979 insertions(+), 1661 deletions(-)
diff --git a/common/lib/settings.ts b/common/lib/settings.ts
@@ -1,11 +1,30 @@
+// THE ONE READER AND THE ONE WRITER OF settings.json.
+//
+// The shape — every field, its default, its clamp, its documentation — is
+// `siteSettingsSchema` in ./settingsSchema.ts (one-core phase 3 slice 4a), and
+// everything that module exports is re-exported here, so the ~200 importers of
+// `lib/settings` (types, constants, clamps, sanitizers) did not move.
+//
+// What is left in this file is exactly what a schema cannot do:
+//
+// - READ STAYS LENIENT. A settings.json is whatever an operator, an older
+// build, or a half-finished write left behind. getSettings never throws:
+// an unreadable or non-object file reads as the empty one, and every field
+// is total. Three migrations are keyed on a field's ABSENCE in the raw file
+// — which a parsed object cannot see, because parsing folds the default in
+// — so they run around the parse, each handed the RAW object.
+// - WRITE STAYS STRICT. writeSettings derives the worker shadow, runs the two
+// validators that THROW (a worker list that cannot transcribe, a social
+// link whose SVG is unsafe), parses through the same schema — which is what
+// drops every key it does not name, retired fields included — and writes
+// atomically (tmp + rename).
+//
+// One schema, both directions: the only differences between what a read and a
+// write produce are those migrations and those two validators.
+
import fs from "node:fs";
import path from "node:path";
import { getPaths } from "./paths";
-import { SYNC_INTERVAL_MAX_MINUTES } from "./channelConfig";
-import {
- isDownloadFormatPreset,
- type DownloadFormatPreset,
-} from "../ytdlp/downloadFormat";
import {
type AppInstanceConfig,
DEFAULT_TRANSCRIBE_ARGS,
@@ -13,1586 +32,98 @@ import {
TRANSCRIPTION_APPS,
} from "./transcriptionApps";
import {
- type Worker,
defaultWorkersFromApps,
- sanitizeWorkerConfig,
sanitizeWorkers,
validateWorkers,
} from "./workers";
-import type { AutoQueueSettings } from "./autoQueueTypes";
-import {
- defaultChannelPriority,
- sanitizeChannelPriority,
- type ChannelPriority,
-} from "./channelPriority";
import { migrateSweepsToLanes } from "./laneMigration";
+import { migrateMediaRootToLocations } from "./storageLocations";
import {
- INTERNAL_LOCATION_ID,
- migrateMediaRootToLocations,
- type StorageLocation,
- type StorageSettings,
- type StorageVolume,
-} from "./storageLocations";
-// The auto-queue's half of the settings schema, in lib/ with the rest of it
-// since one-core phase 3 slice 4a. It used to come from `jobs/autoQueuePolicy`,
-// which was the last back-edge on ../architecture.test.ts's allow-list.
-import {
- defaultAutoQueue,
- sanitizeAutoQueue,
-} from "./autoQueueSchema";
-import {
- DEFAULT_DIARIZATION_ENGINE,
- DEFAULT_DIARIZATION_THRESHOLD,
- DIARIZATION_BACKENDS,
- DIARIZATION_ENGINE_IDS,
- type DiarizationBackend,
- type DiarizationEngineId,
-} from "./diarization";
-import { ATTRIBUTION_PROMPT_VERSION } from "./attribution";
-import {
- DEFAULT_COOKIE_MODE,
- isCookieMode,
- type CookieMode,
-} from "./cookiePolicy";
-// From the CLIENT-SAFE digest module, deliberately — digestApps.ts imports execa,
-// and settings.ts must stay reachable from anywhere.
-import {
- CLAUDE_DIGEST_APP_ID,
- DEFAULT_DIGEST_APP_ID,
- DEFAULT_DIGEST_TIMESTAMP_MODE,
- DIGEST_SECTION_KINDS,
- DIGEST_TIMESTAMP_MODES,
- isDigestSectionKind,
- isDigestTimestampMode,
- type DigestAppConfig,
- type DigestSectionKind,
- type DigestTimestampMode,
-} from "./digest";
-
-export type { Worker } from "./workers";
-export type { AutoQueueSettings } from "./autoQueueTypes";
-export type { ChannelPriority } from "./channelPriority";
-
-// Transcribe placeholder/arg helpers now live with the whisper-cpp app in
-// transcriptionApps.ts. Re-exported here so existing import sites keep working.
-export {
- type AppInstanceConfig,
- TRANSCRIBE_PLACEHOLDER_AUDIO,
- TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE,
- TRANSCRIBE_PLACEHOLDER_MODEL,
- TRANSCRIBE_KNOWN_PLACEHOLDERS,
- DEFAULT_TRANSCRIBE_ARGS,
- validateTranscribeArgs,
-} from "./transcriptionApps";
-
-// Global, OPERATIONAL settings shared across every site this editor powers.
-// Per-site presentation (branding, social links, channel groups, membership)
-// lives in sites/<siteId>/site.json — see common/lib/site.ts.
-export type SiteSettings = {
- // Title for the EDITOR admin shell only (the editor manages all sites and so
- // is not tied to any one site's branding). Public sites get their own titles
- // from site.json.
- adminTitle: string;
- maxTranscriptPageBytes: number;
- // Active transcription app id (key into TRANSCRIPTION_APPS, e.g. "whisper-cpp"
- // or "chough"). Selected globally; see common/lib/transcriptionApps.ts.
- transcriptionApp: string;
- // Per-app configuration, keyed by app id. Each app reads only its own block;
- // a missing block means "use the app's defaults". DEPRECATED in favor of
- // `workers` (each local worker carries its own config); kept one release to
- // drive migration and allow rollback. See common/lib/workers.ts.
- transcriptionApps: Record<string, AppInstanceConfig>;
- // Configured transcription workers (named processing slots). The scheduler
- // distributes each video to the highest-priority free worker. A settings.json
- // predating this field is migrated to a single enabled worker from the active
- // app (see defaultWorkersFromApps). See common/lib/workers.ts.
- workers: Worker[];
- // Browser spec (e.g. "firefox", "chrome:Default") passed to
- // `yt-dlp --cookies-from-browser`. WHEN it is passed is governed by
- // `cookieMode` below. Empty string = no cookies configured. Per-channel
- // override available (ChannelConfig.cookiesFromBrowser).
- cookiesFromBrowser: string;
- // How yt-dlp invocations use the configured cookies (see
- // common/lib/cookiePolicy.ts): "always" passes them on every invocation,
- // "when-required" (default; the historical behavior) only to retry an
- // auth/age failure, "defer" never in normal runs — auth-gated videos are
- // excluded from batches and collected into the per-channel "Needs cookies"
- // bucket for a manual cookie run. Per-channel override available
- // (ChannelConfig.cookieMode).
- cookieMode: CookieMode;
- // Pause (seconds) inserted between per-video yt-dlp invocations in
- // managed batch downloads. yt-dlp's own `-t sleep` only paces requests
- // within one invocation, so without this the managed loop hammers the
- // source IP back-to-back. 0 disables. Per-channel override available.
- sleepBetweenDownloadsSeconds: number;
- // Default yt-dlp `-f` download format for every channel that doesn't set its
- // own (ChannelConfig.downloadFormat). "auto" picks per-source: `original` for
- // Odysee (whose HLS rungs are CDN-truncated), `bestaudio/worst` elsewhere. See
- // common/ytdlp/downloadFormat.ts.
- downloadFormat: DownloadFormatPreset;
- // Minimum free disk space (GB) required on the transcripts data directory for
- // downloads to run. When free space is below this floor, a download job is
- // prevented from starting and a running batch stops launching new videos
- // (the in-flight one finishes). 0 disables the gate. See common/lib/diskSpace.ts.
- minFreeDiskGB: number;
- // Extra headroom (GB) above minFreeDiskGB that a stopped pipeline must see
- // before it resumes. Resuming at the same number we stopped at flaps — the
- // first restarted download pushes free space back under the floor. This is
- // the hysteresis margin, so "resumed" means the operator actually freed
- // something rather than a scratch file being cleaned up. 0 disables the
- // hysteresis (resume at the floor). See diskGate() in common/lib/diskSpace.ts.
- resumeMarginGB: number;
- // Default number of videos transcribed in parallel when a "Transcribe
- // missing" / bucket run doesn't specify its own concurrency. The per-run
- // Concurrency input in the channel UI overrides this for a single run.
- parallelTranscriptions: number;
- // When true, the no-subs fallback in the managed downloader runs whisper
- // inline immediately after the audio download succeeds. When false
- // (default), audio is left for the next "Transcribe missing" pass so a
- // batch download finishes faster and whisper can parallelize.
- inlineTranscribeOnFallback: boolean;
- // When true (default), managed downloads skip videos that are currently live
- // or scheduled/upcoming, decided from a metadata-only prefetch pass. Finished
- // livestream VODs (was_live) are NOT skipped and download normally. A skip is
- // recorded but not archived, so the next sync/download-missing retries the
- // video once the stream ends. Per-channel override available
- // (ChannelConfig.skipLiveDownloads).
- skipLiveDownloads: boolean;
- // Whether the transcribed-audio cleanup sweep checks each candidate is still
- // available upstream before deleting its audio, pinning (do-not-clean) any
- // video found permanently gone. The delete is irreversible and a gone video's
- // audio is irreplaceable, so this defaults to true. Turn it off for an offline
- // or URL-less setup, where the check can never resolve and cleanup would
- // otherwise never delete anything. See verifyBeforeClean.ts.
- verifyAvailabilityBeforeClean: boolean;
- // Whether site builds generate downloadable transcript/live-chat archive zips
- // (into public/archives, linked on the Downloads page). Global default; a site
- // can opt out via site.json `archives: false`, and a single build can skip via
- // the "Skip archive zips" build control. Opt-out: default true.
- buildArchives: boolean;
- // Overflow object storage (Cloudflare R2) for archive zips that exceed the
- // Pages per-file size cap (see Site.archiveMaxBytes). When both fields are set,
- // an oversize archive is uploaded here on deploy — via `wrangler r2 object put`,
- // keyed `<siteId>/archives/<file>` — instead of being dropped, and the Downloads
- // page links to `<publicBaseUrl>/<key>`. Blank/absent → no overflow, so oversize
- // archives stay unavailable ("Too large to host").
- archiveStorage?: { bucket: string; publicBaseUrl: string };
- // Debounce preset for the global snapshot scheduler: how long it waits after
- // the last report-changing action before regenerating affected channel
- // reports. See REPORT_DEBOUNCE_PRESETS. Default "fast" (~1s, no cap).
- reportDebouncePreset: ReportDebouncePreset;
- // How often (seconds) the editor UI passively re-fetches the current page's
- // server-rendered data via router.refresh(), so sidebar badges and reports
- // stay live without a manual reload. Mounted globally; pauses while the tab is
- // hidden. 0 disables passive refresh entirely. See AUTO_REFRESH_INTERVAL_*.
- autoRefreshIntervalSeconds: number;
- // Global configuration for the scheduled (cron-driven) channel sync system.
- // The per-channel cadence lives on ChannelConfig.syncIntervalMinutes; this
- // block holds the defaults and guard rails the scheduler applies across all
- // channels. See common/jobs/syncScheduler.ts.
- syncScheduler: SyncSchedulerSettings;
- // Configuration for the automatic priority-queue runners (auto-transcribe /
- // auto-download). Each holds a tree policy that decides which channel's video
- // to process next, cross-channel, by priority/round-robin/weighted-fair rules.
- // Independent of syncScheduler (which decides staleness, not work order). See
- // common/jobs/autoQueuePolicy.ts.
- autoQueue: AutoQueueSettings;
- // THE OPERATOR-FACING PRIORITY MODEL: one tier per channel plus one
- // corpus-wide focus selector. It is the SOURCE the four `autoQueue[lane].root`
- // trees are compiled from (common/lib/channelPriority.ts), not a second
- // mechanism beside them — and its `paused` tier is the one part that is not a
- // tree shape, filtering the runner's channel list instead. An empty document
- // (the default) is today's behaviour exactly: no focus, every channel normal,
- // the stored trees stand.
- channelPriority: ChannelPriority;
- // Default social links applied to every site that doesn't define its own.
- // A site inherits these unless its site.json carries an explicit
- // `socialLinks` array — see Site.socialLinks / resolveSocialLinks in
- // common/lib/site.ts. The one presentation field that lives globally so a
- // shared footer doesn't have to be repeated per site.
- socialLinks: SocialLink[];
- // Absolute public URL of the family hub/homepage (e.g. "https://archilyzer.pages.dev").
- // Every export site links back to it ("the family" backlink) when set. Empty =
- // no hub link rendered. Normalized to a trailing-slash-free http(s) URL.
- homepageUrl: string;
- // Backup configuration for the saved-video store (Phase 4 of the
- // video-persistence feature). When enabled with a destination, the store is
- // mirrored there (additively, no deletes) with a per-backup manifest, and the
- // sync scheduler runs the backup on the configured cadence. See
- // common/controller/backupSavedVideos.ts.
- savedVideoBackup: SavedVideoBackupSettings;
- // Where a channel's downloaded media goes when it is relocated off the corpus
- // disk. A DEFAULT ONLY: the relocate controller never reads it and always
- // takes an explicit root, so this is the value the per-channel Storage panel
- // prefills and the /channels bulk move falls back to. Blank = no default.
- // See StorageSettings.
- storage: StorageSettings;
- // How the static export is built: "basic" reuses the single export/ tree and
- // serializes builds on one queue (the long-standing behavior); "docker" runs
- // each site's build in an isolated container for safe parallelism. The Docker
- // pipeline itself is a follow-up; this block persists the chosen mode plus the
- // container/concurrency knobs the deploy page and the future orchestrator read.
- buildPipeline: BuildPipelineSettings;
- // AI digest generation (chapters + topic tags over the existing transcripts).
- // Local-first: the metered lane is off by default. See DigestSettings.
- digest: DigestSettings;
- // Speaker diarization captured right after transcription, while the audio is
- // still on disk. OFF by default. See DiarizationSettings.
- diarization: DiarizationSettings;
- // The generic catch-up lane for derived data the existing corpus predates.
- // OFF by default, and idle-only when on. See BackfillSettings.
- backfill: BackfillSettings;
- // Naming the speakers diarization found (or reconstructing them from the
- // transcript when it found none). OFF by default. See AttributionSettings.
- attribution: AttributionSettings;
-};
-
-// Configuration for speaker attribution — putting names to the speaker turns.
-//
-// OFF by default, and that default is doing real work rather than being
-// cautious. The text-only lane costs roughly one model call per transcript
-// CHUNK, which on this corpus is ~194,000 calls, the same order as the digest
-// sweep — and the digest sweep has completed 0.17% of its own. Arming both at
-// once puts two multi-week lanes on the same 8 GB card with nothing to arbitrate
-// between them (the backfill lane's yield deliberately watches only the
-// transcription lane). Nothing here arms anything; a pilot decides whether the
-// corpus-wide text-only pass is worth 25-55 GPU-days at all.
-export type AttributionSettings = {
- // Master switch. Off means the backfill registry reports no attribution work
- // at all — the feature gate every Operation has.
- enabled: boolean;
- // Which digest app runs the naming. Attribution IS a digest-app workload —
- // constrained JSON decoding over transcript text — so it reuses that registry
- // and that per-app config (settings.digest.apps[appId]) rather than growing a
- // second copy of the ollama URL, context size and timeout.
- appId: string;
- // Model override. Empty = the app's configured model, then its default. It is
- // separate from the digest's because the two workloads may want different
- // sizes, and because it is part of the freshness identity: sharing the digest's
- // model field would make a digest bake-off invalidate every attribution record
- // on disk as a side effect.
- model: string;
- // The lanes, separately. Both default OFF even when `enabled` is on, so
- // turning the feature on to look at it cannot start a corpus sweep.
- //
- // They are not a fallback pair. `diarized` is one call per video and grounded
- // in acoustic clustering; `textOnly` is ~30 calls and guesses at identity
- // across chunk seams. An operator may reasonably want the first forever and
- // the second never.
- diarizedEnabled: boolean;
- textOnlyEnabled: boolean;
- // The prompt generation a record must match to count as fresh.
- //
- // Defaults to (and is floored at) ATTRIBUTION_PROMPT_VERSION, the shipped
- // constant. Raising it forces a corpus-wide regeneration without a code
- // change, which is the honest way to redo everything after a prompt tweak.
- // It cannot be set BELOW the shipped constant, and that floor is the lesson
- // from digestPrompt.ts's version 1 -> 2 note: pinning freshness to an older
- // generation freezes output from a superseded prompt into the corpus, looking
- // identical to output from the current one.
- promptVersion: number;
-};
-
-// Configuration for the backfill lane — the generic answer to "a derived-data
-// feature landed and 77,000 existing videos do not have it".
-//
-// WHAT THE LANE GETS is its own queueKey (concurrency with transcription) plus a
-// limit() that returns 0 to stand aside — the same mechanism the digest yield
-// uses, which fails OPEN so a bad read costs contention rather than a deadlock.
-// There is no priority system to join: the registry submits every named queue at
-// concurrency 1 and SchedulerTier only orders work within a single key.
-//
-// THE SHARE IS `concurrency` AND THE LANE'S `autoQueue.backfill.maxWorkers`; the
-// yield is the operation's declared `contendsFor`. Slice 1.3 retired the
-// `weight` scalar that used to mean both — see backfillLimit().
-export type BackfillSettings = {
- // Slots the lane may use when it is not standing aside. Kept at 1 by default
- // for the same reason diarization.concurrency is: this is CPU-bound work
- // competing with GPU feeding and the digest sweep for the same 8 threads.
- concurrency: number;
- // Re-acquire media for videos whose input is GONE (audio deleted after
- // transcription). OFF by default and deliberately so: measured on this corpus,
- // 836 videos still have media and ~76,270 would need a re-download — 91x the
- // reachable work, against 45 GB free at 97% full. When on, each re-fetched
- // file is removed in a `finally` as soon as the backfill has used it, unless
- // the video is marked do-not-clean, or unless the auto-transcribe policy would
- // replace its auto-captions (`replaceAutoSubs`, or a leaf on
- // `downloadedAutoSubsOnly`), in which case the audio is kept for that runner.
- //
- // WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: "youtube"`
- // channel — which normally only fetches subtitles — the re-acquire applies a
- // PER-VIDEO transcribe override so yt-dlp lands audio a diarizer can read;
- // the channel's stored config is not changed. Without that override the fetch
- // re-downloads the captions the video already has and lands nothing, which is
- // what happened to ~16,000 videos on eight channels in 2026-08.
- allowRedownload: boolean;
-};
-
-// Configuration for the speaker-diarization capture lane.
-//
-// This exists because AUDIO IS THE ONE PERISHABLE INPUT in the pipeline:
-// cleanAudioFromTranscribed deletes it once a video is transcribed, so
-// diarization has to happen while the audio is still there or not at all. The
-// capture half is deliberately all that ships here — attribution, LLM speaker
-// naming, viewer badges and quote filtering can all be redone later from the
-// saved JSON, whereas the audio cannot.
-export type DiarizationSettings = {
- // Master switch. OFF by default so a transcription batch can start before this
- // lands, with diarization backfilled over the retained audio afterwards.
- //
- // Turning it ON also arms the cleanup guard: the Clean-audio sweep stops
- // deleting audio for a transcribed video that has no diarization.json yet.
- // That is the point — it is what keeps the perishable input alive long enough
- // to be captured — but it means enabling this holds disk.
- enabled: boolean;
- // Run diarization inline in the post-transcribe hook.
- //
- // OFF by default, and that default is a MEASURED decision, not caution.
- // Measured on this box: GPU transcription runs at 221 s/audio-hour (16.3x
- // realtime, over 3,602 real videos), CPU diarization at ~500-680 s/audio-hour.
- // Diarization is therefore ~2-3x SLOWER than the transcription it follows, so
- // running it inline drops whole-pipeline throughput by roughly 3-4x and leaves
- // the GPU idle while the CPU catches up.
- //
- // The intended sequence for a large batch is the opposite: leave this off, let
- // the batch transcribe at full GPU speed with `enabled` holding the audio, and
- // diarize afterwards with the backfill pass. Turn it on for steady state, once
- // the arrival rate is a few videos a day rather than a corpus.
- inlineAfterTranscribe: boolean;
- // Clustering threshold — the single most consequential knob, since it decides
- // how many speakers come out. Larger merges more aggressively.
- //
- // The default is 0.9, NOT sherpa-onnx's own 0.5, and that is measured on this
- // corpus. On a 6-minute excerpt of a two-person interview (known ground truth:
- // 2 speakers), sherpa's default produced 22 clusters; 0.9 produced 6, with the
- // top two at 40%/40% of talk time — recognizably the two hosts. Sweep on the
- // same clip: 0.4→23, 0.5→22, 0.6→17, 0.7→12, 0.8→10, 0.9→6.
- //
- // It still over-splits, which is why this is a capture lane and not an answer:
- // the turns are recorded with the threshold that produced them, so a later
- // attribution pass can re-cluster or re-run without needing the audio back.
- threshold: number;
- // Engine threads per diarize run.
- threads: number;
- // Which engine runs. "sherpa-onnx" is the shipped default and what every
- // sidecar on disk was produced by; "sortformer" is the ggml engine built by
- // scripts/build-sortformer.sh.
- //
- // CHANGING THIS RESTATES THE FRESHNESS IDENTITY (see diarizationTarget), so
- // every sidecar written by the other engine becomes stale and the backfill lane
- // offers to redo it. That is intended — the two disagree about how many
- // speakers exist, and a corpus half-diarized by each is not one corpus — but on
- // the retained audio it is weeks of work, not a toggle.
- //
- // Why anyone would: on the same file, sherpa at its tuned threshold returns 13
- // speakers and sortformer returns 4, agreeing on the dominant speaker's share
- // to within half a point (73.1% vs 73.5%). On the corpus's worst case sherpa
- // returns 35 and sortformer 4. Over-splitting is the failure mode this lane has
- // always had, and sortformer is end-to-end rather than clustered, so it does
- // not have it. The cost is a hard ceiling of 4 speakers and ~1.8x the wall
- // clock.
- engine: DiarizationEngineId;
- // Compute device for the sortformer engine; ignored by sherpa-onnx, which has
- // no Vulkan compute path on Linux.
- //
- // "vulkan" is 1.5x faster than a thread-tuned CPU run (894 vs 1305
- // s/audio-hour, measured on this box) and holds 558 MB resident instead of
- // 4.84 GB by keeping weights and activations in VRAM. It also takes ~4.4 GB of
- // an 8 GB card, which is why the lane YIELDS to transcription rather than
- // sharing — see controller/digestYield.ts.
- backend: DiarizationBackend;
- // Python interpreter for the default sherpa-onnx engine. sherpa-onnx ships
- // wheels only up to cp313, and this box's system python is 3.14 — so this
- // usually points at a dedicated venv rather than `python3`.
- python: string;
- // ONNX model paths for the default engine. Empty = the lane cannot run, which
- // is reported as a skip rather than a failure.
- segModel: string;
- embModel: string;
- // Binary and model for the sortformer engine, both produced by
- // scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the
- // same "not-configured" skip as an unset segModel/embModel.
- sortformerBin: string;
- sortformerModel: string;
- // How many diarize runs may execute at once in the backfill pass. Kept low by
- // default: diarization is CPU-bound and competes with GPU feeding and the
- // digest sweep for the same 8 threads.
- concurrency: number;
- // Videos longer than this are DEFERRED rather than diarized: reported as a
- // third number that is never summed into reachable work, so a capped corpus
- // can never read as finished.
- //
- // THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering holds a
- // pairwise distance matrix over speech-segment embeddings — O(n^2) in SEGMENT
- // count — and speaker-turn density varies 40x across this corpus (33-1364
- // turns/hour), so duration does not actually predict the blowup: a sparse
- // 7h42m video completed while a dense 6h12m one was OOM-killed. Duration is
- // merely the only predictor available for free, from metadata already on disk,
- // BEFORE spending 45 minutes to find out. n^2 at 30k segments is 6.7 GiB and
- // at 40k is 11.9 GiB, which brackets the 10.6 GB and 9.6 GB peaks measured on
- // this 16 GB box.
- //
- // 0 disables the cap. That is where this goes once windowed diarization lands:
- // windowing divides per-window n by the window count, so the matrix falls by
- // its square, and the cap stops being needed rather than being tuned.
- maxAudioHours: number;
-};
-
-// Configuration for the derived-corpus digest layer. Local-first by decision:
-// `remoteEnabled` gates the metered lane and defaults to false, so nothing here
-// can spend money until it is explicitly turned on.
-export type DigestSettings = {
- // Master switch for the metered (remote-api) lane. OFF by default — an opt-in
- // overflow for the long tail or a channel where local quality is poor, never
- // the default path.
- remoteEnabled: boolean;
- // Videos longer than this are "long tail": 8.2% of the corpus by count, 46% of
- // all transcript tokens. The batch's duration-aware ordering and the optional
- // remote overflow both key off it.
- longTailSeconds: number;
- // The engine each lane uses (ids from common/lib/digestApps.ts).
- localAppId: string;
- remoteAppId: string;
- // Per-app config, keyed by app id — the same id-keyed sub-record shape as
- // transcriptionApps.
- apps: Record<string, DigestAppConfig>;
- // Yield the GPU to the transcription lane: while transcription is working, the
- // digest batch's limit() returns 0 and the pool idle-waits. ON by default,
- // because `digest:local` is deliberately on a different queue from
- // TRANSCRIPTION_QUEUE and so would otherwise run ollama and the transcription
- // engine on the same 8 GB card. See controller/digestYield.ts.
- yieldToTranscription: boolean;
- // Whether a busy worker pinned to `device: "cpu"` counts as GPU contention.
- //
- // OFF by default, which is the FIX for a real bug: the yield originally tested
- // only `kind === "local"`, so on a box with one GPU worker and two CPU-pinned
- // ones (this box, at parallelTranscriptions 2) the digest lane stopped dead for
- // transcription that competes for zero GPU shaders.
- //
- // Only an EXPLICIT "cpu" is treated as non-contending. A worker with no device
- // set is using the engine binary's own default, which may be the GPU, so it
- // still triggers the yield — the unknown case fails safe.
- //
- // Composes with `yieldToTranscription`: that is the master switch, this only
- // narrows which workers it reacts to.
- yieldToCpuWorkers: boolean;
- // Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. Only ever
- // consulted for a metered app.
- spendCapUsd: number;
- // Which sections a sweep generates.
- //
- // Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, and that
- // is not a contradiction: a tag call sends the same transcript as the chapter
- // call before it, so it hits the engine's cached prefix and pays essentially no
- // prefill (+0.4 s across 4 extra calls, against 22.4 s for the first 4). All it
- // pays is decode, and a tag list is ~30 output tokens where a chapter list is
- // ~200–290.
- //
- // The corollary matters more than the number: run them in the SAME pass. Tags
- // generated later, on their own, pay full prefill again — measured at 44% of a
- // whole chapters pass, i.e. 3–9× the marginal cost of just including them now.
- sections: DigestSectionKind[];
- // How each chunk's transcript markers are numbered — see DigestTimestampMode.
- // Was a scored variable in the bake-off rather than a pre-applied fix; the
- // measurement is in and "chunk-local" is now the shipped default.
- timestampMode: DigestTimestampMode;
- // A free-text label for a non-default prompt shape, folded into the recorded
- // provenance by digestPromptVariant(). Setting it invalidates every digest
- // generated under a different label, which is exactly what makes a bake-off
- // round re-run its sample instead of skipping it as fresh. Empty = default.
- promptVariant: string;
-};
-
-// "basic" — `pnpm run build` in export/, serialized on the build queue (shared
-// output tree → no safe parallelism).
-// "docker" — isolated per-site container builds (follow-up); enables real
-// parallel multi-site builds capped by maxParallelBuilds.
-export type BuildMode = "basic" | "docker";
-
-export type BuildPipelineSettings = {
- mode: BuildMode;
- // Cap on concurrent per-site container builds in docker mode. Ignored in basic
- // mode (which is always serial). Clamped to [1, BUILD_MAX_PARALLEL_MAX].
- maxParallelBuilds: number;
- // Tag of the reusable build image (built once, reused for every site).
- dockerImage: string;
- // Dockerfile path relative to the monorepo root, used to (re)build the image.
- dockerfile: string;
-};
-
-export type SavedVideoBackupSettings = {
- // Master switch for the scheduled backup. A backup can still be run manually
- // when this is false, as long as a destination is set.
- enabled: boolean;
- // Destination root the store is mirrored into (a local path or any rsync
- // target). Empty disables both scheduled and manual backups.
- dest: string;
- // Cadence (minutes) for the scheduled backup when enabled. Clamped into the
- // sync-interval window; default daily.
- intervalMinutes: number;
-};
-
-export type SyncSchedulerSettings = {
- // Master switch. When false, a tick selects nothing (manual sync still works).
- enabled: boolean;
- // Fallback cadence (minutes) for channels with no per-channel override.
- defaultIntervalMinutes: number;
- // Cap on sync jobs running/queued at once. A tick queues at most
- // (cap - currently-active) channels; the rest roll to the next tick. This is
- // also the stagger mechanism that keeps a big due-batch from hitting the
- // source all at once.
- maxConcurrentSyncs: number;
- // Optional local-clock quiet window during which auto-sync is suppressed.
- // Both null = always allowed. The window may wrap past midnight
- // (e.g. start=22, end=6). Hours are [0,23]; the window is [start, end).
- quietHoursStart: number | null;
- quietHoursEnd: number | null;
- // Failure backoff bounds. After N consecutive failed scheduled syncs a
- // channel waits min(base * 2^(N-1), max) minutes before it's eligible again.
- backoffBaseMinutes: number;
- backoffMaxMinutes: number;
- // Cadence (seconds) for the editor's in-process heartbeat — the internal timer
- // armed by the instrumentation hook (editor/instrumentation.ts) that calls the
- // scheduler tick directly, so no external cron is needed. 0 = off: rely on the
- // external `pnpm sync:tick` heartbeat instead. Any positive value is clamped to
- // [SYNC_HEARTBEAT_MIN_SECONDS, SYNC_HEARTBEAT_MAX_SECONDS]. The env var
- // SYNC_HEARTBEAT_SECONDS overrides this at runtime. See SCHEDULED_SYNC.md.
- heartbeatSeconds: number;
- // Cadence (minutes) for the scheduled keep-latest deletion check. For each
- // channel with ChannelConfig.keepLatest > 0, the tick re-probes the kept
- // window for source deletion (checkKeptDeletedAction) at most this often and
- // pins any gone videos as do-not-clean. Clamped into the sync-interval window;
- // default daily. The check shares the same concurrency cap and quiet-hours
- // window as scheduled syncs. See editor/app/scheduler/runTick.ts.
- keepLatestCheckIntervalMinutes: number;
- // Default cadence (minutes) for the sync FULL SWEEP — the deep pass that
- // re-enumerates a channel's whole listing in one yt-dlp spawn, refreshes the
- // stored `playlist` file, and flags videos that have left the listing into
- // maybe-missing.json. Ordinary syncs stay on the cheap newest-first paged
- // walk; a sync only upgrades itself to a sweep when this interval has elapsed
- // since the channel's lastFullSweepAt. Per-channel override:
- // ChannelConfig.fullSweepIntervalMinutes. 0 = never sweep. Default daily.
- // See common/jobs/deepSync.ts.
- fullSweepIntervalMinutes: number;
- // Upper bound on how many maybe-missing suspects a full sweep will resolve
- // in-line with the per-video availability probe (deleted vs private vs
- // unlisted). At or under the cap the sweep runs the targeted check itself, so
- // "Sync all" surfaces upstream deletions with no extra clicks; over it, the
- // suspects are flagged and left for a manual check rather than firing hundreds
- // of probes inside a sync. 0 = never auto-confirm.
- fullSweepConfirmMaxSuspects: number;
- // Shrink guard: how far a fresh listing may fall below the stored one before
- // it is treated as suspect rather than acted on. Expressed as a percentage of
- // the previous count, floored at SHRINK_ABS_FLOOR entries so ordinary churn on
- // a small channel doesn't trip it. A suspect listing does not rewrite
- // `playlist` or maybe-missing.json and does not count as a sweep — but a
- // SECOND enumeration reporting a similar count confirms it and is accepted, so
- // a genuine mass deletion costs at most one cadence period. 0 = off (the
- // empty-listing rejection still applies). See controller/acceptListing.ts.
- fullSweepShrinkGuardPercent: number;
-};
-
-export type SocialLink = {
- label: string;
- url: string;
- svg: string;
-};
-
-export const SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS = 600;
-export const SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS = 10;
-
-export const MIN_FREE_DISK_GB_DEFAULT = 5;
-export const MIN_FREE_DISK_GB_MAX = 100000;
-
-// Hysteresis margin for the low-disk gate. 2 GB is deliberately larger than any
-// single scratch file the pipeline writes, so cleaning one up cannot by itself
-// reopen the gate.
-export const RESUME_MARGIN_GB_DEFAULT = 2;
-export const RESUME_MARGIN_GB_MAX = 1000;
-
-export const PARALLEL_TRANSCRIPTIONS_MAX = 16;
-export const PARALLEL_TRANSCRIPTIONS_DEFAULT = 2;
-
-// Passive UI auto-refresh cadence. 0 is a sentinel meaning "disabled"; any other
-// value is clamped into [MIN, MAX] seconds.
-export const AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS = 5;
-export const AUTO_REFRESH_INTERVAL_MIN_SECONDS = 1;
-export const AUTO_REFRESH_INTERVAL_MAX_SECONDS = 600;
-
-// Global snapshot-scheduler debounce presets. `debounceMs` is the quiet-period
-// window after the last report-changing action; `maxWaitMs` caps the total
-// delay under continuous activity (null = no cap, fire purely on the quiet
-// period). Consumed by common/jobs/snapshotScheduler.ts and surfaced in the
-// Settings form.
-export type ReportDebouncePreset = "fast" | "balanced" | "lazy";
-
-export const REPORT_DEBOUNCE_PRESETS: Record<
- ReportDebouncePreset,
- { debounceMs: number; maxWaitMs: number | null }
-> = {
- fast: { debounceMs: 1000, maxWaitMs: null },
- balanced: { debounceMs: 3000, maxWaitMs: 30000 },
- lazy: { debounceMs: 10000, maxWaitMs: 60000 },
-};
-
-export const DEFAULT_REPORT_DEBOUNCE_PRESET: ReportDebouncePreset = "fast";
-
-export function isReportDebouncePreset(v: unknown): v is ReportDebouncePreset {
- return v === "fast" || v === "balanced" || v === "lazy";
-}
-
-export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024;
-export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024;
-export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024;
-
-export const DEFAULT_ADMIN_TITLE = "Transcript Browser Admin";
-
-// Sync-scheduler bounds + defaults. Default cadence is daily; concurrency is
-// conservative so a tick doesn't fan out into the source provider all at once.
-export const SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES = 1440;
-export const SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT = 2;
-export const SYNC_SCHEDULER_MAX_CONCURRENT_MAX = 16;
-export const SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES = 30;
-export const SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES = 1440;
-export const KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES = 1440;
-// Full-sweep defaults. Daily: a sweep is one full enumeration of the channel,
-// far more expensive than the 50-entry page an ordinary sync fetches. The
-// confirm cap keeps an unattended sweep from fanning out into hundreds of
-// per-video probes when a channel's listing changes wholesale.
-export const FULL_SWEEP_DEFAULT_INTERVAL_MINUTES = 1440;
-export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT = 25;
-export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX = 10000;
-// Shrink-guard default: a listing that has lost more than a tenth of its
-// entries (and more than SHRINK_ABS_FLOOR of them) needs a second opinion.
-export const FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT = 10;
-export const FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX = 100;
-export const SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES = 1440;
-
-// Internal-heartbeat cadence bounds. 0 means "off" (use an external cron
-// heartbeat); any other value is clamped into [MIN, MAX] seconds. The floor
-// keeps the in-process timer from busy-looping; the ceiling is one hour.
-export const SYNC_HEARTBEAT_DEFAULT_SECONDS = 0;
-export const SYNC_HEARTBEAT_MIN_SECONDS = 15;
-export const SYNC_HEARTBEAT_MAX_SECONDS = 3600;
-
-export function defaultSyncScheduler(): SyncSchedulerSettings {
- return {
- enabled: false,
- defaultIntervalMinutes: SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES,
- maxConcurrentSyncs: SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT,
- quietHoursStart: null,
- quietHoursEnd: null,
- backoffBaseMinutes: SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES,
- backoffMaxMinutes: SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES,
- heartbeatSeconds: SYNC_HEARTBEAT_DEFAULT_SECONDS,
- keepLatestCheckIntervalMinutes: KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES,
- fullSweepIntervalMinutes: FULL_SWEEP_DEFAULT_INTERVAL_MINUTES,
- fullSweepConfirmMaxSuspects: FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT,
- fullSweepShrinkGuardPercent: FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT,
- };
-}
-
-// Clamp an internal-heartbeat cadence: 0 (off) passes through; any positive
-// value is clamped up into [MIN, MAX]; junk falls back to the default.
-export function clampHeartbeatSeconds(value: unknown): number {
- if (typeof value !== "number" || !Number.isFinite(value)) {
- return SYNC_HEARTBEAT_DEFAULT_SECONDS;
+ clampParallelTranscriptions,
+ defaultStorage,
+ normalizeSocialSvg,
+ parseSocialLinks,
+ sanitizeTranscriptionApps,
+ siteSettingsSchema,
+ type SiteSettings,
+ type SocialLink,
+} from "./settingsSchema";
+
+export * from "./settingsSchema";
+
+type RawSettings = Record<string, unknown>;
+
+// The file as JSON, or `undefined` when it is missing or not JSON. Never
+// throws: a settings read is on every request path.
+function readRawSettings(file: string): unknown {
+ try {
+ return JSON.parse(fs.readFileSync(file, "utf8"));
+ } catch {
+ return undefined;
}
- const n = Math.floor(value);
- if (n <= 0) return 0;
- if (n < SYNC_HEARTBEAT_MIN_SECONDS) return SYNC_HEARTBEAT_MIN_SECONDS;
- if (n > SYNC_HEARTBEAT_MAX_SECONDS) return SYNC_HEARTBEAT_MAX_SECONDS;
- return n;
-}
-
-function clampHourOrNull(value: unknown): number | null {
- if (typeof value !== "number" || !Number.isFinite(value)) return null;
- const n = Math.floor(value);
- if (n < 0 || n > 23) return null;
- return n;
-}
-
-// Like clampPositiveInt, but 0 survives as a sentinel ("off"/"never"). Used by
-// the cadences whose disabled state is expressed as a zero rather than a
-// separate boolean.
-function clampIntAllowZero(value: unknown, fallback: number, max: number): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? Math.floor(value)
- : fallback;
- if (n <= 0) return 0;
- if (n > max) return max;
- return n;
}
-function clampPositiveInt(value: unknown, fallback: number, max: number): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? Math.floor(value)
- : fallback;
- if (n < 1) return 1;
- if (n > max) return max;
- return n;
+// Only a plain object is a settings file. `null`, `[]`, `3` and a truncated
+// write all read as the empty file — every field its default — rather than as
+// an exception. (Before slice 4a a file containing `null` threw here.)
+function rawObject(raw: unknown): RawSettings {
+ return raw && typeof raw === "object" && !Array.isArray(raw)
+ ? (raw as RawSettings)
+ : {};
}
-// Coerce a raw settings.syncScheduler value into a clean SyncSchedulerSettings,
-// falling back to defaults for missing/ill-typed fields. Quiet hours are only
-// honored when BOTH endpoints are valid hours; otherwise the window is cleared.
-export function sanitizeSyncScheduler(value: unknown): SyncSchedulerSettings {
- const d = defaultSyncScheduler();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const start = clampHourOrNull(r.quietHoursStart);
- const end = clampHourOrNull(r.quietHoursEnd);
- const backoffBase = clampPositiveInt(
- r.backoffBaseMinutes,
- d.backoffBaseMinutes,
- SYNC_INTERVAL_MAX_MINUTES,
- );
- return {
- enabled: r.enabled === true,
- defaultIntervalMinutes: clampPositiveInt(
- r.defaultIntervalMinutes,
- d.defaultIntervalMinutes,
- SYNC_INTERVAL_MAX_MINUTES,
- ),
- maxConcurrentSyncs: clampPositiveInt(
- r.maxConcurrentSyncs,
- d.maxConcurrentSyncs,
- SYNC_SCHEDULER_MAX_CONCURRENT_MAX,
- ),
- quietHoursStart: start !== null && end !== null ? start : null,
- quietHoursEnd: start !== null && end !== null ? end : null,
- backoffBaseMinutes: backoffBase,
- // Cap can't sit below the base, or backoff would never grow.
- backoffMaxMinutes: Math.max(
- backoffBase,
- clampPositiveInt(
- r.backoffMaxMinutes,
- d.backoffMaxMinutes,
- SYNC_INTERVAL_MAX_MINUTES,
- ),
- ),
- heartbeatSeconds: clampHeartbeatSeconds(r.heartbeatSeconds),
- keepLatestCheckIntervalMinutes: clampPositiveInt(
- r.keepLatestCheckIntervalMinutes,
- d.keepLatestCheckIntervalMinutes,
- SYNC_INTERVAL_MAX_MINUTES,
- ),
- fullSweepIntervalMinutes: clampIntAllowZero(
- r.fullSweepIntervalMinutes,
- d.fullSweepIntervalMinutes,
- SYNC_INTERVAL_MAX_MINUTES,
- ),
- fullSweepConfirmMaxSuspects: clampIntAllowZero(
- r.fullSweepConfirmMaxSuspects,
- d.fullSweepConfirmMaxSuspects,
- FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX,
- ),
- fullSweepShrinkGuardPercent: clampIntAllowZero(
- r.fullSweepShrinkGuardPercent,
- d.fullSweepShrinkGuardPercent,
- FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX,
- ),
- };
-}
-
-export function defaultSavedVideoBackup(): SavedVideoBackupSettings {
- return {
- enabled: false,
- dest: "",
- intervalMinutes: SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES,
- };
-}
-
-// Coerce a raw settings.savedVideoBackup value into a clean
-// SavedVideoBackupSettings. A missing destination forces enabled off, since a
-// backup with nowhere to go is meaningless.
-export function sanitizeSavedVideoBackup(
- value: unknown,
-): SavedVideoBackupSettings {
- const d = defaultSavedVideoBackup();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const dest = typeof r.dest === "string" ? r.dest.trim() : "";
- return {
- enabled: dest !== "" && r.enabled === true,
- dest,
- intervalMinutes: clampPositiveInt(
- r.intervalMinutes,
- d.intervalMinutes,
- SYNC_INTERVAL_MAX_MINUTES,
- ),
- };
-}
-
-// Where relocated channel media goes: the named locations.
-//
-// This used to be ONE FIELD, `mediaRoot` — a single absolute string, the cold
-// drive, typed once. It grew into a list of entities because a root alone
-// cannot answer the two questions the operator actually has: is that disk here,
-// and if it came up somewhere else, how do I point the channels at it without
-// ssh and hand edits? A location carries an id, a label, the root, an opt-in
-// `autoRepoint`, and the volume identity learned at its last probe.
-//
-// Still NOT a policy: a channel on a location is not thereby deprioritized, and
-// nothing auto-relocates anything because a location exists.
+// THE TWO ABSENCE-KEYED MIGRATIONS THAT REWRITE AN INPUT BLOCK, applied to the
+// raw object BEFORE the parse. Each is handed the raw file, never a parsed one,
+// because "the file does not spell this key" is the whole trigger.
//
-// AVAILABILITY IS NEVER STORED HERE. A refresh that wrote "available" would
-// rewrite settings.json — and so bump the pulse revision — every few seconds.
-// The probe (common/lib/storageVolumes.ts) is computed per request; only the
-// `volume` identity is ever written back, and only when it changed.
+// - autoQueue: `migrateSweepsToLanes` fills `autoQueue.digest` / `.backfill`
+// from the retired sweep fields when — and only when — the file does not
+// already carry that lane. It never enables a lane the sweep flag did not.
+// See lib/laneMigration.ts.
+// - storage: `storage.mediaRoot` (one absolute string) becomes a one-entry
+// location list when the file has no `locations` key. An absent `storage`
+// block is the default block, which has a `locations` key and so does not
+// migrate. See lib/storageLocations.ts.
//
-// The types live in lib/storageLocations.ts, which is pure: a `"use client"`
-// file may import them, and must not reach storageVolumes.ts (execa).
-export type { StorageLocation, StorageVolume, StorageSettings };
-
-export function defaultStorage(): StorageSettings {
- return { locations: [], defaultLocationId: "" };
-}
-
-const LOCATION_ID_RE = /^[a-z0-9][a-z0-9-]{0,63}$/;
-
-// "internal" IS TAKEN. It is the synthetic /storage row for the corpus volume
-// (INTERNAL_LOCATION_ID), and the regex above admits it — so a hand-edited
-// settings.json, or an operator typing the obvious word into the New location
-// form, could store a real location under the one id the page assembles for
-// itself. The row would then be built twice, the rollup would count channels
-// into whichever assembled last, and `locationOfDataDir` would start matching
-// unrelocated channels against it.
-function isReservedLocationId(id: string): boolean {
- return id === INTERNAL_LOCATION_ID;
-}
-
-function sanitizeVolume(value: unknown): StorageVolume | undefined {
- if (!value || typeof value !== "object") return undefined;
- const v = value as Record<string, unknown>;
- const uuid = typeof v.uuid === "string" ? v.uuid.trim() : "";
- const mountpoint =
- typeof v.mountpoint === "string" ? v.mountpoint.trim() : "";
- // No uuid is no identity, and no mountpoint means `root === join(mountpoint,
- // relPath)` cannot hold — either way the record is not usable for finding the
- // volume again, so it is dropped rather than half-kept.
- if (!uuid || !mountpoint) return undefined;
- const relPath = typeof v.relPath === "string" ? v.relPath.trim() : "";
- const fstype = typeof v.fstype === "string" ? v.fstype.trim() : "";
- const label = typeof v.label === "string" ? v.label.trim() : "";
+// Both return RAW values; the schema's sanitizers then normalize them exactly as
+// they normalize a hand-edited file.
+function premigrateRaw(raw: RawSettings): RawSettings {
return {
- uuid,
- ...(fstype ? { fstype } : {}),
- ...(label ? { label } : {}),
- mountpoint,
- relPath,
+ ...raw,
+ autoQueue: migrateSweepsToLanes(raw),
+ storage: migrateMediaRootToLocations(raw.storage ?? defaultStorage()),
};
}
-// Coerce a raw settings.storage value into a clean StorageSettings.
-//
-// EXISTENCE IS NOT CHECKED, on purpose: the whole point of a cold location is
-// that it is a drive that may not be mounted when settings are read, and a
-// sanitizer that dropped the root on an unmounted platter would silently erase
-// the operator's choice on the next save.
+// THE TWO ABSENCE-KEYED MIGRATIONS THAT NEED THE PARSED RESULT, in this order:
//
-// ABSOLUTENESS *IS* checked, and a location with a relative root is DROPPED
-// rather than resolved. Resolving it would anchor the location to whatever cwd
-// the reader booted in — a different directory under docker, under a worktree,
-// and under `pnpm dev` — so the same settings.json would name three different
-// drives. The location form rejects a relative path with a message before it
-// ever gets here; this is the last line, not the only one.
-//
-// NESTED ROOTS ARE ALLOWED. "/mnt/platter" and "/mnt/platter/archive" may both
-// be locations; `locationOfDataDir` resolves a channel to the LONGEST matching
-// root. Nothing here rejects the nesting, because the operator who arranges a
-// disk that way means it.
-//
-// A STALE `mediaRoot` SITTING BESIDE `locations` IS IGNORED — it is not merged
-// back in as an extra location. `migrateMediaRootToLocations` reads it exactly
-// once, when `locations` is absent; after that the list is the whole truth, and
-// resurrecting a root the operator deleted would be a bug, not a kindness.
-//
-// ROLLBACK: an older binary sanitizes this block to `{ mediaRoot: "" }` — the
-// locations are dropped and the single cold root comes back blank. One string
-// lost, nothing on disk moved. `cp settings.json settings.json.pre-storage-
-// locations` before the upgrade and a downgrade is a file copy.
-export function sanitizeStorage(value: unknown): StorageSettings {
- const d = defaultStorage();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const rawList = Array.isArray(r.locations) ? r.locations : [];
- const locations: StorageLocation[] = [];
- const seen = new Set<string>();
- for (const entry of rawList) {
- if (!entry || typeof entry !== "object") continue;
- const e = entry as Record<string, unknown>;
- const id = typeof e.id === "string" ? e.id.trim() : "";
- if (!LOCATION_ID_RE.test(id) || isReservedLocationId(id) || seen.has(id)) {
- continue;
- }
- const rawRoot = typeof e.root === "string" ? e.root.trim() : "";
- if (!path.isAbsolute(rawRoot)) continue;
- // "/mnt/platter/" and "/mnt/platter" are one root; "/" stays "/".
- const stripped = rawRoot.replace(/\/+$/, "");
- const root = stripped === "" ? "/" : stripped;
- const label = typeof e.label === "string" ? e.label.trim() : "";
- const volume = sanitizeVolume(e.volume);
- seen.add(id);
- locations.push({
- id,
- label: label || id,
- root,
- autoRepoint: e.autoRepoint === true,
- ...(volume ? { volume } : {}),
- });
- }
- const wanted =
- typeof r.defaultLocationId === "string" ? r.defaultLocationId.trim() : "";
- // A default naming a location that is gone falls back to the first one, not
- // to "": with a location configured, "no default" is never the answer the
- // operator wanted, and a blank default silently disables every prefill.
- const defaultLocationId = locations.some((l) => l.id === wanted)
- ? wanted
- : (locations[0]?.id ?? "");
- // THE SAVED-VIDEO STORE'S LOCATION IS NOT FALLEN BACK, and the asymmetry
- // with `defaultLocationId` above is deliberate. That one is a PREFERENCE, so
- // picking another location when the named one is gone is helpful. This one is
- // a RECORD OF WHERE BYTES ARE: pointing it at a different location because
- // the recorded one was deleted would claim the store had moved when nothing
- // had. A dangling id sanitizes to "" — "in place" — which is what the disk
- // says as soon as anybody looks, and the symlink (if any) keeps working
- // regardless, because the store is reached through it and not through this.
- const savedWanted =
- typeof r.savedVideosLocationId === "string"
- ? r.savedVideosLocationId.trim()
- : "";
- const savedVideosLocationId = locations.some((l) => l.id === savedWanted)
- ? savedWanted
- : "";
- return {
- locations,
- defaultLocationId,
- ...(savedVideosLocationId ? { savedVideosLocationId } : {}),
- };
-}
-
-// 4 hours. Measured: videos over this are 8.2% of the corpus by count but hold
-// 46% of all transcript tokens, so they are where a sweep's wall-clock actually
-// goes and where chunk-seam bugs live.
-export const DIGEST_LONG_TAIL_DEFAULT_SECONDS = 4 * 3600;
-export const DIGEST_LONG_TAIL_MAX_SECONDS = 24 * 3600;
-
-export function defaultDigest(): DigestSettings {
- return {
- // OFF. The metered lane is built but never the default — see PLAN.md.
- remoteEnabled: false,
- longTailSeconds: DIGEST_LONG_TAIL_DEFAULT_SECONDS,
- localAppId: DEFAULT_DIGEST_APP_ID,
- remoteAppId: CLAUDE_DIGEST_APP_ID,
- // Empty on purpose: every per-app knob falls through to its own default
- // constant (resolveNumCtx -> DEFAULT_DIGEST_NUM_CTX, now 8192, and
- // maxCuesForContext sizes the chunk to it). Seeding a copy of those values
- // here would give the same number two homes and let them drift.
- apps: {},
- // ON. Real GPU contention with the transcription engine is a genuine cost
- // (re-priced: 11.2 s/chunk idle against 24.9 s/chunk on a contended box), so
- // the safe default is to step aside; turning it off is the deliberate choice.
- yieldToTranscription: true,
- // OFF. A CPU-pinned worker is not GPU contention, and treating it as such
- // stalled the digest lane for nothing. See DigestSettings.yieldToCpuWorkers.
- yieldToCpuWorkers: false,
- spendCapUsd: 0,
- sections: ["chapters"],
- timestampMode: DEFAULT_DIGEST_TIMESTAMP_MODE,
- promptVariant: "",
- };
-}
-
-// Coerce a raw settings.digest.apps value into a clean keyed map of
-// DigestAppConfig. Mirrors sanitizeTranscriptionApps — INCLUDING its
-// Array.isArray guard, without which a JSON array would pass the typeof check and
-// produce numeric-keyed garbage.
-export function sanitizeDigestApps(
- value: unknown,
-): Record<string, DigestAppConfig> {
- if (!value || typeof value !== "object" || Array.isArray(value)) return {};
- const out: Record<string, DigestAppConfig> = {};
- for (const [id, raw] of Object.entries(value as Record<string, unknown>)) {
- if (!raw || typeof raw !== "object") continue;
- const r = raw as Record<string, unknown>;
- const cfg: DigestAppConfig = {};
- if (typeof r.bin === "string" && r.bin.trim()) cfg.bin = r.bin.trim();
- if (typeof r.baseUrl === "string" && r.baseUrl.trim()) {
- cfg.baseUrl = r.baseUrl.trim();
- }
- if (typeof r.model === "string" && r.model.trim()) cfg.model = r.model.trim();
- if (typeof r.numCtx === "number" && r.numCtx > 0) {
- cfg.numCtx = Math.floor(r.numCtx);
- }
- if (typeof r.temperature === "number" && r.temperature >= 0) {
- cfg.temperature = r.temperature;
- }
- if (typeof r.timeoutMs === "number" && r.timeoutMs > 0) {
- cfg.timeoutMs = Math.floor(r.timeoutMs);
- }
- // Only carried when explicitly set — see DigestAppConfig.think.
- if (typeof r.think === "boolean") cfg.think = r.think;
- out[id] = cfg;
- }
- return out;
-}
-
-export function sanitizeDigest(value: unknown): DigestSettings {
- const d = defaultDigest();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const sections = Array.isArray(r.sections)
- ? (r.sections.filter(isDigestSectionKind) as DigestSectionKind[])
- : [];
- return {
- remoteEnabled: r.remoteEnabled === true,
- longTailSeconds: clampPositiveInt(
- r.longTailSeconds,
- d.longTailSeconds,
- DIGEST_LONG_TAIL_MAX_SECONDS,
- ),
- // Unknown app ids are not rejected here: getDigestApp() is total and falls
- // back to the local default, so a stale id degrades rather than breaking.
- localAppId:
- typeof r.localAppId === "string" && r.localAppId.trim()
- ? r.localAppId.trim()
- : d.localAppId,
- remoteAppId:
- typeof r.remoteAppId === "string" && r.remoteAppId.trim()
- ? r.remoteAppId.trim()
- : d.remoteAppId,
- apps: sanitizeDigestApps(r.apps),
- // Defaults to ON when absent — `=== false` rather than `!== true`, so a
- // settings file written before this field existed keeps the GPU-safe
- // behaviour instead of silently opting into contention.
- yieldToTranscription: r.yieldToTranscription !== false,
- // The OPPOSITE idiom, and deliberately so: `=== true`, so absence falls to
- // OFF. The field's absence means a settings file written before the CPU-worker
- // bug was found, and for those files OFF is the FIXED behaviour, not a silent
- // change of intent — nobody ever asked to stall the digest lane for a CPU
- // transcription. `yieldToTranscription` still gates the whole thing, so the
- // GPU-safe default is untouched.
- yieldToCpuWorkers: r.yieldToCpuWorkers === true,
- spendCapUsd:
- typeof r.spendCapUsd === "number" && r.spendCapUsd > 0
- ? Math.round(r.spendCapUsd * 100) / 100
- : 0,
- // An empty/garbage list would silently generate nothing, so fall back to the
- // default rather than honoring it.
- sections: sections.length > 0 ? sections : d.sections,
- timestampMode: isDigestTimestampMode(r.timestampMode)
- ? r.timestampMode
- : d.timestampMode,
- // Trimmed and length-capped: it goes into provenance on every record, and a
- // runaway value would bloat 119k sidecars.
- promptVariant:
- typeof r.promptVariant === "string"
- ? r.promptVariant.trim().slice(0, 40)
- : d.promptVariant,
- };
-}
-
-// Every known section kind, for the settings UI's checkbox list.
-export const DIGEST_SECTION_OPTIONS = DIGEST_SECTION_KINDS;
-export const DIGEST_TIMESTAMP_MODE_OPTIONS = DIGEST_TIMESTAMP_MODES;
-
-export const BUILD_MAX_PARALLEL_DEFAULT = 2;
-export const BUILD_MAX_PARALLEL_MAX = 16;
-export const DEFAULT_BUILD_IMAGE = "yt-dlp-transcript-browser-build";
-export const DEFAULT_BUILD_DOCKERFILE = "Dockerfile.build";
-
-export function isBuildMode(v: unknown): v is BuildMode {
- return v === "basic" || v === "docker";
-}
-
-export function defaultBuildPipeline(): BuildPipelineSettings {
- return {
- mode: "basic",
- maxParallelBuilds: BUILD_MAX_PARALLEL_DEFAULT,
- dockerImage: DEFAULT_BUILD_IMAGE,
- dockerfile: DEFAULT_BUILD_DOCKERFILE,
- };
-}
-
-// Coerce a raw settings.buildPipeline value into a clean BuildPipelineSettings,
-// falling back to defaults for missing/ill-typed fields.
-export function sanitizeBuildPipeline(value: unknown): BuildPipelineSettings {
- const d = defaultBuildPipeline();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const dockerImage =
- typeof r.dockerImage === "string" && r.dockerImage.trim()
- ? r.dockerImage.trim()
- : d.dockerImage;
- const dockerfile =
- typeof r.dockerfile === "string" && r.dockerfile.trim()
- ? r.dockerfile.trim()
- : d.dockerfile;
- return {
- mode: isBuildMode(r.mode) ? r.mode : d.mode,
- maxParallelBuilds: clampPositiveInt(
- r.maxParallelBuilds,
- d.maxParallelBuilds,
- BUILD_MAX_PARALLEL_MAX,
- ),
- dockerImage,
- dockerfile,
- };
-}
-
-function defaults(): SiteSettings {
- return {
- adminTitle: DEFAULT_ADMIN_TITLE,
- maxTranscriptPageBytes: TRANSCRIPT_PAGE_DEFAULT_BYTES,
- transcriptionApp: DEFAULT_TRANSCRIPTION_APP_ID,
- transcriptionApps: {},
- workers: [],
- cookiesFromBrowser: "",
- cookieMode: DEFAULT_COOKIE_MODE,
- sleepBetweenDownloadsSeconds: SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS,
- downloadFormat: "auto",
- minFreeDiskGB: MIN_FREE_DISK_GB_DEFAULT,
- resumeMarginGB: RESUME_MARGIN_GB_DEFAULT,
- parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT,
- inlineTranscribeOnFallback: false,
- skipLiveDownloads: true,
- verifyAvailabilityBeforeClean: true,
- buildArchives: true,
- archiveStorage: { bucket: "", publicBaseUrl: "" },
- reportDebouncePreset: DEFAULT_REPORT_DEBOUNCE_PRESET,
- autoRefreshIntervalSeconds: AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS,
- syncScheduler: defaultSyncScheduler(),
- autoQueue: defaultAutoQueue(),
- channelPriority: defaultChannelPriority(),
- socialLinks: [],
- homepageUrl: "",
- savedVideoBackup: defaultSavedVideoBackup(),
- storage: defaultStorage(),
- buildPipeline: defaultBuildPipeline(),
- // Must be listed here or the allowlist loop in getSettings() drops the key
- // entirely and the whole section is never read from disk.
- digest: defaultDigest(),
- diarization: defaultDiarization(),
- backfill: defaultBackfill(),
- attribution: defaultAttribution(),
- };
-}
-
-// The whole default settings object, without touching disk. Exported so a test
-// (or any caller that needs a settings-SHAPED value rather than the operator's
-// actual configuration) can build one without a settings.json.
-export function defaultSiteSettings(): SiteSettings {
- return defaults();
-}
-
-export function defaultBackfill(): BackfillSettings {
- return {
- concurrency: 1,
- // See BackfillSettings.allowRedownload — this one holds disk.
- allowRedownload: false,
- };
-}
-
-export function sanitizeBackfill(value: unknown): BackfillSettings {
- const d = defaultBackfill();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- return {
- // Clamped rather than rejected: a hand-edited 5 means "as much as possible",
- // and reading it as 0 would be the opposite of the intent.
- concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16),
- allowRedownload: r.allowRedownload === true,
- };
-}
-
-export function defaultAttribution(): AttributionSettings {
- return {
- // OFF, and both lanes OFF under it. See AttributionSettings.
- enabled: false,
- appId: DEFAULT_DIGEST_APP_ID,
- model: "",
- diarizedEnabled: false,
- textOnlyEnabled: false,
- promptVersion: ATTRIBUTION_PROMPT_VERSION,
- };
-}
-
-export function sanitizeAttribution(value: unknown): AttributionSettings {
- const d = defaultAttribution();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const str = (v: unknown, fallback: string) =>
- typeof v === "string" && v.trim() ? v.trim() : fallback;
- return {
- enabled: r.enabled === true,
- appId: str(r.appId, d.appId),
- // NOT `str(...)` with a fallback: empty is a meaningful value here ("use the
- // app's own model"), so an empty string must survive rather than reverting
- // to a default that is also empty by coincidence.
- model: typeof r.model === "string" ? r.model.trim() : d.model,
- diarizedEnabled: r.diarizedEnabled === true,
- textOnlyEnabled: r.textOnlyEnabled === true,
- // FLOORED at the shipped constant, never merely defaulted. A hand-edited
- // value below it would pin freshness to a superseded prompt generation and
- // freeze its output into the corpus — see AttributionSettings.promptVersion.
- promptVersion:
- typeof r.promptVersion === "number" && Number.isFinite(r.promptVersion)
- ? Math.max(ATTRIBUTION_PROMPT_VERSION, Math.floor(r.promptVersion))
- : d.promptVersion,
- };
-}
-
-export function defaultDiarization(): DiarizationSettings {
- return {
- // OFF. Capture is opt-in: turning it on makes the cleanup sweep start
- // refusing to delete audio for transcribed-but-undiarized videos, which is
- // correct but is a disk-pressure decision an operator should make.
- enabled: false,
- // OFF — see DiarizationSettings.inlineAfterTranscribe. Diarization is slower
- // than the transcription it would follow, so inline is the exception.
- inlineAfterTranscribe: false,
- // Measured, not sherpa-onnx's 0.5 — see DiarizationSettings.threshold. The
- // constant lives in lib/diarization.ts because isDiarizationFresh needs it
- // to normalize an absent recorded threshold; importing it keeps the default
- // and the comparator from drifting apart.
- threshold: DEFAULT_DIARIZATION_THRESHOLD,
- threads: 4,
- // The engine every sidecar on disk was produced by. Switching is an explicit
- // decision that restates the freshness identity — see DiarizationSettings.
- engine: DEFAULT_DIARIZATION_ENGINE,
- // Only consulted when engine is "sortformer". Defaulting to the GPU is safe
- // because the lane yields the card to transcription rather than sharing it.
- backend: "vulkan",
- python: "python3",
- segModel: "",
- embModel: "",
- sortformerBin: "",
- sortformerModel: "",
- concurrency: 1,
- // OFF, because windowing made it unnecessary — which is what it was always
- // for. It shipped at 4 hours as a stopgap while long recordings were being
- // OOM-killed; the engine now processes them in windows and the 6h12m file
- // that was being killed at 10.5 GB completes at ~2 GB. The knob stays, and
- // stays honest about what it does, for a machine smaller than this one or a
- // recording longer than anything measured here.
- maxAudioHours: 0,
- };
-}
-
-export function sanitizeDiarization(value: unknown): DiarizationSettings {
- const d = defaultDiarization();
- if (!value || typeof value !== "object") return d;
- const r = value as Record<string, unknown>;
- const str = (v: unknown, fallback: string) =>
- typeof v === "string" && v.trim() ? v.trim() : fallback;
- return {
- enabled: r.enabled === true,
- inlineAfterTranscribe: r.inlineAfterTranscribe === true,
- threshold:
- typeof r.threshold === "number" &&
- Number.isFinite(r.threshold) &&
- r.threshold > 0
- ? r.threshold
- : d.threshold,
- threads: clampPositiveInt(r.threads, d.threads, 64),
- // An unknown engine falls back to the default rather than disabling the lane:
- // a typo in settings.json must not silently stop diarization, and the default
- // is the one every existing sidecar already matches.
- engine: DIARIZATION_ENGINE_IDS.includes(r.engine as DiarizationEngineId)
- ? (r.engine as DiarizationEngineId)
- : d.engine,
- backend: DIARIZATION_BACKENDS.includes(r.backend as DiarizationBackend)
- ? (r.backend as DiarizationBackend)
- : d.backend,
- python: str(r.python, d.python),
- segModel: str(r.segModel, d.segModel),
- embModel: str(r.embModel, d.embModel),
- sortformerBin: str(r.sortformerBin, d.sortformerBin),
- sortformerModel: str(r.sortformerModel, d.sortformerModel),
- concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16),
- // 0 is meaningful here (cap off), so this cannot use clampPositiveInt.
- // Fractional hours are allowed — the knob is a duration, not a count.
- maxAudioHours:
- typeof r.maxAudioHours === "number" &&
- Number.isFinite(r.maxAudioHours) &&
- r.maxAudioHours >= 0
- ? r.maxAudioHours
- : d.maxAudioHours,
- };
-}
-
-const SOCIAL_URL_RE = /^(https?:\/\/|mailto:|\/)/i;
-
-// Normalize the family hub URL into a trailing-slash-free absolute http(s) URL.
-// Returns "" for anything that isn't a usable absolute URL (the "no hub" state).
-// Kept local to settings.ts to avoid a settings<->homepage import cycle; mirrors
-// parseHomepageUrl() in homepage.ts.
-export function normalizeHomepageUrl(input: unknown): string {
- if (typeof input !== "string") return "";
- const trimmed = input.trim().replace(/\/+$/, "");
- return /^https?:\/\/\S+/i.test(trimmed) ? trimmed : "";
-}
-
-export function parseSocialLinks(input: unknown): SocialLink[] {
- if (!Array.isArray(input)) return [];
- const out: SocialLink[] = [];
- for (const raw of input) {
- if (!raw || typeof raw !== "object") continue;
- const r = raw as Record<string, unknown>;
- const label = typeof r.label === "string" ? r.label.trim() : "";
- const url = typeof r.url === "string" ? r.url.trim() : "";
- const svg = typeof r.svg === "string" ? r.svg : "";
- if (!label || !url || !svg) continue;
- if (!SOCIAL_URL_RE.test(url)) continue;
- out.push({ label, url, svg });
- }
- return out;
-}
-
-// Normalize an admin-provided SVG snippet for inline use in the export
-// footer. Returns null on anything that looks unsafe or unrenderable.
-// Steps: trim, allowlist-check, strip width/height, force fill="currentColor"
-// + aria-hidden on the root <svg>. Requires a viewBox so the icon scales.
-export function normalizeSocialSvg(raw: string): string | null {
- if (typeof raw !== "string") return null;
- const trimmed = raw.trim();
- if (!trimmed.startsWith("<svg") || !trimmed.endsWith("</svg>")) return null;
- if (/<script\b/i.test(trimmed)) return null;
- if (/<foreignObject\b/i.test(trimmed)) return null;
- if (/<iframe\b/i.test(trimmed)) return null;
- if (/javascript:/i.test(trimmed)) return null;
- if (/\son[a-z]+\s*=/i.test(trimmed)) return null;
- if (/<\?|<!ENTITY/i.test(trimmed)) return null;
-
- const openEnd = trimmed.indexOf(">");
- if (openEnd < 0) return null;
- let opening = trimmed.slice(0, openEnd);
- const rest = trimmed.slice(openEnd);
-
- if (!/\sviewBox\s*=\s*"/i.test(opening)) return null;
-
- opening = opening.replace(/\s(width|height)\s*=\s*"[^"]*"/gi, "");
- opening = opening.replace(/\s(width|height)\s*=\s*'[^']*'/gi, "");
-
- if (!/\sfill\s*=/i.test(opening)) {
- opening = opening.replace(/^<svg/i, '<svg fill="currentColor"');
- }
- if (!/\saria-hidden\s*=/i.test(opening)) {
- opening = opening.replace(/^<svg/i, '<svg aria-hidden="true"');
- }
- return opening + rest;
-}
-
-// normalizeSocialSvg() deliberately STRIPS width/height so the icon scales to its
-// wrapper. The cost is that a viewBox-only <svg> has no intrinsic size, so before
-// the stylesheet loads on a static host it paints at the replaced-element default
-// (huge) — the "flash of giant social icons" FOUC. sizeSocialSvg() re-injects an
-// intrinsic pixel size at RENDER time (existing site.json files already have the
-// attributes stripped, so this must run on read, not just on write). The size is
-// an *attribute*, not inline style, so a wrapper's `w-*`/`h-*` utilities still win
-// once CSS loads — it only governs the pre-CSS first paint.
-export function sizeSocialSvg(svg: string, px = 20): string {
- if (typeof svg !== "string") return svg;
- if (/^<svg[^>]*\swidth\s*=/i.test(svg)) return svg; // already sized
- return svg.replace(/^<svg\b/i, `<svg width="${px}" height="${px}"`);
-}
-
-export function clampSleepBetweenDownloadsSeconds(value: unknown): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? Math.floor(value)
- : SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS;
- if (n < 0) return 0;
- if (n > SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS) {
- return SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS;
- }
- return n;
-}
-
-export function clampMinFreeDiskGB(value: unknown): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? Math.floor(value)
- : MIN_FREE_DISK_GB_DEFAULT;
- if (n < 0) return 0;
- if (n > MIN_FREE_DISK_GB_MAX) return MIN_FREE_DISK_GB_MAX;
- return n;
-}
-
-export function clampResumeMarginGB(value: unknown): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? Math.floor(value)
- : RESUME_MARGIN_GB_DEFAULT;
- if (n < 0) return 0;
- if (n > RESUME_MARGIN_GB_MAX) return RESUME_MARGIN_GB_MAX;
- return n;
-}
-
-export function clampParallelTranscriptions(value: unknown): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? Math.floor(value)
- : PARALLEL_TRANSCRIPTIONS_DEFAULT;
- if (n < 1) return 1;
- if (n > PARALLEL_TRANSCRIPTIONS_MAX) return PARALLEL_TRANSCRIPTIONS_MAX;
- return n;
-}
-
-// 0 means "disabled" and is preserved as-is. Anything else is clamped into the
-// [MIN, MAX] window; a non-finite value falls back to the default cadence.
-export function clampAutoRefreshIntervalSeconds(value: unknown): number {
- if (typeof value !== "number" || !Number.isFinite(value)) {
- return AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS;
- }
- const n = Math.floor(value);
- if (n <= 0) return 0;
- if (n < AUTO_REFRESH_INTERVAL_MIN_SECONDS) {
- return AUTO_REFRESH_INTERVAL_MIN_SECONDS;
- }
- if (n > AUTO_REFRESH_INTERVAL_MAX_SECONDS) {
- return AUTO_REFRESH_INTERVAL_MAX_SECONDS;
- }
- return n;
-}
-
-export function getSettings(): SiteSettings {
- const file = getPaths().settingsFile;
- let parsed: Partial<SiteSettings> = {};
- try {
- parsed = JSON.parse(fs.readFileSync(file, "utf8")) as Partial<SiteSettings>;
- } catch {
- parsed = {};
- }
- // Pick only known operational keys — a pre-multi-site settings.json may still
- // carry siteTitle/groups/socialLinks, which now live per-site in site.json.
- const merged: SiteSettings = { ...defaults() };
- const knownKeys = Object.keys(merged) as (keyof SiteSettings)[];
- for (const key of knownKeys) {
- if (parsed[key] !== undefined) {
- (merged as Record<string, unknown>)[key] = parsed[key];
- }
- }
- if (typeof merged.adminTitle !== "string" || !merged.adminTitle.trim()) {
- merged.adminTitle = DEFAULT_ADMIN_TITLE;
- }
- merged.maxTranscriptPageBytes = clampPageBytes(merged.maxTranscriptPageBytes);
- merged.transcriptionApps = sanitizeTranscriptionApps(merged.transcriptionApps);
- // Migrate a pre-multi-app settings.json (transcribeBin/transcribeArgs/
- // transcribeModel, with no transcriptionApp key) onto the app registry.
- if (parsed.transcriptionApp === undefined) {
- migrateLegacyTranscription(parsed as LegacyTranscribeFields, merged);
- }
- if (
- typeof merged.transcriptionApp !== "string" ||
- !TRANSCRIPTION_APPS[merged.transcriptionApp]
- ) {
- merged.transcriptionApp = DEFAULT_TRANSCRIPTION_APP_ID;
- }
- if (typeof merged.cookiesFromBrowser !== "string") {
- merged.cookiesFromBrowser = "";
- }
- // A settings.json predating cookieMode (or carrying junk) gets the default,
- // which preserves the historical retry-only behavior.
- if (!isCookieMode(merged.cookieMode)) {
- merged.cookieMode = DEFAULT_COOKIE_MODE;
- }
- merged.sleepBetweenDownloadsSeconds = clampSleepBetweenDownloadsSeconds(
- merged.sleepBetweenDownloadsSeconds,
- );
- if (!isDownloadFormatPreset(merged.downloadFormat)) {
- merged.downloadFormat = "auto";
- }
- merged.minFreeDiskGB = clampMinFreeDiskGB(merged.minFreeDiskGB);
- merged.resumeMarginGB = clampResumeMarginGB(merged.resumeMarginGB);
- merged.parallelTranscriptions = clampParallelTranscriptions(
- merged.parallelTranscriptions,
- );
- if (typeof merged.inlineTranscribeOnFallback !== "boolean") {
- merged.inlineTranscribeOnFallback = false;
- }
- if (typeof merged.skipLiveDownloads !== "boolean") {
- merged.skipLiveDownloads = true;
- }
- if (typeof merged.verifyAvailabilityBeforeClean !== "boolean") {
- merged.verifyAvailabilityBeforeClean = true;
- }
- if (typeof merged.buildArchives !== "boolean") {
- merged.buildArchives = true;
- }
- {
- const s = merged.archiveStorage;
- merged.archiveStorage = {
- bucket: s && typeof s.bucket === "string" ? s.bucket : "",
- publicBaseUrl:
- s && typeof s.publicBaseUrl === "string" ? s.publicBaseUrl : "",
- };
- }
- if (!isReportDebouncePreset(merged.reportDebouncePreset)) {
- merged.reportDebouncePreset = DEFAULT_REPORT_DEBOUNCE_PRESET;
- }
- merged.autoRefreshIntervalSeconds = clampAutoRefreshIntervalSeconds(
- merged.autoRefreshIntervalSeconds,
- );
- merged.syncScheduler = sanitizeSyncScheduler(merged.syncScheduler);
- // THE SWEEPS' SCOPE, ON READ. `migrateSweepsToLanes` fills in
- // `autoQueue.digest` / `.backfill` from the retired sweep fields when — and
- // only when — the FILE does not already spell them, which is why it is handed
- // `parsed` rather than `merged`: `merged` has had defaults folded in and can
- // no longer tell "absent" from "default". It never enables a lane the sweep
- // flag did not. See lib/laneMigration.ts.
- merged.autoQueue = sanitizeAutoQueue(migrateSweepsToLanes(parsed));
- // No migration beside it: the legacy read (`channelPriorityFromLegacy`) needs
- // 68 config.json files and getSettings is synchronous and reads one. It is a
- // one-shot offline script instead, and an absent document sanitizes to the
- // empty one, which means today's behaviour.
- merged.channelPriority = sanitizeChannelPriority(merged.channelPriority);
- merged.socialLinks = parseSocialLinks(merged.socialLinks);
- merged.homepageUrl = normalizeHomepageUrl(merged.homepageUrl);
- merged.savedVideoBackup = sanitizeSavedVideoBackup(merged.savedVideoBackup);
- // THE COLD ROOT, ON READ. `storage.mediaRoot` — one absolute string — becomes
- // a one-entry location list. Handed `parsed.storage` rather than
- // `merged.storage` for the same reason `migrateSweepsToLanes` is handed
- // `parsed`: `merged` has had `defaultStorage()` folded in and can no longer
- // tell "the file has no locations key" from "the file has an empty list", and
- // the migration must only fire on the former. See lib/storageLocations.ts.
- merged.storage = sanitizeStorage(
- migrateMediaRootToLocations(
- (parsed as Record<string, unknown>).storage ?? merged.storage,
- ),
- );
- merged.buildPipeline = sanitizeBuildPipeline(merged.buildPipeline);
- merged.digest = sanitizeDigest(merged.digest);
- merged.diarization = sanitizeDiarization(merged.diarization);
- merged.backfill = sanitizeBackfill(merged.backfill);
- merged.attribution = sanitizeAttribution(merged.attribution);
- // Workers. When the file predates the worker model (no `workers` key),
- // synthesize a default list from the (now-settled) active app + per-app
- // configs so existing installs behave identically. Otherwise sanitize the
- // stored list.
- if (parsed.workers === undefined) {
- merged.workers = defaultWorkersFromApps(
- merged.transcriptionApp,
- merged.transcriptionApps,
- merged.parallelTranscriptions,
+// 1. A pre-multi-app file (transcribeBin/transcribeArgs/transcribeModel, no
+// `transcriptionApp` key) is migrated onto the app registry. It reads the
+// already-sanitized `transcriptionApps`, so it runs after the parse.
+// 2. A file predating the worker model (no `workers` key) gets a worker list
+// synthesized from the now-settled active app, so existing installs behave
+// identically. A file that HAS the key keeps its sanitized list, even an
+// empty one.
+function finishRawMigrations(
+ parsed: SiteSettings,
+ raw: RawSettings,
+): SiteSettings {
+ if (raw.transcriptionApp === undefined) {
+ migrateLegacyTranscription(raw as LegacyTranscribeFields, parsed);
+ }
+ if (raw.workers === undefined) {
+ parsed.workers = defaultWorkersFromApps(
+ parsed.transcriptionApp,
+ parsed.transcriptionApps,
+ parsed.parallelTranscriptions,
);
- } else {
- merged.workers = sanitizeWorkers(merged.workers);
}
- return merged;
+ return parsed;
}
-function clampPageBytes(value: unknown): number {
- const n =
- typeof value === "number" && Number.isFinite(value)
- ? value
- : TRANSCRIPT_PAGE_DEFAULT_BYTES;
- if (n < TRANSCRIPT_PAGE_MIN_BYTES) return TRANSCRIPT_PAGE_MIN_BYTES;
- if (n > TRANSCRIPT_PAGE_HARD_CAP_BYTES) return TRANSCRIPT_PAGE_HARD_CAP_BYTES;
- return Math.floor(n);
+export function getSettings(): SiteSettings {
+ const raw = rawObject(readRawSettings(getPaths().settingsFile));
+ return finishRawMigrations(siteSettingsSchema.parse(premigrateRaw(raw)), raw);
}
type LegacyTranscribeFields = {
@@ -1601,20 +132,6 @@ type LegacyTranscribeFields = {
transcribeModel?: unknown;
};
-// Coerce a raw settings.transcriptionApps value into a clean keyed map of
-// AppInstanceConfig, dropping unknown/ill-typed fields.
-export function sanitizeTranscriptionApps(
- value: unknown,
-): Record<string, AppInstanceConfig> {
- if (!value || typeof value !== "object" || Array.isArray(value)) return {};
- const out: Record<string, AppInstanceConfig> = {};
- for (const [id, raw] of Object.entries(value as Record<string, unknown>)) {
- if (!raw || typeof raw !== "object") continue;
- out[id] = sanitizeWorkerConfig(raw);
- }
- return out;
-}
-
function argsAreDefault(args: string[]): boolean {
return (
args.length === DEFAULT_TRANSCRIBE_ARGS.length &&
@@ -1658,10 +175,16 @@ function migrateLegacyTranscription(
merged.transcriptionApps = { ...merged.transcriptionApps, [appId]: cfg };
}
-export async function writeSettings(next: SiteSettings): Promise<void> {
- // Workers are the source of truth. A caller that still sets only the
- // deprecated transcriptionApp/transcriptionApps (no `workers`) gets a list
- // synthesized from them, so old call sites keep working during the transition.
+// THE WORKER SHADOW, derived on every save.
+//
+// Workers are the source of truth. A caller that still sets only the deprecated
+// transcriptionApp/transcriptionApps (no `workers`) gets a list synthesized from
+// them, so old call sites keep working. The list is then VALIDATED — this throws
+// — and the deprecated pair is rewritten as a faithful shadow of it (used only
+// for rollback to a pre-worker build; a file with a `workers` key is never
+// re-migrated): the active app is the first enabled local worker, and the per-app
+// map mirrors each local worker's config.
+function deriveWorkerShadow(next: SiteSettings): SiteSettings {
let workers = sanitizeWorkers(next.workers);
if (workers.length === 0) {
const fallbackAppId =
@@ -1678,14 +201,10 @@ export async function writeSettings(next: SiteSettings): Promise<void> {
const workersErr = validateWorkers(workers);
if (workersErr) throw new Error(workersErr);
- // Keep the deprecated transcriptionApp/transcriptionApps as a faithful shadow
- // of the workers (used only for rollback to a pre-worker build — a file with a
- // `workers` key is never re-migrated). The active app is the first enabled
- // local worker; the per-app map mirrors each local worker's config.
const firstLocal =
workers.find((w) => w.enabled && w.kind === "local") ??
workers.find((w) => w.kind === "local");
- const appId =
+ const transcriptionApp =
firstLocal?.appId && TRANSCRIPTION_APPS[firstLocal.appId]
? firstLocal.appId
: DEFAULT_TRANSCRIPTION_APP_ID;
@@ -1695,87 +214,36 @@ export async function writeSettings(next: SiteSettings): Promise<void> {
transcriptionApps[w.appId] = w.config ?? {};
}
}
- const socialLinks: SocialLink[] = [];
- for (const link of parseSocialLinks(next.socialLinks)) {
+ return { ...next, workers, transcriptionApp, transcriptionApps };
+}
+
+// Every social link's SVG normalized for inline use, or a THROW naming the
+// first one that is not safe to inline. The schema's own `parseSocialLinks`
+// only checks shape; this is the write-side half.
+function validatedSocialLinks(value: unknown): SocialLink[] {
+ const out: SocialLink[] = [];
+ for (const link of parseSocialLinks(value)) {
const svg = normalizeSocialSvg(link.svg);
if (svg === null) {
throw new Error(`Social link "${link.label}" has an invalid SVG`);
}
- socialLinks.push({ ...link, svg });
+ out.push({ ...link, svg });
}
- const file = getPaths().settingsFile;
- // Build the output from ONLY the known operational fields. Do NOT spread
- // `next`: getSettings() spreads the raw file, so a settings.json still
- // carrying pre-multi-site keys (siteTitle/groups/socialLinks) would otherwise
- // smuggle those stale keys back onto disk on every save.
- //
- // THIS IS ALSO WHAT RETIRES A FIELD. The four legacy pause flags
- // (`transcriptionsPaused`, `downloadsPaused`, `digest.digestsPaused` and the
- // inverted `backfill.enabled`) are gone from the type, from the sanitizers
- // and from this literal, so a settings.json that still spells one is read
- // past on load and loses it on the next write. The gate is
+ return out;
+}
+
+export async function writeSettings(next: SiteSettings): Promise<void> {
+ // THE SCHEMA IS ALSO WHAT RETIRES A FIELD. It names only the known
+ // operational fields and strips the rest, so a settings.json still carrying
+ // pre-multi-site keys (siteTitle/groups) or the four retired pause flags
+ // (`transcriptionsPaused`, `downloadsPaused`, `digest.digestsPaused`, the
+ // inverted `backfill.enabled`) loses them on this write. The gate is
// `autoQueue[lane].held` and nothing else — see lib/pauseGates.ts.
- const merged: SiteSettings = {
- adminTitle:
- typeof next.adminTitle === "string" && next.adminTitle.trim()
- ? next.adminTitle.trim()
- : DEFAULT_ADMIN_TITLE,
- maxTranscriptPageBytes: clampPageBytes(next.maxTranscriptPageBytes),
- transcriptionApp: appId,
- transcriptionApps,
- workers,
- cookiesFromBrowser:
- typeof next.cookiesFromBrowser === "string"
- ? next.cookiesFromBrowser.trim()
- : "",
- cookieMode: isCookieMode(next.cookieMode)
- ? next.cookieMode
- : DEFAULT_COOKIE_MODE,
- sleepBetweenDownloadsSeconds: clampSleepBetweenDownloadsSeconds(
- next.sleepBetweenDownloadsSeconds,
- ),
- downloadFormat: isDownloadFormatPreset(next.downloadFormat)
- ? next.downloadFormat
- : "auto",
- minFreeDiskGB: clampMinFreeDiskGB(next.minFreeDiskGB),
- resumeMarginGB: clampResumeMarginGB(next.resumeMarginGB),
- parallelTranscriptions: clampParallelTranscriptions(
- next.parallelTranscriptions,
- ),
- inlineTranscribeOnFallback: next.inlineTranscribeOnFallback === true,
- skipLiveDownloads: next.skipLiveDownloads !== false,
- verifyAvailabilityBeforeClean:
- next.verifyAvailabilityBeforeClean !== false,
- buildArchives: next.buildArchives !== false,
- archiveStorage: {
- bucket:
- typeof next.archiveStorage?.bucket === "string"
- ? next.archiveStorage.bucket.trim()
- : "",
- publicBaseUrl:
- typeof next.archiveStorage?.publicBaseUrl === "string"
- ? next.archiveStorage.publicBaseUrl.trim()
- : "",
- },
- reportDebouncePreset: isReportDebouncePreset(next.reportDebouncePreset)
- ? next.reportDebouncePreset
- : DEFAULT_REPORT_DEBOUNCE_PRESET,
- autoRefreshIntervalSeconds: clampAutoRefreshIntervalSeconds(
- next.autoRefreshIntervalSeconds,
- ),
- syncScheduler: sanitizeSyncScheduler(next.syncScheduler),
- autoQueue: sanitizeAutoQueue(next.autoQueue),
- channelPriority: sanitizeChannelPriority(next.channelPriority),
- socialLinks,
- homepageUrl: normalizeHomepageUrl(next.homepageUrl),
- savedVideoBackup: sanitizeSavedVideoBackup(next.savedVideoBackup),
- storage: sanitizeStorage(next.storage),
- buildPipeline: sanitizeBuildPipeline(next.buildPipeline),
- digest: sanitizeDigest(next.digest),
- diarization: sanitizeDiarization(next.diarization),
- backfill: sanitizeBackfill(next.backfill),
- attribution: sanitizeAttribution(next.attribution),
- };
+ const merged = siteSettingsSchema.parse({
+ ...deriveWorkerShadow(next),
+ socialLinks: validatedSocialLinks(next.socialLinks),
+ });
+ const file = getPaths().settingsFile;
const tmp = `${file}.tmp-${process.pid}`;
await fs.promises.writeFile(tmp, JSON.stringify(merged, null, 2) + "\n");
await fs.promises.rename(tmp, file);
diff --git a/common/lib/settingsSchema.test.ts b/common/lib/settingsSchema.test.ts
@@ -0,0 +1,425 @@
+// THE SETTINGS SCHEMA: its shape, its boundaries, its migrations, and the
+// promise that a read never throws.
+//
+// Run with: node_modules/.bin/tsx --test common/lib/settingsSchema.test.ts
+//
+// SETTINGS SEAM, same arrangement as ./settingsWrite.test.ts: `getPaths()`
+// memoizes at module scope, so SETTINGS_FILE is set before anything imports
+// ./settings, and the module is imported dynamically below.
+
+import { mkdtempSync, writeFileSync } from "node:fs";
+import { rm } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { test, after } from "node:test";
+import assert from "node:assert/strict";
+
+const ROOT = mkdtempSync(path.join(os.tmpdir(), "settings-schema-"));
+process.env.TRANSCRIPTS_DIR = ROOT;
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+
+import type { SiteSettings } from "./settings";
+import type { AppInstanceConfig } from "./transcriptionApps";
+import type { Worker } from "./workers";
+import type { AutoQueueSettings } from "./autoQueueTypes";
+import type { ChannelPriority } from "./channelPriority";
+import type { CookieMode } from "./cookiePolicy";
+import type { DownloadFormatPreset } from "../ytdlp/downloadFormat";
+import type { StorageSettings } from "./storageLocations";
+import type {
+ AttributionSettings,
+ BackfillSettings,
+ BuildPipelineSettings,
+ DiarizationSettings,
+ DigestSettings,
+ ReportDebouncePreset,
+ SavedVideoBackupSettings,
+ SocialLink,
+ SyncSchedulerSettings,
+} from "./settingsSchema";
+
+const S = await import("./settings");
+const { siteSettingsSchema, defaultSiteSettings, defaults, getSettings } = S;
+
+after(() => rm(ROOT, { recursive: true, force: true }));
+
+// ── THE SHAPE, PINNED ────────────────────────────────────────────────────────
+//
+// `SiteSettings` is `z.infer<typeof siteSettingsSchema>` now. This is the type
+// as it was hand-written before slice 4a, spelled out literally, so a schema
+// edit that changes what 200 importers see is a tsc error here — not a surprise
+// somewhere else.
+//
+// ONE DELIBERATE DIFFERENCE: `archiveStorage` was `archiveStorage?: {…}`. It was
+// never absent at runtime (defaults(), getSettings and writeSettings all
+// emitted it), and zod 4 cannot express "optional key that is always emitted",
+// so it is required now. Nothing in the repo relied on the `?`.
+type PreSchemaSiteSettings = {
+ adminTitle: string;
+ maxTranscriptPageBytes: number;
+ transcriptionApp: string;
+ transcriptionApps: Record<string, AppInstanceConfig>;
+ workers: Worker[];
+ cookiesFromBrowser: string;
+ cookieMode: CookieMode;
+ sleepBetweenDownloadsSeconds: number;
+ downloadFormat: DownloadFormatPreset;
+ minFreeDiskGB: number;
+ resumeMarginGB: number;
+ parallelTranscriptions: number;
+ inlineTranscribeOnFallback: boolean;
+ skipLiveDownloads: boolean;
+ verifyAvailabilityBeforeClean: boolean;
+ buildArchives: boolean;
+ archiveStorage: { bucket: string; publicBaseUrl: string };
+ reportDebouncePreset: ReportDebouncePreset;
+ autoRefreshIntervalSeconds: number;
+ syncScheduler: SyncSchedulerSettings;
+ autoQueue: AutoQueueSettings;
+ channelPriority: ChannelPriority;
+ socialLinks: SocialLink[];
+ homepageUrl: string;
+ savedVideoBackup: SavedVideoBackupSettings;
+ storage: StorageSettings;
+ buildPipeline: BuildPipelineSettings;
+ digest: DigestSettings;
+ diarization: DiarizationSettings;
+ backfill: BackfillSettings;
+ attribution: AttributionSettings;
+};
+
+// Bracketed so the conditional does not distribute (see commit 8c43231).
+type Same<A, B> = [A] extends [B] ? ([B] extends [A] ? true : false) : false;
+const shapeUnchanged: Same<SiteSettings, PreSchemaSiteSettings> = true;
+
+test("SiteSettings keeps its 31 fields, in file order", () => {
+ assert.equal(shapeUnchanged, true);
+ assert.deepEqual(Object.keys(siteSettingsSchema.shape), [
+ "adminTitle",
+ "maxTranscriptPageBytes",
+ "transcriptionApp",
+ "transcriptionApps",
+ "workers",
+ "cookiesFromBrowser",
+ "cookieMode",
+ "sleepBetweenDownloadsSeconds",
+ "downloadFormat",
+ "minFreeDiskGB",
+ "resumeMarginGB",
+ "parallelTranscriptions",
+ "inlineTranscribeOnFallback",
+ "skipLiveDownloads",
+ "verifyAvailabilityBeforeClean",
+ "buildArchives",
+ "archiveStorage",
+ "reportDebouncePreset",
+ "autoRefreshIntervalSeconds",
+ "syncScheduler",
+ "autoQueue",
+ "channelPriority",
+ "socialLinks",
+ "homepageUrl",
+ "savedVideoBackup",
+ "storage",
+ "buildPipeline",
+ "digest",
+ "diarization",
+ "backfill",
+ "attribution",
+ ]);
+ // A parsed object carries every key, in that order — writeSettings writes
+ // exactly this, so the order is the on-disk order.
+ assert.deepEqual(
+ Object.keys(defaults()),
+ Object.keys(siteSettingsSchema.shape),
+ );
+});
+
+test("every field has a description (SETTINGS.md is generated from them)", () => {
+ for (const [key, field] of Object.entries(siteSettingsSchema.shape)) {
+ assert.ok(
+ typeof field.description === "string" && field.description.length > 20,
+ `${key} has no .describe()`,
+ );
+ }
+});
+
+test("defaults() and defaultSiteSettings() are the schema's answer for an empty file", () => {
+ assert.deepEqual(defaultSiteSettings(), siteSettingsSchema.parse({}));
+ assert.deepEqual(defaults(), defaultSiteSettings());
+ const d = defaults();
+ assert.equal(d.adminTitle, S.DEFAULT_ADMIN_TITLE);
+ assert.equal(d.maxTranscriptPageBytes, S.TRANSCRIPT_PAGE_DEFAULT_BYTES);
+ assert.deepEqual(d.workers, []);
+ assert.deepEqual(d.archiveStorage, { bucket: "", publicBaseUrl: "" });
+ assert.equal(d.skipLiveDownloads, true);
+ assert.equal(d.inlineTranscribeOnFallback, false);
+});
+
+// ── THE CLAMP BOUNDARIES ─────────────────────────────────────────────────────
+
+type Case = [input: unknown, expected: number];
+
+function checkField(
+ key: keyof SiteSettings,
+ cases: Case[],
+): void {
+ for (const [input, expected] of cases) {
+ const got = siteSettingsSchema.parse({ [key]: input })[key];
+ assert.equal(got, expected, `${key}: ${JSON.stringify(input)} -> ${got}`);
+ }
+}
+
+// min / max / NaN / string / null / absent, for each clamped top-level field.
+test("maxTranscriptPageBytes clamps into [256 KiB, 20 MiB]", () => {
+ checkField("maxTranscriptPageBytes", [
+ [0, S.TRANSCRIPT_PAGE_MIN_BYTES],
+ [S.TRANSCRIPT_PAGE_MIN_BYTES - 1, S.TRANSCRIPT_PAGE_MIN_BYTES],
+ [S.TRANSCRIPT_PAGE_MIN_BYTES, S.TRANSCRIPT_PAGE_MIN_BYTES],
+ [S.TRANSCRIPT_PAGE_HARD_CAP_BYTES, S.TRANSCRIPT_PAGE_HARD_CAP_BYTES],
+ [S.TRANSCRIPT_PAGE_HARD_CAP_BYTES + 1, S.TRANSCRIPT_PAGE_HARD_CAP_BYTES],
+ [1_000_000.7, 1_000_000],
+ [NaN, S.TRANSCRIPT_PAGE_DEFAULT_BYTES],
+ ["9000000", S.TRANSCRIPT_PAGE_DEFAULT_BYTES],
+ [null, S.TRANSCRIPT_PAGE_DEFAULT_BYTES],
+ [undefined, S.TRANSCRIPT_PAGE_DEFAULT_BYTES],
+ ]);
+});
+
+test("sleepBetweenDownloadsSeconds: 0 is off, capped at the max", () => {
+ checkField("sleepBetweenDownloadsSeconds", [
+ [-1, 0],
+ [0, 0],
+ [S.SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS, S.SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS],
+ [S.SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS + 1, S.SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS],
+ [2.9, 2],
+ [NaN, S.SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS],
+ ["5", S.SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS],
+ [null, S.SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS],
+ ]);
+});
+
+test("minFreeDiskGB: 0 disables the gate, capped at the max", () => {
+ checkField("minFreeDiskGB", [
+ [-5, 0],
+ [0, 0],
+ [S.MIN_FREE_DISK_GB_MAX, S.MIN_FREE_DISK_GB_MAX],
+ [S.MIN_FREE_DISK_GB_MAX + 1, S.MIN_FREE_DISK_GB_MAX],
+ [NaN, S.MIN_FREE_DISK_GB_DEFAULT],
+ ["0", S.MIN_FREE_DISK_GB_DEFAULT],
+ [null, S.MIN_FREE_DISK_GB_DEFAULT],
+ ]);
+});
+
+test("resumeMarginGB: 0 disables the hysteresis, capped at the max", () => {
+ checkField("resumeMarginGB", [
+ [-1, 0],
+ [0, 0],
+ [S.RESUME_MARGIN_GB_MAX, S.RESUME_MARGIN_GB_MAX],
+ [S.RESUME_MARGIN_GB_MAX + 1, S.RESUME_MARGIN_GB_MAX],
+ [NaN, S.RESUME_MARGIN_GB_DEFAULT],
+ ["2", S.RESUME_MARGIN_GB_DEFAULT],
+ [null, S.RESUME_MARGIN_GB_DEFAULT],
+ ]);
+});
+
+test("parallelTranscriptions: floored at 1, capped at the max", () => {
+ checkField("parallelTranscriptions", [
+ [0, 1],
+ [-3, 1],
+ [1, 1],
+ [S.PARALLEL_TRANSCRIPTIONS_MAX, S.PARALLEL_TRANSCRIPTIONS_MAX],
+ [S.PARALLEL_TRANSCRIPTIONS_MAX + 1, S.PARALLEL_TRANSCRIPTIONS_MAX],
+ [NaN, S.PARALLEL_TRANSCRIPTIONS_DEFAULT],
+ ["4", S.PARALLEL_TRANSCRIPTIONS_DEFAULT],
+ [null, S.PARALLEL_TRANSCRIPTIONS_DEFAULT],
+ ]);
+});
+
+test("autoRefreshIntervalSeconds: 0 survives as off, the rest clamps", () => {
+ checkField("autoRefreshIntervalSeconds", [
+ [0, 0],
+ [-10, 0],
+ [0.5, 0],
+ [S.AUTO_REFRESH_INTERVAL_MIN_SECONDS, S.AUTO_REFRESH_INTERVAL_MIN_SECONDS],
+ [S.AUTO_REFRESH_INTERVAL_MAX_SECONDS, S.AUTO_REFRESH_INTERVAL_MAX_SECONDS],
+ [S.AUTO_REFRESH_INTERVAL_MAX_SECONDS + 1, S.AUTO_REFRESH_INTERVAL_MAX_SECONDS],
+ [NaN, S.AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS],
+ ["5", S.AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS],
+ [null, S.AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS],
+ ]);
+});
+
+test("syncScheduler.heartbeatSeconds: 0 is off, positive clamps into [MIN, MAX]", () => {
+ const hb = (v: unknown) =>
+ siteSettingsSchema.parse({ syncScheduler: { heartbeatSeconds: v } })
+ .syncScheduler.heartbeatSeconds;
+ assert.equal(hb(0), 0);
+ assert.equal(hb(-1), 0);
+ assert.equal(hb(1), S.SYNC_HEARTBEAT_MIN_SECONDS);
+ assert.equal(hb(S.SYNC_HEARTBEAT_MAX_SECONDS + 1), S.SYNC_HEARTBEAT_MAX_SECONDS);
+ assert.equal(hb(NaN), S.SYNC_HEARTBEAT_DEFAULT_SECONDS);
+ assert.equal(hb("60"), S.SYNC_HEARTBEAT_DEFAULT_SECONDS);
+ assert.equal(hb(null), S.SYNC_HEARTBEAT_DEFAULT_SECONDS);
+});
+
+test("sync cadences that allow zero keep it; the ones that do not floor at 1", () => {
+ const s = siteSettingsSchema.parse({
+ syncScheduler: {
+ fullSweepIntervalMinutes: 0,
+ fullSweepConfirmMaxSuspects: 0,
+ fullSweepShrinkGuardPercent: 0,
+ defaultIntervalMinutes: 0,
+ maxConcurrentSyncs: 0,
+ quietHoursStart: 24,
+ quietHoursEnd: 6,
+ },
+ }).syncScheduler;
+ assert.equal(s.fullSweepIntervalMinutes, 0);
+ assert.equal(s.fullSweepConfirmMaxSuspects, 0);
+ assert.equal(s.fullSweepShrinkGuardPercent, 0);
+ assert.equal(s.defaultIntervalMinutes, 1);
+ assert.equal(s.maxConcurrentSyncs, 1);
+ // An out-of-range hour clears BOTH ends of the window.
+ assert.equal(s.quietHoursStart, null);
+ assert.equal(s.quietHoursEnd, null);
+});
+
+test("enum fields fall back to their default on junk", () => {
+ const s = siteSettingsSchema.parse({
+ cookieMode: "sometimes",
+ downloadFormat: "best",
+ reportDebouncePreset: "instant",
+ transcriptionApp: "no-such-app",
+ });
+ const d = defaults();
+ assert.equal(s.cookieMode, d.cookieMode);
+ assert.equal(s.downloadFormat, "auto");
+ assert.equal(s.reportDebouncePreset, d.reportDebouncePreset);
+ assert.equal(s.transcriptionApp, d.transcriptionApp);
+});
+
+test("booleans: only an explicit value moves them off their default", () => {
+ const s = siteSettingsSchema.parse({
+ inlineTranscribeOnFallback: "yes",
+ skipLiveDownloads: "no",
+ verifyAvailabilityBeforeClean: 0,
+ buildArchives: false,
+ });
+ assert.equal(s.inlineTranscribeOnFallback, false);
+ assert.equal(s.skipLiveDownloads, true);
+ assert.equal(s.verifyAvailabilityBeforeClean, true);
+ assert.equal(s.buildArchives, false);
+});
+
+test("an unknown key is stripped, at the top level and inside a block", () => {
+ const s = siteSettingsSchema.parse({
+ siteTitle: "pre-multi-site",
+ transcriptionsPaused: true,
+ backfill: { enabled: true, concurrency: 2 },
+ }) as Record<string, unknown>;
+ assert.equal("siteTitle" in s, false);
+ assert.equal("transcriptionsPaused" in s, false);
+ assert.deepEqual(s.backfill, { concurrency: 2, allowRedownload: false });
+});
+
+// ── THE LANE GATE AND THE ZERO HOLD ──────────────────────────────────────────
+
+test("held defaults to [false, false, false, true] — backfill ships held", () => {
+ const aq = defaults().autoQueue;
+ assert.deepEqual(
+ [aq.transcription.held, aq.download.held, aq.digest.held, aq.backfill.held],
+ [false, false, false, true],
+ );
+ // And through a file that names the lanes but no gate.
+ const bare = siteSettingsSchema.parse({
+ autoQueue: { transcription: {}, download: {}, digest: {}, backfill: {} },
+ }).autoQueue;
+ assert.deepEqual(
+ [bare.transcription.held, bare.download.held, bare.digest.held, bare.backfill.held],
+ [false, false, false, true],
+ );
+});
+
+// ── getSettings: THE RAW MIGRATIONS AND THE NEVER-THROW ──────────────────────
+
+function withFile(contents: string): SiteSettings {
+ writeFileSync(process.env.SETTINGS_FILE!, contents);
+ return getSettings();
+}
+
+test("getSettings never throws on a file that is not a settings object", () => {
+ // THE EMPTY FILE is the reference, not bare defaults(): an empty file still
+ // runs the absence-keyed migrations (workers synthesized, the two sweep lanes
+ // built by migrateSweepsToLanes), and a non-object file must read as exactly
+ // that. Before slice 4a, a file containing `null` threw.
+ const empty = withFile("{}");
+ assert.equal(empty.workers.length, S.PARALLEL_TRANSCRIPTIONS_DEFAULT);
+ for (const contents of ["{", "null", "[]", "3", "", '"text"']) {
+ let got: SiteSettings | undefined;
+ assert.doesNotThrow(() => {
+ got = withFile(contents);
+ }, `contents ${JSON.stringify(contents)}`);
+ assert.deepEqual(got, empty, `contents ${JSON.stringify(contents)}`);
+ }
+});
+
+test("workers are synthesized ONLY when the key is absent", () => {
+ const absent = withFile(JSON.stringify({ parallelTranscriptions: 3 }));
+ assert.equal(absent.workers.length, 3);
+ assert.ok(absent.workers.every((w) => w.enabled && w.kind === "local"));
+ const empty = withFile(JSON.stringify({ workers: [] }));
+ assert.deepEqual(empty.workers, []);
+});
+
+test("legacy transcribe* fields migrate ONLY when transcriptionApp is absent", () => {
+ const legacy = {
+ transcribeBin: "/opt/chough",
+ transcribeModel: "/models/x.bin",
+ };
+ const migrated = withFile(JSON.stringify(legacy));
+ assert.equal(migrated.transcriptionApp, "chough");
+ assert.equal(migrated.transcriptionApps.chough?.bin, "/opt/chough");
+ assert.equal(migrated.workers[0].appId, "chough");
+
+ const named = withFile(
+ JSON.stringify({ ...legacy, transcriptionApp: "whisper-cpp" }),
+ );
+ assert.equal(named.transcriptionApp, "whisper-cpp");
+ assert.equal(named.transcriptionApps.chough, undefined);
+});
+
+test("storage.mediaRoot migrates ONLY when locations is absent", () => {
+ const migrated = withFile(
+ JSON.stringify({ storage: { mediaRoot: "/mnt/cold" } }),
+ );
+ assert.deepEqual(migrated.storage.locations.map((l) => l.root), ["/mnt/cold"]);
+ assert.equal(migrated.storage.defaultLocationId, "default");
+
+ const listed = withFile(
+ JSON.stringify({ storage: { mediaRoot: "/mnt/cold", locations: [] } }),
+ );
+ assert.deepEqual(listed.storage.locations, []);
+
+ const none = withFile(JSON.stringify({}));
+ assert.deepEqual(none.storage, { locations: [], defaultLocationId: "" });
+});
+
+test("the retired sweep fields migrate onto a lane ONLY when the lane is absent", () => {
+ const migrated = withFile(
+ JSON.stringify({ digest: { sweepEnabled: true, sweepChannels: ["a"] } }),
+ );
+ assert.equal(migrated.autoQueue.digest.enabled, true);
+ // Absence was the trigger, not the default: a file that spells the lane
+ // keeps its own answer even though the sweep flag says otherwise.
+ const spelled = withFile(
+ JSON.stringify({
+ digest: { sweepEnabled: true },
+ autoQueue: { digest: { enabled: false } },
+ }),
+ );
+ assert.equal(spelled.autoQueue.digest.enabled, false);
+ // No sweep flag, no lane: the migration never enables anything.
+ assert.equal(withFile("{}").autoQueue.digest.enabled, false);
+ assert.equal(withFile("{}").autoQueue.backfill.enabled, false);
+});
diff --git a/common/lib/settingsSchema.ts b/common/lib/settingsSchema.ts
@@ -0,0 +1,1425 @@
+// THE SETTINGS SCHEMA — one definition of settings.json, used by the reader,
+// the writer, the example file and the key table.
+//
+// one-core phase 3 slice 4a. What used to be three copies of the same list —
+// the `SiteSettings` type, the `defaults()` literal, and the two field-by-field
+// sanitizing literals in getSettings and writeSettings — is `siteSettingsSchema`
+// below. `SiteSettings` is inferred from it; `defaults()` is `parse({})`;
+// getSettings and writeSettings (lib/settings.ts) both parse through it; and
+// `common/bin/settings-example.ts` generates settings.json.example and SETTINGS.md
+// from it, including each field's `.describe()` text — which is where the
+// comments that used to sit on the `SiteSettings` type now live, so they have one
+// home and cannot drift from the key they describe.
+//
+// ZOD SUPPLIES THE PLUMBING, NOT THE ARITHMETIC. Every field is
+// `settingsField(coerce)` — `z.unknown().catch(undefined).transform(coerce)` —
+// and every `coerce` is the clamp or sanitizer that already existed, reused, so
+// no boundary moved. Unknown keys are dropped by zod's default strip (never
+// `.passthrough()`), which is what retires a field: a key the schema does not
+// name cannot survive a read or a save.
+//
+// SERVER-ONLY IN PRACTICE: it imports node:path (sanitizeStorage) and is reached
+// through lib/settings.ts, which imports node:fs at module scope. A `"use
+// client"` form that needs a constant imports the constant — never this schema.
+//
+// NOT HERE: the three migrations keyed on a field's ABSENCE in the raw file
+// (sweeps → lanes, mediaRoot → locations, legacy transcribe* → app registry and
+// synthesized workers), because a parsed object cannot tell "absent" from
+// "default". They run in getSettings around the parse. See lib/settings.ts.
+
+import path from "node:path";
+import { z } from "zod";
+// The auto-queue, workers and channel-priority blocks keep their own
+// sanitizers in their own homes; these are their zod seams.
+import {
+ autoQueueSchema,
+ channelPrioritySchema,
+ settingsField,
+ workersSchema,
+} from "./settingsFieldSchemas";
+import { SYNC_INTERVAL_MAX_MINUTES } from "./channelConfig";
+import {
+ isDownloadFormatPreset,
+ type DownloadFormatPreset,
+} from "../ytdlp/downloadFormat";
+import {
+ type AppInstanceConfig,
+ DEFAULT_TRANSCRIPTION_APP_ID,
+ TRANSCRIPTION_APPS,
+} from "./transcriptionApps";
+import { sanitizeWorkerConfig } from "./workers";
+import {
+ INTERNAL_LOCATION_ID,
+ type StorageLocation,
+ type StorageSettings,
+ type StorageVolume,
+} from "./storageLocations";
+import {
+ DEFAULT_DIARIZATION_ENGINE,
+ DEFAULT_DIARIZATION_THRESHOLD,
+ DIARIZATION_BACKENDS,
+ DIARIZATION_ENGINE_IDS,
+ type DiarizationBackend,
+ type DiarizationEngineId,
+} from "./diarization";
+import { ATTRIBUTION_PROMPT_VERSION } from "./attribution";
+import {
+ DEFAULT_COOKIE_MODE,
+ isCookieMode,
+ type CookieMode,
+} from "./cookiePolicy";
+// From the CLIENT-SAFE digest module, deliberately — digestApps.ts imports execa,
+// and settings.ts must stay reachable from anywhere.
+import {
+ CLAUDE_DIGEST_APP_ID,
+ DEFAULT_DIGEST_APP_ID,
+ DEFAULT_DIGEST_TIMESTAMP_MODE,
+ DIGEST_SECTION_KINDS,
+ DIGEST_TIMESTAMP_MODES,
+ isDigestSectionKind,
+ isDigestTimestampMode,
+ type DigestAppConfig,
+ type DigestSectionKind,
+ type DigestTimestampMode,
+} from "./digest";
+
+export type { Worker } from "./workers";
+export type { AutoQueueSettings } from "./autoQueueTypes";
+export type { ChannelPriority } from "./channelPriority";
+
+// Transcribe placeholder/arg helpers now live with the whisper-cpp app in
+// transcriptionApps.ts. Re-exported here so existing import sites keep working.
+export {
+ type AppInstanceConfig,
+ TRANSCRIBE_PLACEHOLDER_AUDIO,
+ TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE,
+ TRANSCRIBE_PLACEHOLDER_MODEL,
+ TRANSCRIBE_KNOWN_PLACEHOLDERS,
+ DEFAULT_TRANSCRIBE_ARGS,
+ validateTranscribeArgs,
+} from "./transcriptionApps";
+
+// Configuration for speaker attribution — putting names to the speaker turns.
+//
+// OFF by default, and that default is doing real work rather than being
+// cautious. The text-only lane costs roughly one model call per transcript
+// CHUNK, which on this corpus is ~194,000 calls, the same order as the digest
+// sweep — and the digest sweep has completed 0.17% of its own. Arming both at
+// once puts two multi-week lanes on the same 8 GB card with nothing to arbitrate
+// between them (the backfill lane's yield deliberately watches only the
+// transcription lane). Nothing here arms anything; a pilot decides whether the
+// corpus-wide text-only pass is worth 25-55 GPU-days at all.
+export type AttributionSettings = {
+ // Master switch. Off means the backfill registry reports no attribution work
+ // at all — the feature gate every Operation has.
+ enabled: boolean;
+ // Which digest app runs the naming. Attribution IS a digest-app workload —
+ // constrained JSON decoding over transcript text — so it reuses that registry
+ // and that per-app config (settings.digest.apps[appId]) rather than growing a
+ // second copy of the ollama URL, context size and timeout.
+ appId: string;
+ // Model override. Empty = the app's configured model, then its default. It is
+ // separate from the digest's because the two workloads may want different
+ // sizes, and because it is part of the freshness identity: sharing the digest's
+ // model field would make a digest bake-off invalidate every attribution record
+ // on disk as a side effect.
+ model: string;
+ // The lanes, separately. Both default OFF even when `enabled` is on, so
+ // turning the feature on to look at it cannot start a corpus sweep.
+ //
+ // They are not a fallback pair. `diarized` is one call per video and grounded
+ // in acoustic clustering; `textOnly` is ~30 calls and guesses at identity
+ // across chunk seams. An operator may reasonably want the first forever and
+ // the second never.
+ diarizedEnabled: boolean;
+ textOnlyEnabled: boolean;
+ // The prompt generation a record must match to count as fresh.
+ //
+ // Defaults to (and is floored at) ATTRIBUTION_PROMPT_VERSION, the shipped
+ // constant. Raising it forces a corpus-wide regeneration without a code
+ // change, which is the honest way to redo everything after a prompt tweak.
+ // It cannot be set BELOW the shipped constant, and that floor is the lesson
+ // from digestPrompt.ts's version 1 -> 2 note: pinning freshness to an older
+ // generation freezes output from a superseded prompt into the corpus, looking
+ // identical to output from the current one.
+ promptVersion: number;
+};
+
+// Configuration for the backfill lane — the generic answer to "a derived-data
+// feature landed and 77,000 existing videos do not have it".
+//
+// WHAT THE LANE GETS is its own queueKey (concurrency with transcription) plus a
+// limit() that returns 0 to stand aside — the same mechanism the digest yield
+// uses, which fails OPEN so a bad read costs contention rather than a deadlock.
+// There is no priority system to join: the registry submits every named queue at
+// concurrency 1 and SchedulerTier only orders work within a single key.
+//
+// THE SHARE IS `concurrency` AND THE LANE'S `autoQueue.backfill.maxWorkers`; the
+// yield is the operation's declared `contendsFor`. Slice 1.3 retired the
+// `weight` scalar that used to mean both — see backfillLimit().
+export type BackfillSettings = {
+ // Slots the lane may use when it is not standing aside. Kept at 1 by default
+ // for the same reason diarization.concurrency is: this is CPU-bound work
+ // competing with GPU feeding and the digest sweep for the same 8 threads.
+ concurrency: number;
+ // Re-acquire media for videos whose input is GONE (audio deleted after
+ // transcription). OFF by default and deliberately so: measured on this corpus,
+ // 836 videos still have media and ~76,270 would need a re-download — 91x the
+ // reachable work, against 45 GB free at 97% full. When on, each re-fetched
+ // file is removed in a `finally` as soon as the backfill has used it, unless
+ // the video is marked do-not-clean, or unless the auto-transcribe policy would
+ // replace its auto-captions (`replaceAutoSubs`, or a leaf on
+ // `downloadedAutoSubsOnly`), in which case the audio is kept for that runner.
+ //
+ // WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: "youtube"`
+ // channel — which normally only fetches subtitles — the re-acquire applies a
+ // PER-VIDEO transcribe override so yt-dlp lands audio a diarizer can read;
+ // the channel's stored config is not changed. Without that override the fetch
+ // re-downloads the captions the video already has and lands nothing, which is
+ // what happened to ~16,000 videos on eight channels in 2026-08.
+ allowRedownload: boolean;
+};
+
+// Configuration for the speaker-diarization capture lane.
+//
+// This exists because AUDIO IS THE ONE PERISHABLE INPUT in the pipeline:
+// cleanAudioFromTranscribed deletes it once a video is transcribed, so
+// diarization has to happen while the audio is still there or not at all. The
+// capture half is deliberately all that ships here — attribution, LLM speaker
+// naming, viewer badges and quote filtering can all be redone later from the
+// saved JSON, whereas the audio cannot.
+export type DiarizationSettings = {
+ // Master switch. OFF by default so a transcription batch can start before this
+ // lands, with diarization backfilled over the retained audio afterwards.
+ //
+ // Turning it ON also arms the cleanup guard: the Clean-audio sweep stops
+ // deleting audio for a transcribed video that has no diarization.json yet.
+ // That is the point — it is what keeps the perishable input alive long enough
+ // to be captured — but it means enabling this holds disk.
+ enabled: boolean;
+ // Run diarization inline in the post-transcribe hook.
+ //
+ // OFF by default, and that default is a MEASURED decision, not caution.
+ // Measured on this box: GPU transcription runs at 221 s/audio-hour (16.3x
+ // realtime, over 3,602 real videos), CPU diarization at ~500-680 s/audio-hour.
+ // Diarization is therefore ~2-3x SLOWER than the transcription it follows, so
+ // running it inline drops whole-pipeline throughput by roughly 3-4x and leaves
+ // the GPU idle while the CPU catches up.
+ //
+ // The intended sequence for a large batch is the opposite: leave this off, let
+ // the batch transcribe at full GPU speed with `enabled` holding the audio, and
+ // diarize afterwards with the backfill pass. Turn it on for steady state, once
+ // the arrival rate is a few videos a day rather than a corpus.
+ inlineAfterTranscribe: boolean;
+ // Clustering threshold — the single most consequential knob, since it decides
+ // how many speakers come out. Larger merges more aggressively.
+ //
+ // The default is 0.9, NOT sherpa-onnx's own 0.5, and that is measured on this
+ // corpus. On a 6-minute excerpt of a two-person interview (known ground truth:
+ // 2 speakers), sherpa's default produced 22 clusters; 0.9 produced 6, with the
+ // top two at 40%/40% of talk time — recognizably the two hosts. Sweep on the
+ // same clip: 0.4→23, 0.5→22, 0.6→17, 0.7→12, 0.8→10, 0.9→6.
+ //
+ // It still over-splits, which is why this is a capture lane and not an answer:
+ // the turns are recorded with the threshold that produced them, so a later
+ // attribution pass can re-cluster or re-run without needing the audio back.
+ threshold: number;
+ // Engine threads per diarize run.
+ threads: number;
+ // Which engine runs. "sherpa-onnx" is the shipped default and what every
+ // sidecar on disk was produced by; "sortformer" is the ggml engine built by
+ // scripts/build-sortformer.sh.
+ //
+ // CHANGING THIS RESTATES THE FRESHNESS IDENTITY (see diarizationTarget), so
+ // every sidecar written by the other engine becomes stale and the backfill lane
+ // offers to redo it. That is intended — the two disagree about how many
+ // speakers exist, and a corpus half-diarized by each is not one corpus — but on
+ // the retained audio it is weeks of work, not a toggle.
+ //
+ // Why anyone would: on the same file, sherpa at its tuned threshold returns 13
+ // speakers and sortformer returns 4, agreeing on the dominant speaker's share
+ // to within half a point (73.1% vs 73.5%). On the corpus's worst case sherpa
+ // returns 35 and sortformer 4. Over-splitting is the failure mode this lane has
+ // always had, and sortformer is end-to-end rather than clustered, so it does
+ // not have it. The cost is a hard ceiling of 4 speakers and ~1.8x the wall
+ // clock.
+ engine: DiarizationEngineId;
+ // Compute device for the sortformer engine; ignored by sherpa-onnx, which has
+ // no Vulkan compute path on Linux.
+ //
+ // "vulkan" is 1.5x faster than a thread-tuned CPU run (894 vs 1305
+ // s/audio-hour, measured on this box) and holds 558 MB resident instead of
+ // 4.84 GB by keeping weights and activations in VRAM. It also takes ~4.4 GB of
+ // an 8 GB card, which is why the lane YIELDS to transcription rather than
+ // sharing — see controller/digestYield.ts.
+ backend: DiarizationBackend;
+ // Python interpreter for the default sherpa-onnx engine. sherpa-onnx ships
+ // wheels only up to cp313, and this box's system python is 3.14 — so this
+ // usually points at a dedicated venv rather than `python3`.
+ python: string;
+ // ONNX model paths for the default engine. Empty = the lane cannot run, which
+ // is reported as a skip rather than a failure.
+ segModel: string;
+ embModel: string;
+ // Binary and model for the sortformer engine, both produced by
+ // scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the
+ // same "not-configured" skip as an unset segModel/embModel.
+ sortformerBin: string;
+ sortformerModel: string;
+ // How many diarize runs may execute at once in the backfill pass. Kept low by
+ // default: diarization is CPU-bound and competes with GPU feeding and the
+ // digest sweep for the same 8 threads.
+ concurrency: number;
+ // Videos longer than this are DEFERRED rather than diarized: reported as a
+ // third number that is never summed into reachable work, so a capped corpus
+ // can never read as finished.
+ //
+ // THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering holds a
+ // pairwise distance matrix over speech-segment embeddings — O(n^2) in SEGMENT
+ // count — and speaker-turn density varies 40x across this corpus (33-1364
+ // turns/hour), so duration does not actually predict the blowup: a sparse
+ // 7h42m video completed while a dense 6h12m one was OOM-killed. Duration is
+ // merely the only predictor available for free, from metadata already on disk,
+ // BEFORE spending 45 minutes to find out. n^2 at 30k segments is 6.7 GiB and
+ // at 40k is 11.9 GiB, which brackets the 10.6 GB and 9.6 GB peaks measured on
+ // this 16 GB box.
+ //
+ // 0 disables the cap. That is where this goes once windowed diarization lands:
+ // windowing divides per-window n by the window count, so the matrix falls by
+ // its square, and the cap stops being needed rather than being tuned.
+ maxAudioHours: number;
+};
+
+// Configuration for the derived-corpus digest layer. Local-first by decision:
+// `remoteEnabled` gates the metered lane and defaults to false, so nothing here
+// can spend money until it is explicitly turned on.
+export type DigestSettings = {
+ // Master switch for the metered (remote-api) lane. OFF by default — an opt-in
+ // overflow for the long tail or a channel where local quality is poor, never
+ // the default path.
+ remoteEnabled: boolean;
+ // Videos longer than this are "long tail": 8.2% of the corpus by count, 46% of
+ // all transcript tokens. The batch's duration-aware ordering and the optional
+ // remote overflow both key off it.
+ longTailSeconds: number;
+ // The engine each lane uses (ids from common/lib/digestApps.ts).
+ localAppId: string;
+ remoteAppId: string;
+ // Per-app config, keyed by app id — the same id-keyed sub-record shape as
+ // transcriptionApps.
+ apps: Record<string, DigestAppConfig>;
+ // Yield the GPU to the transcription lane: while transcription is working, the
+ // digest batch's limit() returns 0 and the pool idle-waits. ON by default,
+ // because `digest:local` is deliberately on a different queue from
+ // TRANSCRIPTION_QUEUE and so would otherwise run ollama and the transcription
+ // engine on the same 8 GB card. See controller/digestYield.ts.
+ yieldToTranscription: boolean;
+ // Whether a busy worker pinned to `device: "cpu"` counts as GPU contention.
+ //
+ // OFF by default, which is the FIX for a real bug: the yield originally tested
+ // only `kind === "local"`, so on a box with one GPU worker and two CPU-pinned
+ // ones (this box, at parallelTranscriptions 2) the digest lane stopped dead for
+ // transcription that competes for zero GPU shaders.
+ //
+ // Only an EXPLICIT "cpu" is treated as non-contending. A worker with no device
+ // set is using the engine binary's own default, which may be the GPU, so it
+ // still triggers the yield — the unknown case fails safe.
+ //
+ // Composes with `yieldToTranscription`: that is the master switch, this only
+ // narrows which workers it reacts to.
+ yieldToCpuWorkers: boolean;
+ // Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. Only ever
+ // consulted for a metered app.
+ spendCapUsd: number;
+ // Which sections a sweep generates.
+ //
+ // Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, and that
+ // is not a contradiction: a tag call sends the same transcript as the chapter
+ // call before it, so it hits the engine's cached prefix and pays essentially no
+ // prefill (+0.4 s across 4 extra calls, against 22.4 s for the first 4). All it
+ // pays is decode, and a tag list is ~30 output tokens where a chapter list is
+ // ~200–290.
+ //
+ // The corollary matters more than the number: run them in the SAME pass. Tags
+ // generated later, on their own, pay full prefill again — measured at 44% of a
+ // whole chapters pass, i.e. 3–9× the marginal cost of just including them now.
+ sections: DigestSectionKind[];
+ // How each chunk's transcript markers are numbered — see DigestTimestampMode.
+ // Was a scored variable in the bake-off rather than a pre-applied fix; the
+ // measurement is in and "chunk-local" is now the shipped default.
+ timestampMode: DigestTimestampMode;
+ // A free-text label for a non-default prompt shape, folded into the recorded
+ // provenance by digestPromptVariant(). Setting it invalidates every digest
+ // generated under a different label, which is exactly what makes a bake-off
+ // round re-run its sample instead of skipping it as fresh. Empty = default.
+ promptVariant: string;
+};
+
+// "basic" — `pnpm run build` in export/, serialized on the build queue (shared
+// output tree → no safe parallelism).
+// "docker" — isolated per-site container builds (follow-up); enables real
+// parallel multi-site builds capped by maxParallelBuilds.
+export type BuildMode = "basic" | "docker";
+
+export type BuildPipelineSettings = {
+ mode: BuildMode;
+ // Cap on concurrent per-site container builds in docker mode. Ignored in basic
+ // mode (which is always serial). Clamped to [1, BUILD_MAX_PARALLEL_MAX].
+ maxParallelBuilds: number;
+ // Tag of the reusable build image (built once, reused for every site).
+ dockerImage: string;
+ // Dockerfile path relative to the monorepo root, used to (re)build the image.
+ dockerfile: string;
+};
+
+export type SavedVideoBackupSettings = {
+ // Master switch for the scheduled backup. A backup can still be run manually
+ // when this is false, as long as a destination is set.
+ enabled: boolean;
+ // Destination root the store is mirrored into (a local path or any rsync
+ // target). Empty disables both scheduled and manual backups.
+ dest: string;
+ // Cadence (minutes) for the scheduled backup when enabled. Clamped into the
+ // sync-interval window; default daily.
+ intervalMinutes: number;
+};
+
+export type SyncSchedulerSettings = {
+ // Master switch. When false, a tick selects nothing (manual sync still works).
+ enabled: boolean;
+ // Fallback cadence (minutes) for channels with no per-channel override.
+ defaultIntervalMinutes: number;
+ // Cap on sync jobs running/queued at once. A tick queues at most
+ // (cap - currently-active) channels; the rest roll to the next tick. This is
+ // also the stagger mechanism that keeps a big due-batch from hitting the
+ // source all at once.
+ maxConcurrentSyncs: number;
+ // Optional local-clock quiet window during which auto-sync is suppressed.
+ // Both null = always allowed. The window may wrap past midnight
+ // (e.g. start=22, end=6). Hours are [0,23]; the window is [start, end).
+ quietHoursStart: number | null;
+ quietHoursEnd: number | null;
+ // Failure backoff bounds. After N consecutive failed scheduled syncs a
+ // channel waits min(base * 2^(N-1), max) minutes before it's eligible again.
+ backoffBaseMinutes: number;
+ backoffMaxMinutes: number;
+ // Cadence (seconds) for the editor's in-process heartbeat — the internal timer
+ // armed by the instrumentation hook (editor/instrumentation.ts) that calls the
+ // scheduler tick directly, so no external cron is needed. 0 = off: rely on the
+ // external `pnpm sync:tick` heartbeat instead. Any positive value is clamped to
+ // [SYNC_HEARTBEAT_MIN_SECONDS, SYNC_HEARTBEAT_MAX_SECONDS]. The env var
+ // SYNC_HEARTBEAT_SECONDS overrides this at runtime. See SCHEDULED_SYNC.md.
+ heartbeatSeconds: number;
+ // Cadence (minutes) for the scheduled keep-latest deletion check. For each
+ // channel with ChannelConfig.keepLatest > 0, the tick re-probes the kept
+ // window for source deletion (checkKeptDeletedAction) at most this often and
+ // pins any gone videos as do-not-clean. Clamped into the sync-interval window;
+ // default daily. The check shares the same concurrency cap and quiet-hours
+ // window as scheduled syncs. See editor/app/scheduler/runTick.ts.
+ keepLatestCheckIntervalMinutes: number;
+ // Default cadence (minutes) for the sync FULL SWEEP — the deep pass that
+ // re-enumerates a channel's whole listing in one yt-dlp spawn, refreshes the
+ // stored `playlist` file, and flags videos that have left the listing into
+ // maybe-missing.json. Ordinary syncs stay on the cheap newest-first paged
+ // walk; a sync only upgrades itself to a sweep when this interval has elapsed
+ // since the channel's lastFullSweepAt. Per-channel override:
+ // ChannelConfig.fullSweepIntervalMinutes. 0 = never sweep. Default daily.
+ // See common/jobs/deepSync.ts.
+ fullSweepIntervalMinutes: number;
+ // Upper bound on how many maybe-missing suspects a full sweep will resolve
+ // in-line with the per-video availability probe (deleted vs private vs
+ // unlisted). At or under the cap the sweep runs the targeted check itself, so
+ // "Sync all" surfaces upstream deletions with no extra clicks; over it, the
+ // suspects are flagged and left for a manual check rather than firing hundreds
+ // of probes inside a sync. 0 = never auto-confirm.
+ fullSweepConfirmMaxSuspects: number;
+ // Shrink guard: how far a fresh listing may fall below the stored one before
+ // it is treated as suspect rather than acted on. Expressed as a percentage of
+ // the previous count, floored at SHRINK_ABS_FLOOR entries so ordinary churn on
+ // a small channel doesn't trip it. A suspect listing does not rewrite
+ // `playlist` or maybe-missing.json and does not count as a sweep — but a
+ // SECOND enumeration reporting a similar count confirms it and is accepted, so
+ // a genuine mass deletion costs at most one cadence period. 0 = off (the
+ // empty-listing rejection still applies). See controller/acceptListing.ts.
+ fullSweepShrinkGuardPercent: number;
+};
+
+export type SocialLink = {
+ label: string;
+ url: string;
+ svg: string;
+};
+
+export const SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS = 600;
+export const SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS = 10;
+
+export const MIN_FREE_DISK_GB_DEFAULT = 5;
+export const MIN_FREE_DISK_GB_MAX = 100000;
+
+// Hysteresis margin for the low-disk gate. 2 GB is deliberately larger than any
+// single scratch file the pipeline writes, so cleaning one up cannot by itself
+// reopen the gate.
+export const RESUME_MARGIN_GB_DEFAULT = 2;
+export const RESUME_MARGIN_GB_MAX = 1000;
+
+export const PARALLEL_TRANSCRIPTIONS_MAX = 16;
+export const PARALLEL_TRANSCRIPTIONS_DEFAULT = 2;
+
+// Passive UI auto-refresh cadence. 0 is a sentinel meaning "disabled"; any other
+// value is clamped into [MIN, MAX] seconds.
+export const AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS = 5;
+export const AUTO_REFRESH_INTERVAL_MIN_SECONDS = 1;
+export const AUTO_REFRESH_INTERVAL_MAX_SECONDS = 600;
+
+// Global snapshot-scheduler debounce presets. `debounceMs` is the quiet-period
+// window after the last report-changing action; `maxWaitMs` caps the total
+// delay under continuous activity (null = no cap, fire purely on the quiet
+// period). Consumed by common/jobs/snapshotScheduler.ts and surfaced in the
+// Settings form.
+export type ReportDebouncePreset = "fast" | "balanced" | "lazy";
+
+export const REPORT_DEBOUNCE_PRESETS: Record<
+ ReportDebouncePreset,
+ { debounceMs: number; maxWaitMs: number | null }
+> = {
+ fast: { debounceMs: 1000, maxWaitMs: null },
+ balanced: { debounceMs: 3000, maxWaitMs: 30000 },
+ lazy: { debounceMs: 10000, maxWaitMs: 60000 },
+};
+
+export const DEFAULT_REPORT_DEBOUNCE_PRESET: ReportDebouncePreset = "fast";
+
+export function isReportDebouncePreset(v: unknown): v is ReportDebouncePreset {
+ return v === "fast" || v === "balanced" || v === "lazy";
+}
+
+export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024;
+export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024;
+export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024;
+
+export const DEFAULT_ADMIN_TITLE = "Transcript Browser Admin";
+
+// Sync-scheduler bounds + defaults. Default cadence is daily; concurrency is
+// conservative so a tick doesn't fan out into the source provider all at once.
+export const SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES = 1440;
+export const SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT = 2;
+export const SYNC_SCHEDULER_MAX_CONCURRENT_MAX = 16;
+export const SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES = 30;
+export const SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES = 1440;
+export const KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES = 1440;
+// Full-sweep defaults. Daily: a sweep is one full enumeration of the channel,
+// far more expensive than the 50-entry page an ordinary sync fetches. The
+// confirm cap keeps an unattended sweep from fanning out into hundreds of
+// per-video probes when a channel's listing changes wholesale.
+export const FULL_SWEEP_DEFAULT_INTERVAL_MINUTES = 1440;
+export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT = 25;
+export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX = 10000;
+// Shrink-guard default: a listing that has lost more than a tenth of its
+// entries (and more than SHRINK_ABS_FLOOR of them) needs a second opinion.
+export const FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT = 10;
+export const FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX = 100;
+export const SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES = 1440;
+
+// Internal-heartbeat cadence bounds. 0 means "off" (use an external cron
+// heartbeat); any other value is clamped into [MIN, MAX] seconds. The floor
+// keeps the in-process timer from busy-looping; the ceiling is one hour.
+export const SYNC_HEARTBEAT_DEFAULT_SECONDS = 0;
+export const SYNC_HEARTBEAT_MIN_SECONDS = 15;
+export const SYNC_HEARTBEAT_MAX_SECONDS = 3600;
+
+export function defaultSyncScheduler(): SyncSchedulerSettings {
+ return {
+ enabled: false,
+ defaultIntervalMinutes: SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES,
+ maxConcurrentSyncs: SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT,
+ quietHoursStart: null,
+ quietHoursEnd: null,
+ backoffBaseMinutes: SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES,
+ backoffMaxMinutes: SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES,
+ heartbeatSeconds: SYNC_HEARTBEAT_DEFAULT_SECONDS,
+ keepLatestCheckIntervalMinutes: KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES,
+ fullSweepIntervalMinutes: FULL_SWEEP_DEFAULT_INTERVAL_MINUTES,
+ fullSweepConfirmMaxSuspects: FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT,
+ fullSweepShrinkGuardPercent: FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT,
+ };
+}
+
+// Clamp an internal-heartbeat cadence: 0 (off) passes through; any positive
+// value is clamped up into [MIN, MAX]; junk falls back to the default.
+export function clampHeartbeatSeconds(value: unknown): number {
+ if (typeof value !== "number" || !Number.isFinite(value)) {
+ return SYNC_HEARTBEAT_DEFAULT_SECONDS;
+ }
+ const n = Math.floor(value);
+ if (n <= 0) return 0;
+ if (n < SYNC_HEARTBEAT_MIN_SECONDS) return SYNC_HEARTBEAT_MIN_SECONDS;
+ if (n > SYNC_HEARTBEAT_MAX_SECONDS) return SYNC_HEARTBEAT_MAX_SECONDS;
+ return n;
+}
+
+function clampHourOrNull(value: unknown): number | null {
+ if (typeof value !== "number" || !Number.isFinite(value)) return null;
+ const n = Math.floor(value);
+ if (n < 0 || n > 23) return null;
+ return n;
+}
+
+// Like clampPositiveInt, but 0 survives as a sentinel ("off"/"never"). Used by
+// the cadences whose disabled state is expressed as a zero rather than a
+// separate boolean.
+function clampIntAllowZero(value: unknown, fallback: number, max: number): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? Math.floor(value)
+ : fallback;
+ if (n <= 0) return 0;
+ if (n > max) return max;
+ return n;
+}
+
+function clampPositiveInt(value: unknown, fallback: number, max: number): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? Math.floor(value)
+ : fallback;
+ if (n < 1) return 1;
+ if (n > max) return max;
+ return n;
+}
+
+// Coerce a raw settings.syncScheduler value into a clean SyncSchedulerSettings,
+// falling back to defaults for missing/ill-typed fields. Quiet hours are only
+// honored when BOTH endpoints are valid hours; otherwise the window is cleared.
+export function sanitizeSyncScheduler(value: unknown): SyncSchedulerSettings {
+ const d = defaultSyncScheduler();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const start = clampHourOrNull(r.quietHoursStart);
+ const end = clampHourOrNull(r.quietHoursEnd);
+ const backoffBase = clampPositiveInt(
+ r.backoffBaseMinutes,
+ d.backoffBaseMinutes,
+ SYNC_INTERVAL_MAX_MINUTES,
+ );
+ return {
+ enabled: r.enabled === true,
+ defaultIntervalMinutes: clampPositiveInt(
+ r.defaultIntervalMinutes,
+ d.defaultIntervalMinutes,
+ SYNC_INTERVAL_MAX_MINUTES,
+ ),
+ maxConcurrentSyncs: clampPositiveInt(
+ r.maxConcurrentSyncs,
+ d.maxConcurrentSyncs,
+ SYNC_SCHEDULER_MAX_CONCURRENT_MAX,
+ ),
+ quietHoursStart: start !== null && end !== null ? start : null,
+ quietHoursEnd: start !== null && end !== null ? end : null,
+ backoffBaseMinutes: backoffBase,
+ // Cap can't sit below the base, or backoff would never grow.
+ backoffMaxMinutes: Math.max(
+ backoffBase,
+ clampPositiveInt(
+ r.backoffMaxMinutes,
+ d.backoffMaxMinutes,
+ SYNC_INTERVAL_MAX_MINUTES,
+ ),
+ ),
+ heartbeatSeconds: clampHeartbeatSeconds(r.heartbeatSeconds),
+ keepLatestCheckIntervalMinutes: clampPositiveInt(
+ r.keepLatestCheckIntervalMinutes,
+ d.keepLatestCheckIntervalMinutes,
+ SYNC_INTERVAL_MAX_MINUTES,
+ ),
+ fullSweepIntervalMinutes: clampIntAllowZero(
+ r.fullSweepIntervalMinutes,
+ d.fullSweepIntervalMinutes,
+ SYNC_INTERVAL_MAX_MINUTES,
+ ),
+ fullSweepConfirmMaxSuspects: clampIntAllowZero(
+ r.fullSweepConfirmMaxSuspects,
+ d.fullSweepConfirmMaxSuspects,
+ FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX,
+ ),
+ fullSweepShrinkGuardPercent: clampIntAllowZero(
+ r.fullSweepShrinkGuardPercent,
+ d.fullSweepShrinkGuardPercent,
+ FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX,
+ ),
+ };
+}
+
+export function defaultSavedVideoBackup(): SavedVideoBackupSettings {
+ return {
+ enabled: false,
+ dest: "",
+ intervalMinutes: SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES,
+ };
+}
+
+// Coerce a raw settings.savedVideoBackup value into a clean
+// SavedVideoBackupSettings. A missing destination forces enabled off, since a
+// backup with nowhere to go is meaningless.
+export function sanitizeSavedVideoBackup(
+ value: unknown,
+): SavedVideoBackupSettings {
+ const d = defaultSavedVideoBackup();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const dest = typeof r.dest === "string" ? r.dest.trim() : "";
+ return {
+ enabled: dest !== "" && r.enabled === true,
+ dest,
+ intervalMinutes: clampPositiveInt(
+ r.intervalMinutes,
+ d.intervalMinutes,
+ SYNC_INTERVAL_MAX_MINUTES,
+ ),
+ };
+}
+
+// Where relocated channel media goes: the named locations.
+//
+// This used to be ONE FIELD, `mediaRoot` — a single absolute string, the cold
+// drive, typed once. It grew into a list of entities because a root alone
+// cannot answer the two questions the operator actually has: is that disk here,
+// and if it came up somewhere else, how do I point the channels at it without
+// ssh and hand edits? A location carries an id, a label, the root, an opt-in
+// `autoRepoint`, and the volume identity learned at its last probe.
+//
+// Still NOT a policy: a channel on a location is not thereby deprioritized, and
+// nothing auto-relocates anything because a location exists.
+//
+// AVAILABILITY IS NEVER STORED HERE. A refresh that wrote "available" would
+// rewrite settings.json — and so bump the pulse revision — every few seconds.
+// The probe (common/lib/storageVolumes.ts) is computed per request; only the
+// `volume` identity is ever written back, and only when it changed.
+//
+// The types live in lib/storageLocations.ts, which is pure: a `"use client"`
+// file may import them, and must not reach storageVolumes.ts (execa).
+export type { StorageLocation, StorageVolume, StorageSettings };
+
+export function defaultStorage(): StorageSettings {
+ return { locations: [], defaultLocationId: "" };
+}
+
+const LOCATION_ID_RE = /^[a-z0-9][a-z0-9-]{0,63}$/;
+
+// "internal" IS TAKEN. It is the synthetic /storage row for the corpus volume
+// (INTERNAL_LOCATION_ID), and the regex above admits it — so a hand-edited
+// settings.json, or an operator typing the obvious word into the New location
+// form, could store a real location under the one id the page assembles for
+// itself. The row would then be built twice, the rollup would count channels
+// into whichever assembled last, and `locationOfDataDir` would start matching
+// unrelocated channels against it.
+function isReservedLocationId(id: string): boolean {
+ return id === INTERNAL_LOCATION_ID;
+}
+
+function sanitizeVolume(value: unknown): StorageVolume | undefined {
+ if (!value || typeof value !== "object") return undefined;
+ const v = value as Record<string, unknown>;
+ const uuid = typeof v.uuid === "string" ? v.uuid.trim() : "";
+ const mountpoint =
+ typeof v.mountpoint === "string" ? v.mountpoint.trim() : "";
+ // No uuid is no identity, and no mountpoint means `root === join(mountpoint,
+ // relPath)` cannot hold — either way the record is not usable for finding the
+ // volume again, so it is dropped rather than half-kept.
+ if (!uuid || !mountpoint) return undefined;
+ const relPath = typeof v.relPath === "string" ? v.relPath.trim() : "";
+ const fstype = typeof v.fstype === "string" ? v.fstype.trim() : "";
+ const label = typeof v.label === "string" ? v.label.trim() : "";
+ return {
+ uuid,
+ ...(fstype ? { fstype } : {}),
+ ...(label ? { label } : {}),
+ mountpoint,
+ relPath,
+ };
+}
+
+// Coerce a raw settings.storage value into a clean StorageSettings.
+//
+// EXISTENCE IS NOT CHECKED, on purpose: the whole point of a cold location is
+// that it is a drive that may not be mounted when settings are read, and a
+// sanitizer that dropped the root on an unmounted platter would silently erase
+// the operator's choice on the next save.
+//
+// ABSOLUTENESS *IS* checked, and a location with a relative root is DROPPED
+// rather than resolved. Resolving it would anchor the location to whatever cwd
+// the reader booted in — a different directory under docker, under a worktree,
+// and under `pnpm dev` — so the same settings.json would name three different
+// drives. The location form rejects a relative path with a message before it
+// ever gets here; this is the last line, not the only one.
+//
+// NESTED ROOTS ARE ALLOWED. "/mnt/platter" and "/mnt/platter/archive" may both
+// be locations; `locationOfDataDir` resolves a channel to the LONGEST matching
+// root. Nothing here rejects the nesting, because the operator who arranges a
+// disk that way means it.
+//
+// A STALE `mediaRoot` SITTING BESIDE `locations` IS IGNORED — it is not merged
+// back in as an extra location. `migrateMediaRootToLocations` reads it exactly
+// once, when `locations` is absent; after that the list is the whole truth, and
+// resurrecting a root the operator deleted would be a bug, not a kindness.
+//
+// ROLLBACK: an older binary sanitizes this block to `{ mediaRoot: "" }` — the
+// locations are dropped and the single cold root comes back blank. One string
+// lost, nothing on disk moved. `cp settings.json settings.json.pre-storage-
+// locations` before the upgrade and a downgrade is a file copy.
+export function sanitizeStorage(value: unknown): StorageSettings {
+ const d = defaultStorage();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const rawList = Array.isArray(r.locations) ? r.locations : [];
+ const locations: StorageLocation[] = [];
+ const seen = new Set<string>();
+ for (const entry of rawList) {
+ if (!entry || typeof entry !== "object") continue;
+ const e = entry as Record<string, unknown>;
+ const id = typeof e.id === "string" ? e.id.trim() : "";
+ if (!LOCATION_ID_RE.test(id) || isReservedLocationId(id) || seen.has(id)) {
+ continue;
+ }
+ const rawRoot = typeof e.root === "string" ? e.root.trim() : "";
+ if (!path.isAbsolute(rawRoot)) continue;
+ // "/mnt/platter/" and "/mnt/platter" are one root; "/" stays "/".
+ const stripped = rawRoot.replace(/\/+$/, "");
+ const root = stripped === "" ? "/" : stripped;
+ const label = typeof e.label === "string" ? e.label.trim() : "";
+ const volume = sanitizeVolume(e.volume);
+ seen.add(id);
+ locations.push({
+ id,
+ label: label || id,
+ root,
+ autoRepoint: e.autoRepoint === true,
+ ...(volume ? { volume } : {}),
+ });
+ }
+ const wanted =
+ typeof r.defaultLocationId === "string" ? r.defaultLocationId.trim() : "";
+ // A default naming a location that is gone falls back to the first one, not
+ // to "": with a location configured, "no default" is never the answer the
+ // operator wanted, and a blank default silently disables every prefill.
+ const defaultLocationId = locations.some((l) => l.id === wanted)
+ ? wanted
+ : (locations[0]?.id ?? "");
+ // THE SAVED-VIDEO STORE'S LOCATION IS NOT FALLEN BACK, and the asymmetry
+ // with `defaultLocationId` above is deliberate. That one is a PREFERENCE, so
+ // picking another location when the named one is gone is helpful. This one is
+ // a RECORD OF WHERE BYTES ARE: pointing it at a different location because
+ // the recorded one was deleted would claim the store had moved when nothing
+ // had. A dangling id sanitizes to "" — "in place" — which is what the disk
+ // says as soon as anybody looks, and the symlink (if any) keeps working
+ // regardless, because the store is reached through it and not through this.
+ const savedWanted =
+ typeof r.savedVideosLocationId === "string"
+ ? r.savedVideosLocationId.trim()
+ : "";
+ const savedVideosLocationId = locations.some((l) => l.id === savedWanted)
+ ? savedWanted
+ : "";
+ return {
+ locations,
+ defaultLocationId,
+ ...(savedVideosLocationId ? { savedVideosLocationId } : {}),
+ };
+}
+
+// 4 hours. Measured: videos over this are 8.2% of the corpus by count but hold
+// 46% of all transcript tokens, so they are where a sweep's wall-clock actually
+// goes and where chunk-seam bugs live.
+export const DIGEST_LONG_TAIL_DEFAULT_SECONDS = 4 * 3600;
+export const DIGEST_LONG_TAIL_MAX_SECONDS = 24 * 3600;
+
+export function defaultDigest(): DigestSettings {
+ return {
+ // OFF. The metered lane is built but never the default — see PLAN.md.
+ remoteEnabled: false,
+ longTailSeconds: DIGEST_LONG_TAIL_DEFAULT_SECONDS,
+ localAppId: DEFAULT_DIGEST_APP_ID,
+ remoteAppId: CLAUDE_DIGEST_APP_ID,
+ // Empty on purpose: every per-app knob falls through to its own default
+ // constant (resolveNumCtx -> DEFAULT_DIGEST_NUM_CTX, now 8192, and
+ // maxCuesForContext sizes the chunk to it). Seeding a copy of those values
+ // here would give the same number two homes and let them drift.
+ apps: {},
+ // ON. Real GPU contention with the transcription engine is a genuine cost
+ // (re-priced: 11.2 s/chunk idle against 24.9 s/chunk on a contended box), so
+ // the safe default is to step aside; turning it off is the deliberate choice.
+ yieldToTranscription: true,
+ // OFF. A CPU-pinned worker is not GPU contention, and treating it as such
+ // stalled the digest lane for nothing. See DigestSettings.yieldToCpuWorkers.
+ yieldToCpuWorkers: false,
+ spendCapUsd: 0,
+ sections: ["chapters"],
+ timestampMode: DEFAULT_DIGEST_TIMESTAMP_MODE,
+ promptVariant: "",
+ };
+}
+
+// Coerce a raw settings.digest.apps value into a clean keyed map of
+// DigestAppConfig. Mirrors sanitizeTranscriptionApps — INCLUDING its
+// Array.isArray guard, without which a JSON array would pass the typeof check and
+// produce numeric-keyed garbage.
+export function sanitizeDigestApps(
+ value: unknown,
+): Record<string, DigestAppConfig> {
+ if (!value || typeof value !== "object" || Array.isArray(value)) return {};
+ const out: Record<string, DigestAppConfig> = {};
+ for (const [id, raw] of Object.entries(value as Record<string, unknown>)) {
+ if (!raw || typeof raw !== "object") continue;
+ const r = raw as Record<string, unknown>;
+ const cfg: DigestAppConfig = {};
+ if (typeof r.bin === "string" && r.bin.trim()) cfg.bin = r.bin.trim();
+ if (typeof r.baseUrl === "string" && r.baseUrl.trim()) {
+ cfg.baseUrl = r.baseUrl.trim();
+ }
+ if (typeof r.model === "string" && r.model.trim()) cfg.model = r.model.trim();
+ if (typeof r.numCtx === "number" && r.numCtx > 0) {
+ cfg.numCtx = Math.floor(r.numCtx);
+ }
+ if (typeof r.temperature === "number" && r.temperature >= 0) {
+ cfg.temperature = r.temperature;
+ }
+ if (typeof r.timeoutMs === "number" && r.timeoutMs > 0) {
+ cfg.timeoutMs = Math.floor(r.timeoutMs);
+ }
+ // Only carried when explicitly set — see DigestAppConfig.think.
+ if (typeof r.think === "boolean") cfg.think = r.think;
+ out[id] = cfg;
+ }
+ return out;
+}
+
+export function sanitizeDigest(value: unknown): DigestSettings {
+ const d = defaultDigest();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const sections = Array.isArray(r.sections)
+ ? (r.sections.filter(isDigestSectionKind) as DigestSectionKind[])
+ : [];
+ return {
+ remoteEnabled: r.remoteEnabled === true,
+ longTailSeconds: clampPositiveInt(
+ r.longTailSeconds,
+ d.longTailSeconds,
+ DIGEST_LONG_TAIL_MAX_SECONDS,
+ ),
+ // Unknown app ids are not rejected here: getDigestApp() is total and falls
+ // back to the local default, so a stale id degrades rather than breaking.
+ localAppId:
+ typeof r.localAppId === "string" && r.localAppId.trim()
+ ? r.localAppId.trim()
+ : d.localAppId,
+ remoteAppId:
+ typeof r.remoteAppId === "string" && r.remoteAppId.trim()
+ ? r.remoteAppId.trim()
+ : d.remoteAppId,
+ apps: sanitizeDigestApps(r.apps),
+ // Defaults to ON when absent — `=== false` rather than `!== true`, so a
+ // settings file written before this field existed keeps the GPU-safe
+ // behaviour instead of silently opting into contention.
+ yieldToTranscription: r.yieldToTranscription !== false,
+ // The OPPOSITE idiom, and deliberately so: `=== true`, so absence falls to
+ // OFF. The field's absence means a settings file written before the CPU-worker
+ // bug was found, and for those files OFF is the FIXED behaviour, not a silent
+ // change of intent — nobody ever asked to stall the digest lane for a CPU
+ // transcription. `yieldToTranscription` still gates the whole thing, so the
+ // GPU-safe default is untouched.
+ yieldToCpuWorkers: r.yieldToCpuWorkers === true,
+ spendCapUsd:
+ typeof r.spendCapUsd === "number" && r.spendCapUsd > 0
+ ? Math.round(r.spendCapUsd * 100) / 100
+ : 0,
+ // An empty/garbage list would silently generate nothing, so fall back to the
+ // default rather than honoring it.
+ sections: sections.length > 0 ? sections : d.sections,
+ timestampMode: isDigestTimestampMode(r.timestampMode)
+ ? r.timestampMode
+ : d.timestampMode,
+ // Trimmed and length-capped: it goes into provenance on every record, and a
+ // runaway value would bloat 119k sidecars.
+ promptVariant:
+ typeof r.promptVariant === "string"
+ ? r.promptVariant.trim().slice(0, 40)
+ : d.promptVariant,
+ };
+}
+
+// Every known section kind, for the settings UI's checkbox list.
+export const DIGEST_SECTION_OPTIONS = DIGEST_SECTION_KINDS;
+export const DIGEST_TIMESTAMP_MODE_OPTIONS = DIGEST_TIMESTAMP_MODES;
+
+export const BUILD_MAX_PARALLEL_DEFAULT = 2;
+export const BUILD_MAX_PARALLEL_MAX = 16;
+export const DEFAULT_BUILD_IMAGE = "yt-dlp-transcript-browser-build";
+export const DEFAULT_BUILD_DOCKERFILE = "Dockerfile.build";
+
+export function isBuildMode(v: unknown): v is BuildMode {
+ return v === "basic" || v === "docker";
+}
+
+export function defaultBuildPipeline(): BuildPipelineSettings {
+ return {
+ mode: "basic",
+ maxParallelBuilds: BUILD_MAX_PARALLEL_DEFAULT,
+ dockerImage: DEFAULT_BUILD_IMAGE,
+ dockerfile: DEFAULT_BUILD_DOCKERFILE,
+ };
+}
+
+// Coerce a raw settings.buildPipeline value into a clean BuildPipelineSettings,
+// falling back to defaults for missing/ill-typed fields.
+export function sanitizeBuildPipeline(value: unknown): BuildPipelineSettings {
+ const d = defaultBuildPipeline();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const dockerImage =
+ typeof r.dockerImage === "string" && r.dockerImage.trim()
+ ? r.dockerImage.trim()
+ : d.dockerImage;
+ const dockerfile =
+ typeof r.dockerfile === "string" && r.dockerfile.trim()
+ ? r.dockerfile.trim()
+ : d.dockerfile;
+ return {
+ mode: isBuildMode(r.mode) ? r.mode : d.mode,
+ maxParallelBuilds: clampPositiveInt(
+ r.maxParallelBuilds,
+ d.maxParallelBuilds,
+ BUILD_MAX_PARALLEL_MAX,
+ ),
+ dockerImage,
+ dockerfile,
+ };
+}
+
+export function defaultBackfill(): BackfillSettings {
+ return {
+ concurrency: 1,
+ // See BackfillSettings.allowRedownload — this one holds disk.
+ allowRedownload: false,
+ };
+}
+
+export function sanitizeBackfill(value: unknown): BackfillSettings {
+ const d = defaultBackfill();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ return {
+ // Clamped rather than rejected: a hand-edited 5 means "as much as possible",
+ // and reading it as 0 would be the opposite of the intent.
+ concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16),
+ allowRedownload: r.allowRedownload === true,
+ };
+}
+
+export function defaultAttribution(): AttributionSettings {
+ return {
+ // OFF, and both lanes OFF under it. See AttributionSettings.
+ enabled: false,
+ appId: DEFAULT_DIGEST_APP_ID,
+ model: "",
+ diarizedEnabled: false,
+ textOnlyEnabled: false,
+ promptVersion: ATTRIBUTION_PROMPT_VERSION,
+ };
+}
+
+export function sanitizeAttribution(value: unknown): AttributionSettings {
+ const d = defaultAttribution();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const str = (v: unknown, fallback: string) =>
+ typeof v === "string" && v.trim() ? v.trim() : fallback;
+ return {
+ enabled: r.enabled === true,
+ appId: str(r.appId, d.appId),
+ // NOT `str(...)` with a fallback: empty is a meaningful value here ("use the
+ // app's own model"), so an empty string must survive rather than reverting
+ // to a default that is also empty by coincidence.
+ model: typeof r.model === "string" ? r.model.trim() : d.model,
+ diarizedEnabled: r.diarizedEnabled === true,
+ textOnlyEnabled: r.textOnlyEnabled === true,
+ // FLOORED at the shipped constant, never merely defaulted. A hand-edited
+ // value below it would pin freshness to a superseded prompt generation and
+ // freeze its output into the corpus — see AttributionSettings.promptVersion.
+ promptVersion:
+ typeof r.promptVersion === "number" && Number.isFinite(r.promptVersion)
+ ? Math.max(ATTRIBUTION_PROMPT_VERSION, Math.floor(r.promptVersion))
+ : d.promptVersion,
+ };
+}
+
+export function defaultDiarization(): DiarizationSettings {
+ return {
+ // OFF. Capture is opt-in: turning it on makes the cleanup sweep start
+ // refusing to delete audio for transcribed-but-undiarized videos, which is
+ // correct but is a disk-pressure decision an operator should make.
+ enabled: false,
+ // OFF — see DiarizationSettings.inlineAfterTranscribe. Diarization is slower
+ // than the transcription it would follow, so inline is the exception.
+ inlineAfterTranscribe: false,
+ // Measured, not sherpa-onnx's 0.5 — see DiarizationSettings.threshold. The
+ // constant lives in lib/diarization.ts because isDiarizationFresh needs it
+ // to normalize an absent recorded threshold; importing it keeps the default
+ // and the comparator from drifting apart.
+ threshold: DEFAULT_DIARIZATION_THRESHOLD,
+ threads: 4,
+ // The engine every sidecar on disk was produced by. Switching is an explicit
+ // decision that restates the freshness identity — see DiarizationSettings.
+ engine: DEFAULT_DIARIZATION_ENGINE,
+ // Only consulted when engine is "sortformer". Defaulting to the GPU is safe
+ // because the lane yields the card to transcription rather than sharing it.
+ backend: "vulkan",
+ python: "python3",
+ segModel: "",
+ embModel: "",
+ sortformerBin: "",
+ sortformerModel: "",
+ concurrency: 1,
+ // OFF, because windowing made it unnecessary — which is what it was always
+ // for. It shipped at 4 hours as a stopgap while long recordings were being
+ // OOM-killed; the engine now processes them in windows and the 6h12m file
+ // that was being killed at 10.5 GB completes at ~2 GB. The knob stays, and
+ // stays honest about what it does, for a machine smaller than this one or a
+ // recording longer than anything measured here.
+ maxAudioHours: 0,
+ };
+}
+
+export function sanitizeDiarization(value: unknown): DiarizationSettings {
+ const d = defaultDiarization();
+ if (!value || typeof value !== "object") return d;
+ const r = value as Record<string, unknown>;
+ const str = (v: unknown, fallback: string) =>
+ typeof v === "string" && v.trim() ? v.trim() : fallback;
+ return {
+ enabled: r.enabled === true,
+ inlineAfterTranscribe: r.inlineAfterTranscribe === true,
+ threshold:
+ typeof r.threshold === "number" &&
+ Number.isFinite(r.threshold) &&
+ r.threshold > 0
+ ? r.threshold
+ : d.threshold,
+ threads: clampPositiveInt(r.threads, d.threads, 64),
+ // An unknown engine falls back to the default rather than disabling the lane:
+ // a typo in settings.json must not silently stop diarization, and the default
+ // is the one every existing sidecar already matches.
+ engine: DIARIZATION_ENGINE_IDS.includes(r.engine as DiarizationEngineId)
+ ? (r.engine as DiarizationEngineId)
+ : d.engine,
+ backend: DIARIZATION_BACKENDS.includes(r.backend as DiarizationBackend)
+ ? (r.backend as DiarizationBackend)
+ : d.backend,
+ python: str(r.python, d.python),
+ segModel: str(r.segModel, d.segModel),
+ embModel: str(r.embModel, d.embModel),
+ sortformerBin: str(r.sortformerBin, d.sortformerBin),
+ sortformerModel: str(r.sortformerModel, d.sortformerModel),
+ concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16),
+ // 0 is meaningful here (cap off), so this cannot use clampPositiveInt.
+ // Fractional hours are allowed — the knob is a duration, not a count.
+ maxAudioHours:
+ typeof r.maxAudioHours === "number" &&
+ Number.isFinite(r.maxAudioHours) &&
+ r.maxAudioHours >= 0
+ ? r.maxAudioHours
+ : d.maxAudioHours,
+ };
+}
+
+const SOCIAL_URL_RE = /^(https?:\/\/|mailto:|\/)/i;
+
+// Normalize the family hub URL into a trailing-slash-free absolute http(s) URL.
+// Returns "" for anything that isn't a usable absolute URL (the "no hub" state).
+// Kept local to settings.ts to avoid a settings<->homepage import cycle; mirrors
+// parseHomepageUrl() in homepage.ts.
+export function normalizeHomepageUrl(input: unknown): string {
+ if (typeof input !== "string") return "";
+ const trimmed = input.trim().replace(/\/+$/, "");
+ return /^https?:\/\/\S+/i.test(trimmed) ? trimmed : "";
+}
+
+export function parseSocialLinks(input: unknown): SocialLink[] {
+ if (!Array.isArray(input)) return [];
+ const out: SocialLink[] = [];
+ for (const raw of input) {
+ if (!raw || typeof raw !== "object") continue;
+ const r = raw as Record<string, unknown>;
+ const label = typeof r.label === "string" ? r.label.trim() : "";
+ const url = typeof r.url === "string" ? r.url.trim() : "";
+ const svg = typeof r.svg === "string" ? r.svg : "";
+ if (!label || !url || !svg) continue;
+ if (!SOCIAL_URL_RE.test(url)) continue;
+ out.push({ label, url, svg });
+ }
+ return out;
+}
+
+// Normalize an admin-provided SVG snippet for inline use in the export
+// footer. Returns null on anything that looks unsafe or unrenderable.
+// Steps: trim, allowlist-check, strip width/height, force fill="currentColor"
+// + aria-hidden on the root <svg>. Requires a viewBox so the icon scales.
+export function normalizeSocialSvg(raw: string): string | null {
+ if (typeof raw !== "string") return null;
+ const trimmed = raw.trim();
+ if (!trimmed.startsWith("<svg") || !trimmed.endsWith("</svg>")) return null;
+ if (/<script\b/i.test(trimmed)) return null;
+ if (/<foreignObject\b/i.test(trimmed)) return null;
+ if (/<iframe\b/i.test(trimmed)) return null;
+ if (/javascript:/i.test(trimmed)) return null;
+ if (/\son[a-z]+\s*=/i.test(trimmed)) return null;
+ if (/<\?|<!ENTITY/i.test(trimmed)) return null;
+
+ const openEnd = trimmed.indexOf(">");
+ if (openEnd < 0) return null;
+ let opening = trimmed.slice(0, openEnd);
+ const rest = trimmed.slice(openEnd);
+
+ if (!/\sviewBox\s*=\s*"/i.test(opening)) return null;
+
+ opening = opening.replace(/\s(width|height)\s*=\s*"[^"]*"/gi, "");
+ opening = opening.replace(/\s(width|height)\s*=\s*'[^']*'/gi, "");
+
+ if (!/\sfill\s*=/i.test(opening)) {
+ opening = opening.replace(/^<svg/i, '<svg fill="currentColor"');
+ }
+ if (!/\saria-hidden\s*=/i.test(opening)) {
+ opening = opening.replace(/^<svg/i, '<svg aria-hidden="true"');
+ }
+ return opening + rest;
+}
+
+// normalizeSocialSvg() deliberately STRIPS width/height so the icon scales to its
+// wrapper. The cost is that a viewBox-only <svg> has no intrinsic size, so before
+// the stylesheet loads on a static host it paints at the replaced-element default
+// (huge) — the "flash of giant social icons" FOUC. sizeSocialSvg() re-injects an
+// intrinsic pixel size at RENDER time (existing site.json files already have the
+// attributes stripped, so this must run on read, not just on write). The size is
+// an *attribute*, not inline style, so a wrapper's `w-*`/`h-*` utilities still win
+// once CSS loads — it only governs the pre-CSS first paint.
+export function sizeSocialSvg(svg: string, px = 20): string {
+ if (typeof svg !== "string") return svg;
+ if (/^<svg[^>]*\swidth\s*=/i.test(svg)) return svg; // already sized
+ return svg.replace(/^<svg\b/i, `<svg width="${px}" height="${px}"`);
+}
+
+export function clampSleepBetweenDownloadsSeconds(value: unknown): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? Math.floor(value)
+ : SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS;
+ if (n < 0) return 0;
+ if (n > SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS) {
+ return SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS;
+ }
+ return n;
+}
+
+export function clampMinFreeDiskGB(value: unknown): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? Math.floor(value)
+ : MIN_FREE_DISK_GB_DEFAULT;
+ if (n < 0) return 0;
+ if (n > MIN_FREE_DISK_GB_MAX) return MIN_FREE_DISK_GB_MAX;
+ return n;
+}
+
+export function clampResumeMarginGB(value: unknown): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? Math.floor(value)
+ : RESUME_MARGIN_GB_DEFAULT;
+ if (n < 0) return 0;
+ if (n > RESUME_MARGIN_GB_MAX) return RESUME_MARGIN_GB_MAX;
+ return n;
+}
+
+export function clampParallelTranscriptions(value: unknown): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? Math.floor(value)
+ : PARALLEL_TRANSCRIPTIONS_DEFAULT;
+ if (n < 1) return 1;
+ if (n > PARALLEL_TRANSCRIPTIONS_MAX) return PARALLEL_TRANSCRIPTIONS_MAX;
+ return n;
+}
+
+// 0 means "disabled" and is preserved as-is. Anything else is clamped into the
+// [MIN, MAX] window; a non-finite value falls back to the default cadence.
+export function clampAutoRefreshIntervalSeconds(value: unknown): number {
+ if (typeof value !== "number" || !Number.isFinite(value)) {
+ return AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS;
+ }
+ const n = Math.floor(value);
+ if (n <= 0) return 0;
+ if (n < AUTO_REFRESH_INTERVAL_MIN_SECONDS) {
+ return AUTO_REFRESH_INTERVAL_MIN_SECONDS;
+ }
+ if (n > AUTO_REFRESH_INTERVAL_MAX_SECONDS) {
+ return AUTO_REFRESH_INTERVAL_MAX_SECONDS;
+ }
+ return n;
+}
+
+export function clampPageBytes(value: unknown): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? value
+ : TRANSCRIPT_PAGE_DEFAULT_BYTES;
+ if (n < TRANSCRIPT_PAGE_MIN_BYTES) return TRANSCRIPT_PAGE_MIN_BYTES;
+ if (n > TRANSCRIPT_PAGE_HARD_CAP_BYTES) return TRANSCRIPT_PAGE_HARD_CAP_BYTES;
+ return Math.floor(n);
+}
+
+
+// Coerce a raw settings.transcriptionApps value into a clean keyed map of
+// AppInstanceConfig, dropping unknown/ill-typed fields.
+export function sanitizeTranscriptionApps(
+ value: unknown,
+): Record<string, AppInstanceConfig> {
+ if (!value || typeof value !== "object" || Array.isArray(value)) return {};
+ const out: Record<string, AppInstanceConfig> = {};
+ for (const [id, raw] of Object.entries(value as Record<string, unknown>)) {
+ if (!raw || typeof raw !== "object") continue;
+ out[id] = sanitizeWorkerConfig(raw);
+ }
+ return out;
+}
+
+
+// --- The schema -------------------------------------------------------------
+
+// Global, OPERATIONAL settings shared across every site this editor powers.
+// Per-site presentation (branding, social links, channel groups, membership)
+// lives in sites/<siteId>/site.json — see common/lib/site.ts.
+//
+// FIELD ORDER IS FILE ORDER: zod emits keys in the order they are declared, and
+// writeSettings writes what the schema emits, so reordering these reorders every
+// settings.json on its next save.
+export const siteSettingsSchema = z.object({
+ adminTitle: settingsField((v): string =>
+ typeof v === "string" && v.trim() ? v.trim() : DEFAULT_ADMIN_TITLE).describe(
+ "Title for the EDITOR admin shell only (the editor manages all sites and so is not tied to any one site's branding). Public sites get their own titles from site.json.",
+ ),
+ maxTranscriptPageBytes: settingsField((v): number => clampPageBytes(v)).describe(
+ "Target size (bytes) of one exported transcript page shard — the unit the export site fetches. Clamped into [TRANSCRIPT_PAGE_MIN_BYTES, TRANSCRIPT_PAGE_HARD_CAP_BYTES] (256 KiB – 20 MiB); default 8 MiB.",
+ ),
+ transcriptionApp: settingsField((v): string =>
+ typeof v === "string" && TRANSCRIPTION_APPS[v] ? v : DEFAULT_TRANSCRIPTION_APP_ID).describe(
+ "Active transcription app id (key into TRANSCRIPTION_APPS, e.g. \"whisper-cpp\" or \"chough\"). Selected globally; see common/lib/transcriptionApps.ts.",
+ ),
+ transcriptionApps: settingsField((v): Record<string, AppInstanceConfig> => sanitizeTranscriptionApps(v)).describe(
+ "Per-app configuration, keyed by app id. Each app reads only its own block; a missing block means \"use the app's defaults\". DEPRECATED in favor of `workers` (each local worker carries its own config); kept one release to drive migration and allow rollback. See common/lib/workers.ts.",
+ ),
+ workers: workersSchema.describe(
+ "Configured transcription workers (named processing slots). The scheduler distributes each video to the highest-priority free worker. A settings.json predating this field is migrated to a single enabled worker from the active app (see defaultWorkersFromApps). See common/lib/workers.ts.",
+ ),
+ cookiesFromBrowser: settingsField((v): string => (typeof v === "string" ? v.trim() : "")).describe(
+ "Browser spec (e.g. \"firefox\", \"chrome:Default\") passed to `yt-dlp --cookies-from-browser`. WHEN it is passed is governed by `cookieMode` below. Empty string = no cookies configured. Per-channel override available (ChannelConfig.cookiesFromBrowser).",
+ ),
+ cookieMode: settingsField((v): CookieMode => (isCookieMode(v) ? v : DEFAULT_COOKIE_MODE)).describe(
+ "How yt-dlp invocations use the configured cookies (see common/lib/cookiePolicy.ts): \"always\" passes them on every invocation, \"when-required\" (default; the historical behavior) only to retry an auth/age failure, \"defer\" never in normal runs — auth-gated videos are excluded from batches and collected into the per-channel \"Needs cookies\" bucket for a manual cookie run. Per-channel override available (ChannelConfig.cookieMode).",
+ ),
+ sleepBetweenDownloadsSeconds: settingsField((v): number => clampSleepBetweenDownloadsSeconds(v)).describe(
+ "Pause (seconds) inserted between per-video yt-dlp invocations in managed batch downloads. yt-dlp's own `-t sleep` only paces requests within one invocation, so without this the managed loop hammers the source IP back-to-back. 0 disables. Per-channel override available.",
+ ),
+ downloadFormat: settingsField((v): DownloadFormatPreset =>
+ isDownloadFormatPreset(v) ? v : "auto").describe(
+ "Default yt-dlp `-f` download format for every channel that doesn't set its own (ChannelConfig.downloadFormat). \"auto\" picks per-source: `original` for Odysee (whose HLS rungs are CDN-truncated), `bestaudio/worst` elsewhere. See common/ytdlp/downloadFormat.ts.",
+ ),
+ minFreeDiskGB: settingsField((v): number => clampMinFreeDiskGB(v)).describe(
+ "Minimum free disk space (GB) required on the transcripts data directory for downloads to run. When free space is below this floor, a download job is prevented from starting and a running batch stops launching new videos (the in-flight one finishes). 0 disables the gate. See common/lib/diskSpace.ts.",
+ ),
+ resumeMarginGB: settingsField((v): number => clampResumeMarginGB(v)).describe(
+ "Extra headroom (GB) above minFreeDiskGB that a stopped pipeline must see before it resumes. Resuming at the same number we stopped at flaps — the first restarted download pushes free space back under the floor. This is the hysteresis margin, so \"resumed\" means the operator actually freed something rather than a scratch file being cleaned up. 0 disables the hysteresis (resume at the floor). See diskGate() in common/lib/diskSpace.ts.",
+ ),
+ parallelTranscriptions: settingsField((v): number => clampParallelTranscriptions(v)).describe(
+ "Default number of videos transcribed in parallel when a \"Transcribe missing\" / bucket run doesn't specify its own concurrency. The per-run Concurrency input in the channel UI overrides this for a single run.",
+ ),
+ inlineTranscribeOnFallback: settingsField((v): boolean => v === true).describe(
+ "When true, the no-subs fallback in the managed downloader runs whisper inline immediately after the audio download succeeds. When false (default), audio is left for the next \"Transcribe missing\" pass so a batch download finishes faster and whisper can parallelize.",
+ ),
+ skipLiveDownloads: settingsField((v): boolean => v !== false).describe(
+ "When true (default), managed downloads skip videos that are currently live or scheduled/upcoming, decided from a metadata-only prefetch pass. Finished livestream VODs (was_live) are NOT skipped and download normally. A skip is recorded but not archived, so the next sync/download-missing retries the video once the stream ends. Per-channel override available (ChannelConfig.skipLiveDownloads).",
+ ),
+ verifyAvailabilityBeforeClean: settingsField((v): boolean => v !== false).describe(
+ "Whether the transcribed-audio cleanup sweep checks each candidate is still available upstream before deleting its audio, pinning (do-not-clean) any video found permanently gone. The delete is irreversible and a gone video's audio is irreplaceable, so this defaults to true. Turn it off for an offline or URL-less setup, where the check can never resolve and cleanup would otherwise never delete anything. See verifyBeforeClean.ts.",
+ ),
+ buildArchives: settingsField((v): boolean => v !== false).describe(
+ "Whether site builds generate downloadable transcript/live-chat archive zips (into public/archives, linked on the Downloads page). Global default; a site can opt out via site.json `archives: false`, and a single build can skip via the \"Skip archive zips\" build control. Opt-out: default true.",
+ ),
+ archiveStorage: settingsField((v): { bucket: string; publicBaseUrl: string } => {
+ const r = (v && typeof v === "object" ? v : {}) as Record<string, unknown>;
+ return {
+ bucket: typeof r.bucket === "string" ? r.bucket.trim() : "",
+ publicBaseUrl:
+ typeof r.publicBaseUrl === "string" ? r.publicBaseUrl.trim() : "",
+ };
+ }).describe(
+ "Overflow object storage (Cloudflare R2) for archive zips that exceed the Pages per-file size cap (see Site.archiveMaxBytes). When both fields are set, an oversize archive is uploaded here on deploy — via `wrangler r2 object put`, keyed `<siteId>/archives/<file>` — instead of being dropped, and the Downloads page links to `<publicBaseUrl>/<key>`. Blank/absent → no overflow, so oversize archives stay unavailable (\"Too large to host\").",
+ ),
+ reportDebouncePreset: settingsField((v): ReportDebouncePreset =>
+ isReportDebouncePreset(v) ? v : DEFAULT_REPORT_DEBOUNCE_PRESET).describe(
+ "Debounce preset for the global snapshot scheduler: how long it waits after the last report-changing action before regenerating affected channel reports. See REPORT_DEBOUNCE_PRESETS. Default \"fast\" (~1s, no cap).",
+ ),
+ autoRefreshIntervalSeconds: settingsField((v): number => clampAutoRefreshIntervalSeconds(v)).describe(
+ "How often (seconds) the editor UI passively re-fetches the current page's server-rendered data via router.refresh(), so sidebar badges and reports stay live without a manual reload. Mounted globally; pauses while the tab is hidden. 0 disables passive refresh entirely. See AUTO_REFRESH_INTERVAL_*.",
+ ),
+ syncScheduler: settingsField((v): SyncSchedulerSettings => sanitizeSyncScheduler(v)).describe(
+ "Global configuration for the scheduled (cron-driven) channel sync system. The per-channel cadence lives on ChannelConfig.syncIntervalMinutes; this block holds the defaults and guard rails the scheduler applies across all channels. See common/jobs/syncScheduler.ts.",
+ ),
+ autoQueue: autoQueueSchema.describe(
+ "Configuration for the automatic priority-queue runners (auto-transcribe / auto-download). Each holds a tree policy that decides which channel's video to process next, cross-channel, by priority/round-robin/weighted-fair rules. Independent of syncScheduler (which decides staleness, not work order). See common/jobs/autoQueuePolicy.ts.",
+ ),
+ channelPriority: channelPrioritySchema.describe(
+ "THE OPERATOR-FACING PRIORITY MODEL: one tier per channel plus one corpus-wide focus selector. It is the SOURCE the four `autoQueue[lane].root` trees are compiled from (common/lib/channelPriority.ts), not a second mechanism beside them — and its `paused` tier is the one part that is not a tree shape, filtering the runner's channel list instead. An empty document (the default) is today's behaviour exactly: no focus, every channel normal, the stored trees stand.",
+ ),
+ socialLinks: settingsField((v): SocialLink[] => parseSocialLinks(v)).describe(
+ "Default social links applied to every site that doesn't define its own. A site inherits these unless its site.json carries an explicit `socialLinks` array — see Site.socialLinks / resolveSocialLinks in common/lib/site.ts. The one presentation field that lives globally so a shared footer doesn't have to be repeated per site.",
+ ),
+ homepageUrl: settingsField((v): string => normalizeHomepageUrl(v)).describe(
+ "Absolute public URL of the family hub/homepage (e.g. \"https://archilyzer.pages.dev\"). Every export site links back to it (\"the family\" backlink) when set. Empty = no hub link rendered. Normalized to a trailing-slash-free http(s) URL.",
+ ),
+ savedVideoBackup: settingsField((v): SavedVideoBackupSettings => sanitizeSavedVideoBackup(v)).describe(
+ "Backup configuration for the saved-video store (Phase 4 of the video-persistence feature). When enabled with a destination, the store is mirrored there (additively, no deletes) with a per-backup manifest, and the sync scheduler runs the backup on the configured cadence. See common/controller/backupSavedVideos.ts.",
+ ),
+ storage: settingsField((v): StorageSettings => sanitizeStorage(v)).describe(
+ "Where a channel's downloaded media goes when it is relocated off the corpus disk. A DEFAULT ONLY: the relocate controller never reads it and always takes an explicit root, so this is the value the per-channel Storage panel prefills and the /channels bulk move falls back to. Blank = no default. See StorageSettings.",
+ ),
+ buildPipeline: settingsField((v): BuildPipelineSettings => sanitizeBuildPipeline(v)).describe(
+ "How the static export is built: \"basic\" reuses the single export/ tree and serializes builds on one queue (the long-standing behavior); \"docker\" runs each site's build in an isolated container for safe parallelism. The Docker pipeline itself is a follow-up; this block persists the chosen mode plus the container/concurrency knobs the deploy page and the future orchestrator read.",
+ ),
+ digest: settingsField((v): DigestSettings => sanitizeDigest(v)).describe(
+ "AI digest generation (chapters + topic tags over the existing transcripts). Local-first: the metered lane is off by default. See DigestSettings.",
+ ),
+ diarization: settingsField((v): DiarizationSettings => sanitizeDiarization(v)).describe(
+ "Speaker diarization captured right after transcription, while the audio is still on disk. OFF by default. See DiarizationSettings.",
+ ),
+ backfill: settingsField((v): BackfillSettings => sanitizeBackfill(v)).describe(
+ "The generic catch-up lane for derived data the existing corpus predates. OFF by default, and idle-only when on. See BackfillSettings.",
+ ),
+ attribution: settingsField((v): AttributionSettings => sanitizeAttribution(v)).describe(
+ "Naming the speakers diarization found (or reconstructing them from the transcript when it found none). OFF by default. See AttributionSettings.",
+ ),
+});
+
+export type SiteSettings = z.infer<typeof siteSettingsSchema>;
+
+// The whole default settings object, without touching disk: the schema's answer
+// for an empty file. Not a second literal — a default that lived anywhere but in
+// the field's own coercion would be a second place to change it.
+export function defaults(): SiteSettings {
+ return siteSettingsSchema.parse({});
+}
+
+// Exported under this name too, because tests and e2e helpers already say it:
+// a caller that needs a settings-SHAPED value rather than the operator's actual
+// configuration builds one here without a settings.json.
+export function defaultSiteSettings(): SiteSettings {
+ return defaults();
+}