// THE SETTINGS SCHEMA — one definition of settings.json, used by the reader, // the writer, the example file and the key table. // // one-core phase 3 slice 4a. What used to be three copies of the same list — // the `SiteSettings` type, the `defaults()` literal, and the two field-by-field // sanitizing literals in getSettings and writeSettings — is `siteSettingsSchema` // below. `SiteSettings` is inferred from it; `defaults()` is `parse({})`; // getSettings and writeSettings (lib/settings.ts) both parse through it; and // `common/bin/settings-example.ts` generates settings.json.example and SETTINGS.md // from it, including each field's `.describe()` text — which is where the // comments that used to sit on the `SiteSettings` type now live, so they have one // home and cannot drift from the key they describe. // // ZOD SUPPLIES THE PLUMBING, NOT THE ARITHMETIC. Every field is // `settingsField(coerce)` — `z.unknown().catch(undefined).transform(coerce)` — // and every `coerce` is the clamp or sanitizer that already existed, reused, so // no boundary moved. Each is total over `unknown`, and that — not zod, whose // `.catch` cannot fire on `z.unknown()` — is what makes a read never throw. // The nested blocks' own fields are documented in the `*_FIELD_DOCS` record // beside each block type (here and in workers.ts, channelPriority.ts, // autoQueueTypes.ts, storageLocations.ts, transcriptionApps.ts, digest.ts), // type-checked complete. Unknown keys are dropped by zod's default strip (never // `.passthrough()`), which is what retires a field: a key the schema does not // name cannot survive a read or a save. // // SERVER-ONLY IN PRACTICE: it imports node:path (sanitizeStorage) and is reached // through lib/settings.ts, which imports node:fs at module scope. A `"use // client"` form that needs a constant imports the constant — never this schema. // // NOT HERE: the three migrations keyed on a field's ABSENCE in the raw file // (sweeps → lanes, mediaRoot → locations, legacy transcribe* → app registry and // synthesized workers), because a parsed object cannot tell "absent" from // "default". They run in getSettings around the parse. See lib/settings.ts. import path from "node:path"; import { z } from "zod"; // The auto-queue, workers and channel-priority blocks keep their own // sanitizers in their own homes; these are their zod seams. import { autoQueueSchema, channelPrioritySchema, settingsField, workersSchema, } from "./settingsFieldSchemas"; import { SYNC_INTERVAL_MAX_MINUTES } from "./channelConfig"; import { DEFAULT_SOURCE_VIDEO_QUALITY, isDownloadFormatPreset, isSourceVideoQuality, type DownloadFormatPreset, type SourceVideoQuality, } from "../ytdlp/downloadFormat"; import { type AppInstanceConfig, DEFAULT_TRANSCRIPTION_APP_ID, TRANSCRIPTION_APPS, } from "./transcriptionApps"; import { sanitizeWorkerConfig } from "./workers"; import { INTERNAL_LOCATION_ID, type StorageLocation, type StorageSettings, type StorageVolume, } from "./storageLocations"; import { sanitizeStorageHealth } from "./storageHealthTimings"; import { DEFAULT_DIARIZATION_ENGINE, DEFAULT_DIARIZATION_THRESHOLD, DIARIZATION_BACKENDS, DIARIZATION_ENGINE_IDS, type DiarizationBackend, type DiarizationEngineId, } from "./diarization"; import { ATTRIBUTION_PROMPT_VERSION } from "./attribution"; import { DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, type ArchiveOrgFetchSettings, } from "./archiveOrgTorrent"; import { sanitizeSeeder, type SeederSettings } from "./seederSettings"; export { SEEDER_SETTINGS_FIELD_DOCS, type SeederSettings } from "./seederSettings"; import { DEFAULT_COOKIE_MODE, isCookieMode, type CookieMode, } from "./cookiePolicy"; // From the CLIENT-SAFE digest module, deliberately — digestApps.ts imports execa, // and settings.ts must stay reachable from anywhere. import { CLAUDE_DIGEST_APP_ID, DEFAULT_DIGEST_APP_ID, DEFAULT_DIGEST_TIMESTAMP_MODE, DIGEST_SECTION_KINDS, DIGEST_TIMESTAMP_MODES, isDigestSectionKind, isDigestTimestampMode, type DigestAppConfig, type DigestSectionKind, type DigestTimestampMode, } from "./digest"; import type { FieldDocs } from "./fieldDocs"; import { previewBranchProblem } from "./pagesDeploy"; import { sanitizeSocial, type SocialSettings, type XSocialSettings, } from "../social/xCookieSource"; export type { Worker } from "./workers"; export type { SocialSettings, XSocialSettings } from "../social/xCookieSource"; // The `social` block (release 16 slice XL). The types and the coercion live // with the source they choose, in social/xCookieSource.ts (pure, client-safe); // the documentation lives here with every other block's. export const SOCIAL_SETTINGS_FIELD_DOCS: FieldDocs = { x: "X / Twitter. See `social.x` below.", }; export const X_SOCIAL_SETTINGS_FIELD_DOCS: FieldDocs = { cookieSource: "Where the X fetchers' login comes from. `\"browser\"`: the operator's everyday browser, " + "named by `cookiesFromBrowser` — gallery-dl is handed `--cookies-from-browser ` and " + "reads it on every run, and the Playwright fallback reads the same store (Firefox only; " + "common/social/xBrowserLogin.ts), so the login lasts as long as the browser's. " + "`\"profile\"`: the session broker's persistent profile (\"Connect X account\" on /settings) " + "and the cookie jar it exports. ABSENT (the default) is resolved at read time, never " + "stored: `\"browser\"` when `cookiesFromBrowser` is set and no profile is connected (no " + "exported jar carrying an auth_token), else `\"profile\"`. The source, not `cookieMode`, " + "governs the X fetchers.", visibility: "Where X posts may appear. `\"public\"` (the default; absent): an X channel's posts are " + "built into every site that has the channel. `\"private\"`: every X channel's posts (a " + "channel with `sourceKind: \"social\"` and `platform: \"twitter\"`) are left out of every " + "PUBLIC site build — the channel with them, since posts are all an X channel holds — and " + "built only into PRIVATE sites (`site.json` `audience`). Nothing on disk changes and " + "fetching does not; a site already published changes on its next build and deploy, and " + "flipping back is a rebuild. Chosen in the X account session section of /settings; the " + "rule is common/lib/postsVisibility.ts.", }; export type { AutoQueueSettings } from "./autoQueueTypes"; export type { ChannelPriority } from "./channelPriority"; // Transcribe placeholder/arg helpers now live with the whisper-cpp app in // transcriptionApps.ts. Re-exported here so existing import sites keep working. export { type AppInstanceConfig, TRANSCRIBE_PLACEHOLDER_AUDIO, TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE, TRANSCRIBE_PLACEHOLDER_MODEL, TRANSCRIBE_KNOWN_PLACEHOLDERS, DEFAULT_TRANSCRIBE_ARGS, validateTranscribeArgs, } from "./transcriptionApps"; // Configuration for speaker attribution — putting names to the speaker turns. // // OFF by default, and that default is doing real work rather than being // cautious. The text-only lane costs roughly one model call per transcript // CHUNK, which on this corpus is ~194,000 calls, the same order as the digest // sweep — and the digest sweep has completed 0.17% of its own. Arming both at // once puts two multi-week lanes on the same 8 GB card with nothing to arbitrate // between them (the backfill lane's yield deliberately watches only the // transcription lane). Nothing here arms anything; a pilot decides whether the // corpus-wide text-only pass is worth 25-55 GPU-days at all. // Each field is documented in ATTRIBUTION_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). export type AttributionSettings = { enabled: boolean; appId: string; model: string; diarizedEnabled: boolean; textOnlyEnabled: boolean; promptVersion: number; }; export const ATTRIBUTION_SETTINGS_FIELD_DOCS: FieldDocs = { enabled: "Master switch. Off means the backfill registry reports no attribution " + "work at all — the feature gate every Operation has.", appId: "Which digest app runs the naming. Attribution IS a digest-app workload" + " — constrained JSON decoding over transcript text — so it reuses that " + "registry and that per-app config (settings.digest.apps[appId]) rather " + "than growing a second copy of the ollama URL, context size and " + "timeout.", model: "Model override. Empty = the app's configured model, then its default. " + "It is separate from the digest's because the two workloads may want " + "different sizes, and because it is part of the freshness identity: " + "sharing the digest's model field would make a digest bake-off " + "invalidate every attribution record on disk as a side effect.", diarizedEnabled: "The lanes, separately. Both default OFF even when `enabled` is on, so " + "turning the feature on to look at it cannot start a corpus sweep.\n\n" + "They are not a fallback pair. `diarized` is one call per video and " + "grounded in acoustic clustering; `textOnly` is ~30 calls and guesses " + "at identity across chunk seams. An operator may reasonably want the " + "first forever and the second never.", textOnlyEnabled: "The text-only attribution lane: names speakers from the transcript alone (~30 model calls per video, guessing identity across chunk seams). Default OFF even when `enabled` is on. See `diarizedEnabled` — the two are separate lanes, not a fallback pair.", promptVersion: "The prompt generation a record must match to count as fresh.\n\n" + "Defaults to (and is floored at) ATTRIBUTION_PROMPT_VERSION, the " + "shipped constant. Raising it forces a corpus-wide regeneration without" + " a code change, which is the honest way to redo everything after a " + "prompt tweak. It cannot be set BELOW the shipped constant, and that " + "floor is the lesson from digestPrompt.ts's version 1 -> 2 note: " + "pinning freshness to an older generation freezes output from a " + "superseded prompt into the corpus, looking identical to output from " + "the current one.", }; // Configuration for the backfill lane — the generic answer to "a derived-data // feature landed and 77,000 existing videos do not have it". // // WHAT THE LANE GETS is its own queueKey (concurrency with transcription) plus a // limit() that returns 0 to stand aside — the same mechanism the digest yield // uses, which fails OPEN so a bad read costs contention rather than a deadlock. // There is no priority system to join: the registry submits every named queue at // concurrency 1 and SchedulerTier only orders work within a single key. // // THE SHARE IS `concurrency` AND THE LANE'S `autoQueue.backfill.maxWorkers`; the // yield is the operation's declared `contendsFor`. Slice 1.3 retired the // `weight` scalar that used to mean both — see backfillLimit(). // Each field is documented in BACKFILL_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). export type BackfillSettings = { concurrency: number; allowRedownload: boolean; }; export const BACKFILL_SETTINGS_FIELD_DOCS: FieldDocs = { concurrency: "Slots the lane may use when it is not standing aside. Kept at 1 by " + "default for the same reason diarization.concurrency is: this is CPU-" + "bound work competing with GPU feeding and the digest sweep for the " + "same 8 threads.", allowRedownload: "Re-acquire media for videos whose input is GONE (audio deleted after " + "transcription). OFF by default and deliberately so: measured on this " + "corpus, 836 videos still have media and ~76,270 would need a re-" + "download — 91x the reachable work, against 45 GB free at 97% full. " + "When on, each re-fetched file is removed in a `finally` as soon as the" + " backfill has used it, unless the video is marked do-not-clean, or " + "unless the auto-transcribe policy would replace its auto-captions " + "(`replaceAutoSubs`, or a leaf on `downloadedAutoSubsOnly`), in which " + "case the audio is kept for that runner.\n\n" + "WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: " + "\"youtube\"` channel — which normally only fetches subtitles — the re-" + "acquire applies a PER-VIDEO transcribe override so yt-dlp lands audio " + "a diarizer can read; the channel's stored config is not changed. " + "Without that override the fetch re-downloads the captions the video " + "already has and lands nothing, which is what happened to ~16,000 " + "videos on eight channels in 2026-08.", }; // Configuration for the speaker-diarization capture lane. // // This exists because AUDIO IS THE ONE PERISHABLE INPUT in the pipeline: // cleanAudioFromTranscribed deletes it once a video is transcribed, so // diarization has to happen while the audio is still there or not at all. The // capture half is deliberately all that ships here — attribution, LLM speaker // naming, viewer badges and quote filtering can all be redone later from the // saved JSON, whereas the audio cannot. // Each field is documented in DIARIZATION_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). export type DiarizationSettings = { enabled: boolean; inlineAfterTranscribe: boolean; threshold: number; threads: number; engine: DiarizationEngineId; backend: DiarizationBackend; python: string; segModel: string; embModel: string; sortformerBin: string; sortformerModel: string; concurrency: number; maxAudioHours: number; }; export const DIARIZATION_SETTINGS_FIELD_DOCS: FieldDocs = { enabled: "Master switch. OFF by default so a transcription batch can start " + "before this lands, with diarization backfilled over the retained audio" + " afterwards.\n\n" + "Turning it ON also arms the cleanup guard: the Clean-audio sweep stops" + " deleting audio for a transcribed video that has no diarization.json " + "yet. That is the point — it is what keeps the perishable input alive " + "long enough to be captured — but it means enabling this holds disk.", inlineAfterTranscribe: "Run diarization inline in the post-transcribe hook.\n\n" + "OFF by default, and that default is a MEASURED decision, not caution. " + "Measured on this box: GPU transcription runs at 221 s/audio-hour " + "(16.3x realtime, over 3,602 real videos), CPU diarization at ~500-680 " + "s/audio-hour. Diarization is therefore ~2-3x SLOWER than the " + "transcription it follows, so running it inline drops whole-pipeline " + "throughput by roughly 3-4x and leaves the GPU idle while the CPU " + "catches up.\n\n" + "The intended sequence for a large batch is the opposite: leave this " + "off, let the batch transcribe at full GPU speed with `enabled` holding" + " the audio, and diarize afterwards with the backfill pass. Turn it on " + "for steady state, once the arrival rate is a few videos a day rather " + "than a corpus.", threshold: "Clustering threshold — the single most consequential knob, since it " + "decides how many speakers come out. Larger merges more aggressively.\n\n" + "The default is 0.9, NOT sherpa-onnx's own 0.5, and that is measured on" + " this corpus. On a 6-minute excerpt of a two-person interview (known " + "ground truth: 2 speakers), sherpa's default produced 22 clusters; 0.9 " + "produced 6, with the top two at 40%/40% of talk time — recognizably " + "the two hosts. Sweep on the same clip: 0.4→23, 0.5→22, 0.6→17, 0.7→12," + " 0.8→10, 0.9→6.\n\n" + "It still over-splits, which is why this is a capture lane and not an " + "answer: the turns are recorded with the threshold that produced them, " + "so a later attribution pass can re-cluster or re-run without needing " + "the audio back.", threads: "Engine threads per diarize run.", engine: "Which engine runs. \"sherpa-onnx\" is the shipped default and what every" + " sidecar on disk was produced by; \"sortformer\" is the ggml engine " + "built by scripts/build-sortformer.sh.\n\n" + "CHANGING THIS RESTATES THE FRESHNESS IDENTITY (see diarizationTarget)," + " so every sidecar written by the other engine becomes stale and the " + "backfill lane offers to redo it. That is intended — the two disagree " + "about how many speakers exist, and a corpus half-diarized by each is " + "not one corpus — but on the retained audio it is weeks of work, not a " + "toggle.\n\n" + "Why anyone would: on the same file, sherpa at its tuned threshold " + "returns 13 speakers and sortformer returns 4, agreeing on the dominant" + " speaker's share to within half a point (73.1% vs 73.5%). On the " + "corpus's worst case sherpa returns 35 and sortformer 4. Over-splitting" + " is the failure mode this lane has always had, and sortformer is end-" + "to-end rather than clustered, so it does not have it. The cost is a " + "hard ceiling of 4 speakers and ~1.8x the wall clock.", backend: "Compute device for the sortformer engine; ignored by sherpa-onnx, " + "which has no Vulkan compute path on Linux.\n\n" + "\"vulkan\" is 1.5x faster than a thread-tuned CPU run (894 vs 1305 " + "s/audio-hour, measured on this box) and holds 558 MB resident instead " + "of 4.84 GB by keeping weights and activations in VRAM. It also takes " + "~4.4 GB of an 8 GB card, which is why the lane YIELDS to transcription" + " rather than sharing — see controller/digestYield.ts.", python: "Python interpreter for the default sherpa-onnx engine. sherpa-onnx " + "ships wheels only up to cp313, and this box's system python is 3.14 — " + "so this usually points at a dedicated venv rather than `python3`.", segModel: "ONNX model paths for the default engine. Empty = the lane cannot run, " + "which is reported as a skip rather than a failure.", embModel: "ONNX speaker-embedding model path for the sherpa-onnx engine. Empty = the lane cannot run, reported as a `not-configured` skip rather than a failure (same as `segModel`).", sortformerBin: "Binary and model for the sortformer engine, both produced by " + "scripts/build-sortformer.sh. Empty = that engine cannot run, reported " + "as the same \"not-configured\" skip as an unset segModel/embModel.", sortformerModel: "Model for the sortformer engine, produced by scripts/build-sortformer.sh. Empty = that engine cannot run, reported as the same `not-configured` skip as an unset `sortformerBin`.", concurrency: "How many diarize runs may execute at once in the backfill pass. Kept " + "low by default: diarization is CPU-bound and competes with GPU feeding" + " and the digest sweep for the same 8 threads.", maxAudioHours: "Videos longer than this are DEFERRED rather than diarized: reported as" + " a third number that is never summed into reachable work, so a capped " + "corpus can never read as finished.\n\n" + "THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering " + "holds a pairwise distance matrix over speech-segment embeddings — " + "O(n^2) in SEGMENT count — and speaker-turn density varies 40x across " + "this corpus (33-1364 turns/hour), so duration does not actually " + "predict the blowup: a sparse 7h42m video completed while a dense 6h12m" + " one was OOM-killed. Duration is merely the only predictor available " + "for free, from metadata already on disk, BEFORE spending 45 minutes to" + " find out. n^2 at 30k segments is 6.7 GiB and at 40k is 11.9 GiB, " + "which brackets the 10.6 GB and 9.6 GB peaks measured on this 16 GB " + "box.\n\n" + "0 disables the cap. That is where this goes once windowed diarization " + "lands: windowing divides per-window n by the window count, so the " + "matrix falls by its square, and the cap stops being needed rather than" + " being tuned.", }; // Configuration for the derived-corpus digest layer. Local-first by decision: // `remoteEnabled` gates the metered lane and defaults to false, so nothing here // can spend money until it is explicitly turned on. // Each field is documented in DIGEST_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). export type DigestSettings = { remoteEnabled: boolean; longTailSeconds: number; localAppId: string; remoteAppId: string; apps: Record; yieldToTranscription: boolean; yieldToCpuWorkers: boolean; spendCapUsd: number; sections: DigestSectionKind[]; timestampMode: DigestTimestampMode; promptVariant: string; }; export const DIGEST_SETTINGS_FIELD_DOCS: FieldDocs = { remoteEnabled: "Master switch for the metered (remote-api) lane. OFF by default — an " + "opt-in overflow for the long tail or a channel where local quality is " + "poor, never the default path.", longTailSeconds: "Videos longer than this are \"long tail\": 8.2% of the corpus by count, " + "46% of all transcript tokens. The batch's duration-aware ordering and " + "the optional remote overflow both key off it.", localAppId: "The engine each lane uses (ids from common/lib/digestApps.ts).", remoteAppId: "The engine the metered (remote-api) lane uses — an id from common/lib/digestApps.ts. Unknown ids degrade to the default app rather than failing.", apps: "Per-app config, keyed by app id — the same id-keyed sub-record shape " + "as transcriptionApps.", yieldToTranscription: "Yield the GPU to the transcription lane: while transcription is " + "working, the digest batch's limit() returns 0 and the pool idle-waits." + " ON by default, because `digest:local` is deliberately on a different " + "queue from TRANSCRIPTION_QUEUE and so would otherwise run ollama and " + "the transcription engine on the same 8 GB card. See " + "controller/digestYield.ts.", yieldToCpuWorkers: "Whether a busy worker pinned to `device: \"cpu\"` counts as GPU " + "contention.\n\n" + "OFF by default, which is the FIX for a real bug: the yield originally " + "tested only `kind === \"local\"`, so on a box with one GPU worker and " + "two CPU-pinned ones (this box, at parallelTranscriptions 2) the digest" + " lane stopped dead for transcription that competes for zero GPU " + "shaders.\n\n" + "Only an EXPLICIT \"cpu\" is treated as non-contending. A worker with no " + "device set is using the engine binary's own default, which may be the " + "GPU, so it still triggers the yield — the unknown case fails safe.\n\n" + "Composes with `yieldToTranscription`: that is the master switch, this " + "only narrows which workers it reacts to.", spendCapUsd: "Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. " + "Only ever consulted for a metered app.", sections: "Which sections a sweep generates.\n\n" + "Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, " + "and that is not a contradiction: a tag call sends the same transcript " + "as the chapter call before it, so it hits the engine's cached prefix " + "and pays essentially no prefill (+0.4 s across 4 extra calls, against " + "22.4 s for the first 4). All it pays is decode, and a tag list is ~30 " + "output tokens where a chapter list is ~200–290.\n\n" + "The corollary matters more than the number: run them in the SAME pass." + " Tags generated later, on their own, pay full prefill again — measured" + " at 44% of a whole chapters pass, i.e. 3–9× the marginal cost of just " + "including them now.", timestampMode: "How each chunk's transcript markers are numbered — see " + "DigestTimestampMode. Was a scored variable in the bake-off rather than" + " a pre-applied fix; the measurement is in and \"chunk-local\" is now the" + " shipped default.", promptVariant: "A free-text label for a non-default prompt shape, folded into the " + "recorded provenance by digestPromptVariant(). Setting it invalidates " + "every digest generated under a different label, which is exactly what " + "makes a bake-off round re-run its sample instead of skipping it as " + "fresh. Empty = default.", }; // The docker build runner (release 18: `settings.publish.runner: "docker"`, // `archilyzer publish build all --runner docker`) fans every stale site out in // containers (publish/stageBodies.ts, buildAllDocker) on a host whose engine // answers; `--runner auto` (`archilyzer build all`) does so when one answers and // builds serially otherwise. The default runner is "local": one site at a time, // a child of the editor. There is no mode switch: the `mode` key ("basic" | // "docker") was a label nothing read, and it was dropped on 2026-09-28 // (release 11, follow-up O6c). A settings.json // that still carries it loads, and loses it on the next save — the sanitizer // below builds its output from the three real fields only. // // Each field is documented in BUILD_PIPELINE_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). export type BuildPipelineSettings = { maxParallelBuilds: number; dockerImage: string; dockerfile: string; }; export const BUILD_PIPELINE_SETTINGS_FIELD_DOCS: FieldDocs = { maxParallelBuilds: "Cap on concurrent per-site container builds when the docker build runner" + " builds every site (`publish.runner: \"docker\"`, or" + " `archilyzer publish build all --runner docker`)." + " Clamped to [1, " + "BUILD_MAX_PARALLEL_MAX].", dockerImage: "Tag of the reusable build image (built once, reused for every site).", dockerfile: "Dockerfile path relative to the monorepo root, used to (re)build the " + "image.", }; // Each field is documented in ARCHIVE_ORG_SETTINGS_FIELD_DOCS below (rendered // into SETTINGS.md). The type and defaults live with the torrent code // (lib/archiveOrgTorrent.ts), which is pure. export type { ArchiveOrgFetchSettings }; export const ARCHIVE_ORG_SETTINGS_FIELD_DOCS: FieldDocs = { torrent: "Fetch an archive.org file over BitTorrent when the item's " + "`_archive.torrent` carries it and aria2c is installed " + "(ARIA2C_BIN). archive.org is the torrent's web seed, so what other peers " + "give never touches archive.org. False = always the direct download.", seedMinutes: "Minutes to seed the file after it is complete, as a good swarm citizen. " + "The import holds archive.org's queue while it seeds. 0 = do not seed. " + "Clamped to [0, 1440]; default 10.", seedRatio: "Stop seeding sooner once this much has been uploaded relative to the " + "file's size (aria2c --seed-ratio), whichever of the two comes first. " + "0 = no ratio limit. Clamped to [0, 100]; default 1.", stallMinutes: "No download progress for this long stops aria2c and the file is " + "downloaded directly instead. Clamped to [1, 120]; default 5.", maxPeers: "Most peers per torrent (aria2c --bt-max-peers). Clamped to [1, 500]; default 30.", maxDownloadKiBps: "Download rate cap for a torrent fetch, KiB/s (aria2c " + "--max-overall-download-limit). 0 = unlimited.", maxUploadKiBps: "Upload rate cap while downloading and seeding, KiB/s (aria2c " + "--max-overall-upload-limit). 0 = unlimited.", }; function clampNumber(value: unknown, fallback: number, min: number, max: number): number { const n = typeof value === "number" && Number.isFinite(value) ? value : fallback; return Math.min(max, Math.max(min, n)); } // Coerce a raw settings.archiveOrg value into clean ArchiveOrgFetchSettings. export function sanitizeArchiveOrg(value: unknown): ArchiveOrgFetchSettings { const d = DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS; const r = (value && typeof value === "object" ? value : {}) as Record; return { torrent: r.torrent !== false, seedMinutes: clampNumber(r.seedMinutes, d.seedMinutes, 0, 1440), seedRatio: clampNumber(r.seedRatio, d.seedRatio, 0, 100), stallMinutes: clampNumber(r.stallMinutes, d.stallMinutes, 1, 120), maxPeers: Math.floor(clampNumber(r.maxPeers, d.maxPeers, 1, 500)), maxDownloadKiBps: Math.floor(clampNumber(r.maxDownloadKiBps, d.maxDownloadKiBps, 0, 10_000_000)), maxUploadKiBps: Math.floor(clampNumber(r.maxUploadKiBps, d.maxUploadKiBps, 0, 10_000_000)), }; } // Each field is documented in SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). export type SavedVideoBackupSettings = { enabled: boolean; dest: string; intervalMinutes: number; }; export const SAVED_VIDEO_BACKUP_SETTINGS_FIELD_DOCS: FieldDocs = { enabled: "Master switch for the scheduled backup. A backup can still be run " + "manually when this is false, as long as a destination is set.", dest: "Destination root the store is mirrored into (a local path or any rsync" + " target). Empty disables both scheduled and manual backups.", intervalMinutes: "Cadence (minutes) for the scheduled backup when enabled. Clamped into " + "the sync-interval window; default daily.", }; // Each field is documented in SYNC_SCHEDULER_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). export type SyncSchedulerSettings = { enabled: boolean; defaultIntervalMinutes: number; maxConcurrentSyncs: number; quietHoursStart: number | null; quietHoursEnd: number | null; backoffBaseMinutes: number; backoffMaxMinutes: number; heartbeatSeconds: number; keepLatestCheckIntervalMinutes: number; fullSweepIntervalMinutes: number; fullSweepConfirmMaxSuspects: number; fullSweepShrinkGuardPercent: number; }; export const SYNC_SCHEDULER_SETTINGS_FIELD_DOCS: FieldDocs = { enabled: "Master switch. When false, a tick selects nothing (manual sync still " + "works).", defaultIntervalMinutes: "Fallback cadence (minutes) for channels with no per-channel override.", maxConcurrentSyncs: "Cap on sync jobs running/queued at once. A tick queues at most (cap - " + "currently-active) channels; the rest roll to the next tick. This is " + "also the stagger mechanism that keeps a big due-batch from hitting the" + " source all at once.", quietHoursStart: "Optional local-clock quiet window during which auto-sync is " + "suppressed. Both null = always allowed. The window may wrap past " + "midnight (e.g. start=22, end=6). Hours are [0,23]; the window is " + "[start, end).", quietHoursEnd: "End hour of the quiet window, [0,23], exclusive. See `quietHoursStart`: both must be valid hours or the window is cleared (null = always allowed).", backoffBaseMinutes: "Failure backoff bounds. After N consecutive failed scheduled syncs a " + "channel waits min(base * 2^(N-1), max) minutes before it's eligible " + "again.", backoffMaxMinutes: "Ceiling on the failure backoff (see `backoffBaseMinutes`): a channel waits min(base * 2^(N-1), max) minutes after N consecutive failures. Never below the base.", heartbeatSeconds: "Cadence (seconds) for the editor's in-process heartbeat — the internal" + " timer armed by the instrumentation hook (editor/instrumentation.ts) " + "that calls the scheduler tick directly, so no external cron is needed." + " 0 = off: rely on the external `pnpm sync:tick` heartbeat instead. Any" + " positive value is clamped to [SYNC_HEARTBEAT_MIN_SECONDS, " + "SYNC_HEARTBEAT_MAX_SECONDS]. The env var SYNC_HEARTBEAT_SECONDS " + "overrides this at runtime. See SCHEDULED_SYNC.md.", keepLatestCheckIntervalMinutes: "Cadence (minutes) for the scheduled keep-latest deletion check. For " + "each channel with ChannelConfig.keepLatest > 0, the tick re-probes the" + " kept window for source deletion (checkKeptDeletedAction) at most this" + " often and pins any gone videos as do-not-clean. Clamped into the " + "sync-interval window; default daily. The check shares the same " + "concurrency cap and quiet-hours window as scheduled syncs. See " + "editor/app/scheduler/runTick.ts.", fullSweepIntervalMinutes: "Default cadence (minutes) for the sync FULL SWEEP — the deep pass that" + " re-enumerates a channel's whole listing in one yt-dlp spawn, " + "refreshes the stored `playlist` file, and flags videos that have left " + "the listing into maybe-missing.json. Ordinary syncs stay on the cheap " + "newest-first paged walk; a sync only upgrades itself to a sweep when " + "this interval has elapsed since the channel's lastFullSweepAt. Per-" + "channel override: ChannelConfig.fullSweepIntervalMinutes. 0 = never " + "sweep. Default daily. See common/jobs/deepSync.ts.", fullSweepConfirmMaxSuspects: "Upper bound on how many maybe-missing suspects a full sweep will " + "resolve in-line with the per-video availability probe (deleted vs " + "private vs unlisted). At or under the cap the sweep runs the targeted " + "check itself, so \"Sync all\" surfaces upstream deletions with no extra " + "clicks; over it, the suspects are flagged and left for a manual check " + "rather than firing hundreds of probes inside a sync. 0 = never auto-" + "confirm.", fullSweepShrinkGuardPercent: "Shrink guard: how far a fresh listing may fall below the stored one " + "before it is treated as suspect rather than acted on. Expressed as a " + "percentage of the previous count, floored at SHRINK_ABS_FLOOR entries " + "so ordinary churn on a small channel doesn't trip it. A suspect " + "listing does not rewrite `playlist` or maybe-missing.json and does not" + " count as a sweep — but a SECOND enumeration reporting a similar count" + " confirms it and is accepted, so a genuine mass deletion costs at most" + " one cadence period. 0 = off (the empty-listing rejection still " + "applies). See controller/acceptListing.ts.", }; // Each field is documented in SOCIAL_LINK_FIELD_DOCS below (rendered into SETTINGS.md). export type SocialLink = { label: string; url: string; svg: string; // Stored only when true. What a header does with it: lib/socialLinks.ts. featured?: boolean; }; export const SOCIAL_LINK_FIELD_DOCS: FieldDocs = { label: "The link's name: the icon's accessible name and its tooltip, never " + "text beside it. Shown as text only in place of an icon that fails the " + "check at render.", url: "Link target: http(s), mailto: or a site-relative path.", svg: "Inline SVG markup: ONE well-formed `` element, checked when it is " + "saved new or edited and again every time it is rendered (a link whose " + "icon fails at render shows its label instead; `archilyzer doctor` names " + "it). It may contain shapes, groups, defs, gradients, patterns, clip " + "paths, masks, filters, text and animate/animateTransform/set — no " + "script, style block, foreignObject, a, image, title, desc or any HTML " + "element (a title or desc holding text only is removed); SVG " + "presentation attributes plus aria-*, data-* and xmlns:* — no event " + "handler (on…); a `style` attribute of presentation properties only; an " + "href or url(…) only to an id inside the icon, written plainly; no CSS " + "escape, comment or function that loads anything (image-set, image, " + "cross-fade, element, src, paint, @import); ids plain names. A leading " + "XML declaration, a DOCTYPE without an internal subset and comments are " + "removed. Normalized on save: width/height stripped, aria-hidden added, " + "a single-colour icon's fills made fill=\"currentColor\" (an icon of two " + "or more colours keeps them). A root with no viewBox but a numeric width " + "W and height H (unitless or px) is given `viewBox=\"0 0 W H\"`, so a " + "file pasted as downloaded is accepted. A refused save names why; export " + "from a drawing program with presentation attributes rather than a style " + "block (in Inkscape, save as Plain SVG).", featured: "Keep this link in the header on small screens (the editor's \"Keep in " + "header on small screens\"). A narrow header shows only the featured " + "links (up to 4, the last 4 if more are marked; none marked → none, so " + "the name has the room); a wide header shows every link, up to 4, the " + "featured ones kept first, then the last of the rest. The footer shows " + "every link. Written only when true.", }; // Each field is documented in ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). export type ArchiveStorageSettings = { bucket: string; publicBaseUrl: string; }; export const ARCHIVE_STORAGE_SETTINGS_FIELD_DOCS: FieldDocs = { bucket: "Cloudflare R2 bucket an oversize archive zip is uploaded to on deploy " + "(`wrangler r2 object put`, keyed `/archives/`). Blank = " + "no overflow.", publicBaseUrl: "Public base URL of that bucket; the Downloads page links " + "`/`. Both fields must be set for overflow to " + "happen.", }; // Each field is documented in PACING_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md). export type PacingSettingsBlock = { sleepRequestsCapSeconds: number; decayAfterCleanUnits: number; holdAfterFailsAtCap: number; holdProbeMinutes: number; }; export const PACING_SETTINGS_FIELD_DOCS: FieldDocs = { sleepRequestsCapSeconds: "Ceiling (seconds) on a platform's adaptive `--sleep-requests`. The pace " + "starts at the platform's fixed value (1 s for YouTube and Rumble) and " + "doubles on every platform-level rate limit until it reaches this. A " + "subtitle-only 429 never moves it. Clamped 1–120; default 16.", decayAfterCleanUnits: "How many clean auto-download units on a platform halve its pace one step " + "back toward the fixed value. Clamped 1–1000; default 5.", holdAfterFailsAtCap: "How many consecutive failures AT the 30-minute cooldown cap put a " + "platform in a hold: the lane then runs one probe unit per " + "holdProbeMinutes instead of one per cooldown, and a manual Sync or " + "download on it is refused with the next probe's time. A clean probe " + "clears the hold and the backoff. Clamped 1–100; default 3.", holdProbeMinutes: "Minutes between probes while a platform is held. Clamped 1–1440; " + "default 60.", }; export const PACING_DEFAULT_SLEEP_REQUESTS_CAP_SECONDS = 16; export const PACING_DEFAULT_DECAY_AFTER_CLEAN_UNITS = 5; export const PACING_DEFAULT_HOLD_AFTER_FAILS_AT_CAP = 3; export const PACING_DEFAULT_HOLD_PROBE_MINUTES = 60; function clampInt(v: unknown, min: number, max: number, dflt: number): number { if (typeof v !== "number" || !Number.isFinite(v)) return dflt; return Math.min(max, Math.max(min, Math.floor(v))); } export function sanitizePacing(v: unknown): PacingSettingsBlock { const r = (v && typeof v === "object" ? v : {}) as Record; return { sleepRequestsCapSeconds: clampInt( r.sleepRequestsCapSeconds, 1, 120, PACING_DEFAULT_SLEEP_REQUESTS_CAP_SECONDS, ), decayAfterCleanUnits: clampInt( r.decayAfterCleanUnits, 1, 1000, PACING_DEFAULT_DECAY_AFTER_CLEAN_UNITS, ), holdAfterFailsAtCap: clampInt( r.holdAfterFailsAtCap, 1, 100, PACING_DEFAULT_HOLD_AFTER_FAILS_AT_CAP, ), holdProbeMinutes: clampInt( r.holdProbeMinutes, 1, 1440, PACING_DEFAULT_HOLD_PROBE_MINUTES, ), }; } export const SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS = 600; export const SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS = 10; export const MIN_FREE_DISK_GB_DEFAULT = 5; export const MIN_FREE_DISK_GB_MAX = 100000; // Hysteresis margin for the low-disk gate. 2 GB is deliberately larger than any // single scratch file the pipeline writes, so cleaning one up cannot by itself // reopen the gate. export const RESUME_MARGIN_GB_DEFAULT = 2; export const RESUME_MARGIN_GB_MAX = 1000; export const PARALLEL_TRANSCRIPTIONS_MAX = 16; export const PARALLEL_TRANSCRIPTIONS_DEFAULT = 2; // Passive UI auto-refresh cadence. 0 is a sentinel meaning "disabled"; any other // value is clamped into [MIN, MAX] seconds. export const AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS = 5; export const AUTO_REFRESH_INTERVAL_MIN_SECONDS = 1; export const AUTO_REFRESH_INTERVAL_MAX_SECONDS = 600; // Global snapshot-scheduler debounce presets. `debounceMs` is the quiet-period // window after the last report-changing action; `maxWaitMs` caps the total // delay under continuous activity (null = no cap, fire purely on the quiet // period). Consumed by common/jobs/snapshotScheduler.ts and surfaced in the // Settings form. export type ReportDebouncePreset = "fast" | "balanced" | "lazy"; export const REPORT_DEBOUNCE_PRESETS: Record< ReportDebouncePreset, { debounceMs: number; maxWaitMs: number | null } > = { fast: { debounceMs: 1000, maxWaitMs: null }, balanced: { debounceMs: 3000, maxWaitMs: 30000 }, lazy: { debounceMs: 10000, maxWaitMs: 60000 }, }; export const DEFAULT_REPORT_DEBOUNCE_PRESET: ReportDebouncePreset = "fast"; export function isReportDebouncePreset(v: unknown): v is ReportDebouncePreset { return v === "fast" || v === "balanced" || v === "lazy"; } export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024; export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024; export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024; export const DEFAULT_ADMIN_TITLE = "Transcript Browser Admin"; // Sync-scheduler bounds + defaults. Default cadence is daily; concurrency is // conservative so a tick doesn't fan out into the source provider all at once. export const SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES = 1440; export const SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT = 2; export const SYNC_SCHEDULER_MAX_CONCURRENT_MAX = 16; export const SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES = 30; export const SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES = 1440; export const KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES = 1440; // Full-sweep defaults. Daily: a sweep is one full enumeration of the channel, // far more expensive than the 50-entry page an ordinary sync fetches. The // confirm cap keeps an unattended sweep from fanning out into hundreds of // per-video probes when a channel's listing changes wholesale. export const FULL_SWEEP_DEFAULT_INTERVAL_MINUTES = 1440; export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT = 25; export const FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX = 10000; // Shrink-guard default: a listing that has lost more than a tenth of its // entries (and more than SHRINK_ABS_FLOOR of them) needs a second opinion. export const FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT = 10; export const FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX = 100; export const SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES = 1440; // Internal-heartbeat cadence bounds. 0 means "off" (use an external cron // heartbeat); any other value is clamped into [MIN, MAX] seconds. The floor // keeps the in-process timer from busy-looping; the ceiling is one hour. export const SYNC_HEARTBEAT_DEFAULT_SECONDS = 0; export const SYNC_HEARTBEAT_MIN_SECONDS = 15; export const SYNC_HEARTBEAT_MAX_SECONDS = 3600; export function defaultSyncScheduler(): SyncSchedulerSettings { return { enabled: false, defaultIntervalMinutes: SYNC_SCHEDULER_DEFAULT_INTERVAL_MINUTES, maxConcurrentSyncs: SYNC_SCHEDULER_MAX_CONCURRENT_DEFAULT, quietHoursStart: null, quietHoursEnd: null, backoffBaseMinutes: SYNC_SCHEDULER_BACKOFF_BASE_DEFAULT_MINUTES, backoffMaxMinutes: SYNC_SCHEDULER_BACKOFF_MAX_DEFAULT_MINUTES, heartbeatSeconds: SYNC_HEARTBEAT_DEFAULT_SECONDS, keepLatestCheckIntervalMinutes: KEEP_LATEST_CHECK_DEFAULT_INTERVAL_MINUTES, fullSweepIntervalMinutes: FULL_SWEEP_DEFAULT_INTERVAL_MINUTES, fullSweepConfirmMaxSuspects: FULL_SWEEP_CONFIRM_MAX_SUSPECTS_DEFAULT, fullSweepShrinkGuardPercent: FULL_SWEEP_SHRINK_GUARD_PERCENT_DEFAULT, }; } // Clamp an internal-heartbeat cadence: 0 (off) passes through; any positive // value is clamped up into [MIN, MAX]; junk falls back to the default. export function clampHeartbeatSeconds(value: unknown): number { if (typeof value !== "number" || !Number.isFinite(value)) { return SYNC_HEARTBEAT_DEFAULT_SECONDS; } const n = Math.floor(value); if (n <= 0) return 0; if (n < SYNC_HEARTBEAT_MIN_SECONDS) return SYNC_HEARTBEAT_MIN_SECONDS; if (n > SYNC_HEARTBEAT_MAX_SECONDS) return SYNC_HEARTBEAT_MAX_SECONDS; return n; } function clampHourOrNull(value: unknown): number | null { if (typeof value !== "number" || !Number.isFinite(value)) return null; const n = Math.floor(value); if (n < 0 || n > 23) return null; return n; } // Like clampPositiveInt, but 0 survives as a sentinel ("off"/"never"). Used by // the cadences whose disabled state is expressed as a zero rather than a // separate boolean. function clampIntAllowZero(value: unknown, fallback: number, max: number): number { const n = typeof value === "number" && Number.isFinite(value) ? Math.floor(value) : fallback; if (n <= 0) return 0; if (n > max) return max; return n; } function clampPositiveInt(value: unknown, fallback: number, max: number): number { const n = typeof value === "number" && Number.isFinite(value) ? Math.floor(value) : fallback; if (n < 1) return 1; if (n > max) return max; return n; } // Coerce a raw settings.syncScheduler value into a clean SyncSchedulerSettings, // falling back to defaults for missing/ill-typed fields. Quiet hours are only // honored when BOTH endpoints are valid hours; otherwise the window is cleared. export function sanitizeSyncScheduler(value: unknown): SyncSchedulerSettings { const d = defaultSyncScheduler(); if (!value || typeof value !== "object") return d; const r = value as Record; const start = clampHourOrNull(r.quietHoursStart); const end = clampHourOrNull(r.quietHoursEnd); const backoffBase = clampPositiveInt( r.backoffBaseMinutes, d.backoffBaseMinutes, SYNC_INTERVAL_MAX_MINUTES, ); return { enabled: r.enabled === true, defaultIntervalMinutes: clampPositiveInt( r.defaultIntervalMinutes, d.defaultIntervalMinutes, SYNC_INTERVAL_MAX_MINUTES, ), maxConcurrentSyncs: clampPositiveInt( r.maxConcurrentSyncs, d.maxConcurrentSyncs, SYNC_SCHEDULER_MAX_CONCURRENT_MAX, ), quietHoursStart: start !== null && end !== null ? start : null, quietHoursEnd: start !== null && end !== null ? end : null, backoffBaseMinutes: backoffBase, // Cap can't sit below the base, or backoff would never grow. backoffMaxMinutes: Math.max( backoffBase, clampPositiveInt( r.backoffMaxMinutes, d.backoffMaxMinutes, SYNC_INTERVAL_MAX_MINUTES, ), ), heartbeatSeconds: clampHeartbeatSeconds(r.heartbeatSeconds), keepLatestCheckIntervalMinutes: clampPositiveInt( r.keepLatestCheckIntervalMinutes, d.keepLatestCheckIntervalMinutes, SYNC_INTERVAL_MAX_MINUTES, ), fullSweepIntervalMinutes: clampIntAllowZero( r.fullSweepIntervalMinutes, d.fullSweepIntervalMinutes, SYNC_INTERVAL_MAX_MINUTES, ), fullSweepConfirmMaxSuspects: clampIntAllowZero( r.fullSweepConfirmMaxSuspects, d.fullSweepConfirmMaxSuspects, FULL_SWEEP_CONFIRM_MAX_SUSPECTS_MAX, ), fullSweepShrinkGuardPercent: clampIntAllowZero( r.fullSweepShrinkGuardPercent, d.fullSweepShrinkGuardPercent, FULL_SWEEP_SHRINK_GUARD_PERCENT_MAX, ), }; } export function defaultSavedVideoBackup(): SavedVideoBackupSettings { return { enabled: false, dest: "", intervalMinutes: SAVED_VIDEO_BACKUP_DEFAULT_INTERVAL_MINUTES, }; } // Coerce a raw settings.savedVideoBackup value into a clean // SavedVideoBackupSettings. A missing destination forces enabled off, since a // backup with nowhere to go is meaningless. export function sanitizeSavedVideoBackup( value: unknown, ): SavedVideoBackupSettings { const d = defaultSavedVideoBackup(); if (!value || typeof value !== "object") return d; const r = value as Record; const dest = typeof r.dest === "string" ? r.dest.trim() : ""; return { enabled: dest !== "" && r.enabled === true, dest, intervalMinutes: clampPositiveInt( r.intervalMinutes, d.intervalMinutes, SYNC_INTERVAL_MAX_MINUTES, ), }; } // Where relocated channel media goes: the named locations. // // This used to be ONE FIELD, `mediaRoot` — a single absolute string, the cold // drive, typed once. It grew into a list of entities because a root alone // cannot answer the two questions the operator actually has: is that disk here, // and if it came up somewhere else, how do I point the channels at it without // ssh and hand edits? A location carries an id, a label, the root, an opt-in // `autoRepoint`, and the volume identity learned at its last probe. // // Still NOT a policy: a channel on a location is not thereby deprioritized, and // nothing auto-relocates anything because a location exists. // // AVAILABILITY IS NEVER STORED HERE. A refresh that wrote "available" would // rewrite settings.json — and so bump the pulse revision — every few seconds. // The probe (common/lib/storageVolumes.ts) is computed per request; only the // `volume` identity is ever written back, and only when it changed. // // The types live in lib/storageLocations.ts, which is pure: a `"use client"` // file may import them, and must not reach storageVolumes.ts (execa). export type { StorageLocation, StorageVolume, StorageSettings }; export function defaultStorage(): StorageSettings { return { locations: [], defaultLocationId: "" }; } const LOCATION_ID_RE = /^[a-z0-9][a-z0-9-]{0,63}$/; // "internal" IS TAKEN. It is the synthetic /storage row for the corpus volume // (INTERNAL_LOCATION_ID), and the regex above admits it — so a hand-edited // settings.json, or an operator typing the obvious word into the New location // form, could store a real location under the one id the page assembles for // itself. The row would then be built twice, the rollup would count channels // into whichever assembled last, and `locationOfDataDir` would start matching // unrelocated channels against it. function isReservedLocationId(id: string): boolean { return id === INTERNAL_LOCATION_ID; } function sanitizeVolume(value: unknown): StorageVolume | undefined { if (!value || typeof value !== "object") return undefined; const v = value as Record; const uuid = typeof v.uuid === "string" ? v.uuid.trim() : ""; const mountpoint = typeof v.mountpoint === "string" ? v.mountpoint.trim() : ""; // No uuid is no identity, and no mountpoint means `root === join(mountpoint, // relPath)` cannot hold — either way the record is not usable for finding the // volume again, so it is dropped rather than half-kept. if (!uuid || !mountpoint) return undefined; const relPath = typeof v.relPath === "string" ? v.relPath.trim() : ""; const fstype = typeof v.fstype === "string" ? v.fstype.trim() : ""; const label = typeof v.label === "string" ? v.label.trim() : ""; return { uuid, ...(fstype ? { fstype } : {}), ...(label ? { label } : {}), mountpoint, relPath, }; } // Coerce a raw settings.storage value into a clean StorageSettings. // // EXISTENCE IS NOT CHECKED, on purpose: the whole point of a cold location is // that it is a drive that may not be mounted when settings are read, and a // sanitizer that dropped the root on an unmounted platter would silently erase // the operator's choice on the next save. // // ABSOLUTENESS *IS* checked, and a location with a relative root is DROPPED // rather than resolved. Resolving it would anchor the location to whatever cwd // the reader booted in — a different directory under docker, under a worktree, // and under `pnpm dev` — so the same settings.json would name three different // drives. The location form rejects a relative path with a message before it // ever gets here; this is the last line, not the only one. // // NESTED ROOTS ARE ALLOWED. "/mnt/platter" and "/mnt/platter/archive" may both // be locations; `locationOfDataDir` resolves a channel to the LONGEST matching // root. Nothing here rejects the nesting, because the operator who arranges a // disk that way means it. // // A STALE `mediaRoot` SITTING BESIDE `locations` IS IGNORED — it is not merged // back in as an extra location. `migrateMediaRootToLocations` reads it exactly // once, when `locations` is absent; after that the list is the whole truth, and // resurrecting a root the operator deleted would be a bug, not a kindness. // // ROLLBACK: an older binary sanitizes this block to `{ mediaRoot: "" }` — the // locations are dropped and the single cold root comes back blank. One string // lost, nothing on disk moved. `cp settings.json settings.json.pre-storage- // locations` before the upgrade and a downgrade is a file copy. export function sanitizeStorage(value: unknown): StorageSettings { const d = defaultStorage(); if (!value || typeof value !== "object") return d; const r = value as Record; const rawList = Array.isArray(r.locations) ? r.locations : []; const locations: StorageLocation[] = []; const seen = new Set(); for (const entry of rawList) { if (!entry || typeof entry !== "object") continue; const e = entry as Record; const id = typeof e.id === "string" ? e.id.trim() : ""; if (!LOCATION_ID_RE.test(id) || isReservedLocationId(id) || seen.has(id)) { continue; } const rawRoot = typeof e.root === "string" ? e.root.trim() : ""; if (!path.isAbsolute(rawRoot)) continue; // "/mnt/platter/" and "/mnt/platter" are one root; "/" stays "/". const stripped = rawRoot.replace(/\/+$/, ""); const root = stripped === "" ? "/" : stripped; const label = typeof e.label === "string" ? e.label.trim() : ""; const volume = sanitizeVolume(e.volume); seen.add(id); locations.push({ id, label: label || id, root, autoRepoint: e.autoRepoint === true, ...(volume ? { volume } : {}), }); } const wanted = typeof r.defaultLocationId === "string" ? r.defaultLocationId.trim() : ""; // A default naming a location that is gone falls back to the first one, not // to "": with a location configured, "no default" is never the answer the // operator wanted, and a blank default silently disables every prefill. const defaultLocationId = locations.some((l) => l.id === wanted) ? wanted : (locations[0]?.id ?? ""); // THE SAVED-VIDEO STORE'S LOCATION IS NOT FALLEN BACK, and the asymmetry // with `defaultLocationId` above is deliberate. That one is a PREFERENCE, so // picking another location when the named one is gone is helpful. This one is // a RECORD OF WHERE BYTES ARE: pointing it at a different location because // the recorded one was deleted would claim the store had moved when nothing // had. A dangling id sanitizes to "" — "in place" — which is what the disk // says as soon as anybody looks, and the symlink (if any) keeps working // regardless, because the store is reached through it and not through this. const savedWanted = typeof r.savedVideosLocationId === "string" ? r.savedVideosLocationId.trim() : ""; const savedVideosLocationId = locations.some((l) => l.id === savedWanted) ? savedWanted : ""; // THE DRIVE-HEALTH TIMINGS: each clamped into its range, and kept only where // it differs from its default; a block with nothing left is not written // (lib/storageHealthTimings.ts). const health = sanitizeStorageHealth(r.health); return { locations, defaultLocationId, ...(savedVideosLocationId ? { savedVideosLocationId } : {}), ...(Object.keys(health).length > 0 ? { health } : {}), }; } // 4 hours. Measured: videos over this are 8.2% of the corpus by count but hold // 46% of all transcript tokens, so they are where a sweep's wall-clock actually // goes and where chunk-seam bugs live. export const DIGEST_LONG_TAIL_DEFAULT_SECONDS = 4 * 3600; export const DIGEST_LONG_TAIL_MAX_SECONDS = 24 * 3600; export function defaultDigest(): DigestSettings { return { // OFF. The metered lane is built but never the default — see PLAN.md. remoteEnabled: false, longTailSeconds: DIGEST_LONG_TAIL_DEFAULT_SECONDS, localAppId: DEFAULT_DIGEST_APP_ID, remoteAppId: CLAUDE_DIGEST_APP_ID, // Empty on purpose: every per-app knob falls through to its own default // constant (resolveNumCtx -> DEFAULT_DIGEST_NUM_CTX, now 8192, and // maxCuesForContext sizes the chunk to it). Seeding a copy of those values // here would give the same number two homes and let them drift. apps: {}, // ON. Real GPU contention with the transcription engine is a genuine cost // (re-priced: 11.2 s/chunk idle against 24.9 s/chunk on a contended box), so // the safe default is to step aside; turning it off is the deliberate choice. yieldToTranscription: true, // OFF. A CPU-pinned worker is not GPU contention, and treating it as such // stalled the digest lane for nothing. See DigestSettings.yieldToCpuWorkers. yieldToCpuWorkers: false, spendCapUsd: 0, sections: ["chapters"], timestampMode: DEFAULT_DIGEST_TIMESTAMP_MODE, promptVariant: "", }; } // Coerce a raw settings.digest.apps value into a clean keyed map of // DigestAppConfig. Mirrors sanitizeTranscriptionApps — INCLUDING its // Array.isArray guard, without which a JSON array would pass the typeof check and // produce numeric-keyed garbage. export function sanitizeDigestApps( value: unknown, ): Record { if (!value || typeof value !== "object" || Array.isArray(value)) return {}; const out: Record = {}; for (const [id, raw] of Object.entries(value as Record)) { if (!raw || typeof raw !== "object") continue; const r = raw as Record; const cfg: DigestAppConfig = {}; if (typeof r.bin === "string" && r.bin.trim()) cfg.bin = r.bin.trim(); if (typeof r.baseUrl === "string" && r.baseUrl.trim()) { cfg.baseUrl = r.baseUrl.trim(); } if (typeof r.model === "string" && r.model.trim()) cfg.model = r.model.trim(); if (typeof r.numCtx === "number" && r.numCtx > 0) { cfg.numCtx = Math.floor(r.numCtx); } if (typeof r.temperature === "number" && r.temperature >= 0) { cfg.temperature = r.temperature; } if (typeof r.timeoutMs === "number" && r.timeoutMs > 0) { cfg.timeoutMs = Math.floor(r.timeoutMs); } // Only carried when explicitly set — see DigestAppConfig.think. if (typeof r.think === "boolean") cfg.think = r.think; out[id] = cfg; } return out; } export function sanitizeDigest(value: unknown): DigestSettings { const d = defaultDigest(); if (!value || typeof value !== "object") return d; const r = value as Record; const sections = Array.isArray(r.sections) ? (r.sections.filter(isDigestSectionKind) as DigestSectionKind[]) : []; return { remoteEnabled: r.remoteEnabled === true, longTailSeconds: clampPositiveInt( r.longTailSeconds, d.longTailSeconds, DIGEST_LONG_TAIL_MAX_SECONDS, ), // Unknown app ids are not rejected here: getDigestApp() is total and falls // back to the local default, so a stale id degrades rather than breaking. localAppId: typeof r.localAppId === "string" && r.localAppId.trim() ? r.localAppId.trim() : d.localAppId, remoteAppId: typeof r.remoteAppId === "string" && r.remoteAppId.trim() ? r.remoteAppId.trim() : d.remoteAppId, apps: sanitizeDigestApps(r.apps), // Defaults to ON when absent — `=== false` rather than `!== true`, so a // settings file written before this field existed keeps the GPU-safe // behaviour instead of silently opting into contention. yieldToTranscription: r.yieldToTranscription !== false, // The OPPOSITE idiom, and deliberately so: `=== true`, so absence falls to // OFF. The field's absence means a settings file written before the CPU-worker // bug was found, and for those files OFF is the FIXED behaviour, not a silent // change of intent — nobody ever asked to stall the digest lane for a CPU // transcription. `yieldToTranscription` still gates the whole thing, so the // GPU-safe default is untouched. yieldToCpuWorkers: r.yieldToCpuWorkers === true, spendCapUsd: typeof r.spendCapUsd === "number" && r.spendCapUsd > 0 ? Math.round(r.spendCapUsd * 100) / 100 : 0, // An empty/garbage list would silently generate nothing, so fall back to the // default rather than honoring it. sections: sections.length > 0 ? sections : d.sections, timestampMode: isDigestTimestampMode(r.timestampMode) ? r.timestampMode : d.timestampMode, // Trimmed and length-capped: it goes into provenance on every record, and a // runaway value would bloat 119k sidecars. promptVariant: typeof r.promptVariant === "string" ? r.promptVariant.trim().slice(0, 40) : d.promptVariant, }; } // Every known section kind, for the settings UI's checkbox list. export const DIGEST_SECTION_OPTIONS = DIGEST_SECTION_KINDS; export const DIGEST_TIMESTAMP_MODE_OPTIONS = DIGEST_TIMESTAMP_MODES; export const BUILD_MAX_PARALLEL_DEFAULT = 2; export const BUILD_MAX_PARALLEL_MAX = 16; export const DEFAULT_BUILD_IMAGE = "yt-dlp-transcript-browser-build"; export const DEFAULT_BUILD_DOCKERFILE = "Dockerfile.build"; export function defaultBuildPipeline(): BuildPipelineSettings { return { maxParallelBuilds: BUILD_MAX_PARALLEL_DEFAULT, dockerImage: DEFAULT_BUILD_IMAGE, dockerfile: DEFAULT_BUILD_DOCKERFILE, }; } // Coerce a raw settings.buildPipeline value into a clean BuildPipelineSettings, // falling back to defaults for missing/ill-typed fields. Built from the three // known fields only, so a retired key (the old `mode`) never survives. export function sanitizeBuildPipeline(value: unknown): BuildPipelineSettings { const d = defaultBuildPipeline(); if (!value || typeof value !== "object") return d; const r = value as Record; const dockerImage = typeof r.dockerImage === "string" && r.dockerImage.trim() ? r.dockerImage.trim() : d.dockerImage; const dockerfile = typeof r.dockerfile === "string" && r.dockerfile.trim() ? r.dockerfile.trim() : d.dockerfile; return { maxParallelBuilds: clampPositiveInt( r.maxParallelBuilds, d.maxParallelBuilds, BUILD_MAX_PARALLEL_MAX, ), dockerImage, dockerfile, }; } export function defaultBackfill(): BackfillSettings { return { concurrency: 1, // See BackfillSettings.allowRedownload — this one holds disk. allowRedownload: false, }; } export function sanitizeBackfill(value: unknown): BackfillSettings { const d = defaultBackfill(); if (!value || typeof value !== "object") return d; const r = value as Record; return { // Clamped rather than rejected: a hand-edited 5 means "as much as possible", // and reading it as 0 would be the opposite of the intent. concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16), allowRedownload: r.allowRedownload === true, }; } export function defaultAttribution(): AttributionSettings { return { // OFF, and both lanes OFF under it. See AttributionSettings. enabled: false, appId: DEFAULT_DIGEST_APP_ID, model: "", diarizedEnabled: false, textOnlyEnabled: false, promptVersion: ATTRIBUTION_PROMPT_VERSION, }; } export function sanitizeAttribution(value: unknown): AttributionSettings { const d = defaultAttribution(); if (!value || typeof value !== "object") return d; const r = value as Record; const str = (v: unknown, fallback: string) => typeof v === "string" && v.trim() ? v.trim() : fallback; return { enabled: r.enabled === true, appId: str(r.appId, d.appId), // NOT `str(...)` with a fallback: empty is a meaningful value here ("use the // app's own model"), so an empty string must survive rather than reverting // to a default that is also empty by coincidence. model: typeof r.model === "string" ? r.model.trim() : d.model, diarizedEnabled: r.diarizedEnabled === true, textOnlyEnabled: r.textOnlyEnabled === true, // FLOORED at the shipped constant, never merely defaulted. A hand-edited // value below it would pin freshness to a superseded prompt generation and // freeze its output into the corpus — see AttributionSettings.promptVersion. promptVersion: typeof r.promptVersion === "number" && Number.isFinite(r.promptVersion) ? Math.max(ATTRIBUTION_PROMPT_VERSION, Math.floor(r.promptVersion)) : d.promptVersion, }; } export function defaultDiarization(): DiarizationSettings { return { // OFF. Capture is opt-in: turning it on makes the cleanup sweep start // refusing to delete audio for transcribed-but-undiarized videos, which is // correct but is a disk-pressure decision an operator should make. enabled: false, // OFF — see DiarizationSettings.inlineAfterTranscribe. Diarization is slower // than the transcription it would follow, so inline is the exception. inlineAfterTranscribe: false, // Measured, not sherpa-onnx's 0.5 — see DiarizationSettings.threshold. The // constant lives in lib/diarization.ts because isDiarizationFresh needs it // to normalize an absent recorded threshold; importing it keeps the default // and the comparator from drifting apart. threshold: DEFAULT_DIARIZATION_THRESHOLD, threads: 4, // The engine every sidecar on disk was produced by. Switching is an explicit // decision that restates the freshness identity — see DiarizationSettings. engine: DEFAULT_DIARIZATION_ENGINE, // Only consulted when engine is "sortformer". Defaulting to the GPU is safe // because the lane yields the card to transcription rather than sharing it. backend: "vulkan", python: "python3", segModel: "", embModel: "", sortformerBin: "", sortformerModel: "", concurrency: 1, // OFF, because windowing made it unnecessary — which is what it was always // for. It shipped at 4 hours as a stopgap while long recordings were being // OOM-killed; the engine now processes them in windows and the 6h12m file // that was being killed at 10.5 GB completes at ~2 GB. The knob stays, and // stays honest about what it does, for a machine smaller than this one or a // recording longer than anything measured here. maxAudioHours: 0, }; } export function sanitizeDiarization(value: unknown): DiarizationSettings { const d = defaultDiarization(); if (!value || typeof value !== "object") return d; const r = value as Record; const str = (v: unknown, fallback: string) => typeof v === "string" && v.trim() ? v.trim() : fallback; return { enabled: r.enabled === true, inlineAfterTranscribe: r.inlineAfterTranscribe === true, threshold: typeof r.threshold === "number" && Number.isFinite(r.threshold) && r.threshold > 0 ? r.threshold : d.threshold, threads: clampPositiveInt(r.threads, d.threads, 64), // An unknown engine falls back to the default rather than disabling the lane: // a typo in settings.json must not silently stop diarization, and the default // is the one every existing sidecar already matches. engine: DIARIZATION_ENGINE_IDS.includes(r.engine as DiarizationEngineId) ? (r.engine as DiarizationEngineId) : d.engine, backend: DIARIZATION_BACKENDS.includes(r.backend as DiarizationBackend) ? (r.backend as DiarizationBackend) : d.backend, python: str(r.python, d.python), segModel: str(r.segModel, d.segModel), embModel: str(r.embModel, d.embModel), sortformerBin: str(r.sortformerBin, d.sortformerBin), sortformerModel: str(r.sortformerModel, d.sortformerModel), concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16), // 0 is meaningful here (cap off), so this cannot use clampPositiveInt. // Fractional hours are allowed — the knob is a duration, not a count. maxAudioHours: typeof r.maxAudioHours === "number" && Number.isFinite(r.maxAudioHours) && r.maxAudioHours >= 0 ? r.maxAudioHours : d.maxAudioHours, }; } const SOCIAL_URL_RE = /^(https?:\/\/|mailto:|\/)/i; // Normalize the family hub URL into a trailing-slash-free absolute http(s) URL. // Returns "" for anything that isn't a usable absolute URL (the "no hub" state). // Kept local to settings.ts to avoid a settings<->homepage import cycle; mirrors // parseHomepageUrl() in homepage.ts. export function normalizeHomepageUrl(input: unknown): string { if (typeof input !== "string") return ""; const trimmed = input.trim().replace(/\/+$/, ""); return /^https?:\/\/\S+/i.test(trimmed) ? trimmed : ""; } export function parseSocialLinks(input: unknown): SocialLink[] { if (!Array.isArray(input)) return []; const out: SocialLink[] = []; for (const raw of input) { if (!raw || typeof raw !== "object") continue; const r = raw as Record; const label = typeof r.label === "string" ? r.label.trim() : ""; const url = typeof r.url === "string" ? r.url.trim() : ""; const svg = typeof r.svg === "string" ? r.svg : ""; if (!label || !url || !svg) continue; if (!SOCIAL_URL_RE.test(url)) continue; // `featured` only when it is exactly `true`: absent and false read the // same, and a file that never marked a link parses as it always did. out.push({ label, url, svg, ...(r.featured === true ? { featured: true } : {}) }); } return out; } // The icon's SVG — what it may contain, and its normalized form — is // lib/socialSvg.ts (pure, so the render path runs it too). Re-exported here for // the importers that reach it through lib/settings. export { normalizeSocialSvg, socialSvgProblem } from "./socialSvg"; // The render-time half (sizing, id scoping, the header's selection, the // render-time check) lives in lib/socialLinks.ts, which imports only the pure // socialSvg.ts, so a client tree can use it. Re-exported here for the importers // that reach it through lib/settings. export { sizeSocialSvg } from "./socialLinks"; export function clampSleepBetweenDownloadsSeconds(value: unknown): number { const n = typeof value === "number" && Number.isFinite(value) ? Math.floor(value) : SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS; if (n < 0) return 0; if (n > SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS) { return SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS; } return n; } export function clampMinFreeDiskGB(value: unknown): number { const n = typeof value === "number" && Number.isFinite(value) ? Math.floor(value) : MIN_FREE_DISK_GB_DEFAULT; if (n < 0) return 0; if (n > MIN_FREE_DISK_GB_MAX) return MIN_FREE_DISK_GB_MAX; return n; } export function clampResumeMarginGB(value: unknown): number { const n = typeof value === "number" && Number.isFinite(value) ? Math.floor(value) : RESUME_MARGIN_GB_DEFAULT; if (n < 0) return 0; if (n > RESUME_MARGIN_GB_MAX) return RESUME_MARGIN_GB_MAX; return n; } export function clampParallelTranscriptions(value: unknown): number { const n = typeof value === "number" && Number.isFinite(value) ? Math.floor(value) : PARALLEL_TRANSCRIPTIONS_DEFAULT; if (n < 1) return 1; if (n > PARALLEL_TRANSCRIPTIONS_MAX) return PARALLEL_TRANSCRIPTIONS_MAX; return n; } // 0 means "disabled" and is preserved as-is. Anything else is clamped into the // [MIN, MAX] window; a non-finite value falls back to the default cadence. export function clampAutoRefreshIntervalSeconds(value: unknown): number { if (typeof value !== "number" || !Number.isFinite(value)) { return AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS; } const n = Math.floor(value); if (n <= 0) return 0; if (n < AUTO_REFRESH_INTERVAL_MIN_SECONDS) { return AUTO_REFRESH_INTERVAL_MIN_SECONDS; } if (n > AUTO_REFRESH_INTERVAL_MAX_SECONDS) { return AUTO_REFRESH_INTERVAL_MAX_SECONDS; } return n; } export function clampPageBytes(value: unknown): number { const n = typeof value === "number" && Number.isFinite(value) ? value : TRANSCRIPT_PAGE_DEFAULT_BYTES; if (n < TRANSCRIPT_PAGE_MIN_BYTES) return TRANSCRIPT_PAGE_MIN_BYTES; if (n > TRANSCRIPT_PAGE_HARD_CAP_BYTES) return TRANSCRIPT_PAGE_HARD_CAP_BYTES; return Math.floor(n); } // --- The publish lane (release 18) ------------------------------------------ // What the lane (and Publish now) may do to one target when it is stale: // nothing, build it, build it and deploy it as a preview, or build it and // deploy it to production. A site carries its own in site.json // (`publish.auto`, lib/siteSchema.ts); the hub and the homepage carry theirs // here. A manual Build or Deploy button never asks it. export type PublishPolicy = "off" | "build" | "preview" | "production"; export const PUBLISH_POLICIES: readonly PublishPolicy[] = ["off", "build", "preview", "production"]; export function isPublishPolicy(v: unknown): v is PublishPolicy { return v === "off" || v === "build" || v === "preview" || v === "production"; } export type PublishQuietHours = { start: number; end: number }; // Each field is documented in PUBLISH_SETTINGS_FIELD_DOCS below. export type PublishSettings = { enabled: boolean; held: boolean; checkEveryMinutes: number; refreshEveryMinutes: number; quietHours: PublishQuietHours | null; runner: "local" | "docker"; previewBranch: string; hub: PublishPolicy; homepage: PublishPolicy; }; export const PUBLISH_SETTINGS_FIELD_DOCS: FieldDocs = { enabled: "Whether the publish lane's runner runs (`auto-publish` on /jobs). Off by default. Turning it on starts nothing by itself until the index is stale (or there is no index stamp yet) — see `refreshEveryMinutes`.", held: "The lane's pause gate (lib/pauseGates.ts, lane `publish`). A hold stops the runner DISPATCHING: the stage in flight finishes, no next one starts. It never kills a stage.", checkEveryMinutes: "How often (minutes) the runner wakes to ask whether a pass is due. Clamped to [1, 1440]; default 10.", refreshEveryMinutes: "The least time (minutes) between two index updates the lane starts: a pass runs when the index is stale and the last update is at least this old, or when there is no index stamp. Clamped to [0, 43200]; default 360. 0 = whenever the index is stale.", quietHours: "A local-clock window in which the lane starts no pass, `{ \"start\": 22, \"end\": 6 }` (hours [0,23], `[start, end)`, may wrap midnight), or null (default). A pass already running finishes its stage and then waits.", runner: "Which build runner the lane's builds ask for: \"local\" (default; each site in turn, as a child of the editor) or \"docker\" (every stale site in containers — a host with a container engine only; in a container it is refused). See PUBLISH.md.", previewBranch: "The Pages preview branch a `preview` policy deploys to (`wrangler pages deploy --branch `). Lowercase letters, digits and dashes, never `main`; an invalid name reads as the default, \"preview\".", hub: "The hub's policy: \"off\" (default), \"build\", \"preview\" or \"production\". The hub is built when the index or the listed sites changed, and deployed as the policy says — to production only with a Pages project in homepage.json.", homepage: "The homepage's policy, as `hub` (default \"off\"). Building the homepage also publishes the source mirror (the operator's scrub and denylist files must exist).", }; export const PUBLISH_CHECK_EVERY_DEFAULT_MINUTES = 10; export const PUBLISH_CHECK_EVERY_MAX_MINUTES = 1440; export const PUBLISH_REFRESH_EVERY_DEFAULT_MINUTES = 360; export const PUBLISH_REFRESH_EVERY_MAX_MINUTES = 43200; export const PUBLISH_DEFAULT_PREVIEW_BRANCH = "preview"; export function defaultPublish(): PublishSettings { return { enabled: false, held: false, checkEveryMinutes: PUBLISH_CHECK_EVERY_DEFAULT_MINUTES, refreshEveryMinutes: PUBLISH_REFRESH_EVERY_DEFAULT_MINUTES, quietHours: null, runner: "local", previewBranch: PUBLISH_DEFAULT_PREVIEW_BRANCH, hub: "off", homepage: "off", }; } function sanitizeQuietHours(value: unknown): PublishQuietHours | null { if (!value || typeof value !== "object" || Array.isArray(value)) return null; const r = value as Record; const start = clampHourOrNull(r.start); const end = clampHourOrNull(r.end); // Both valid hours and a non-empty window, or no window at all. if (start === null || end === null || start === end) return null; return { start, end }; } // Coerce a raw settings.publish into a clean PublishSettings: every field its // default when missing or ill-typed, numbers clamped, the two switches true // only when exactly true, a bad preview name the default. export function sanitizePublish(value: unknown): PublishSettings { const d = defaultPublish(); if (!value || typeof value !== "object" || Array.isArray(value)) return d; const r = value as Record; const previewBranch = typeof r.previewBranch === "string" && previewBranchProblem(r.previewBranch) === null ? r.previewBranch.trim() : d.previewBranch; return { enabled: r.enabled === true, held: r.held === true, checkEveryMinutes: clampPositiveInt(r.checkEveryMinutes, d.checkEveryMinutes, PUBLISH_CHECK_EVERY_MAX_MINUTES), refreshEveryMinutes: clampIntAllowZero( r.refreshEveryMinutes, d.refreshEveryMinutes, PUBLISH_REFRESH_EVERY_MAX_MINUTES, ), quietHours: sanitizeQuietHours(r.quietHours), runner: r.runner === "docker" ? "docker" : "local", previewBranch, hub: isPublishPolicy(r.hub) ? r.hub : d.hub, homepage: isPublishPolicy(r.homepage) ? r.homepage : d.homepage, }; } // Coerce a raw settings.transcriptionApps value into a clean keyed map of // AppInstanceConfig, dropping unknown/ill-typed fields. export function sanitizeTranscriptionApps( value: unknown, ): Record { if (!value || typeof value !== "object" || Array.isArray(value)) return {}; const out: Record = {}; for (const [id, raw] of Object.entries(value as Record)) { if (!raw || typeof raw !== "object") continue; out[id] = sanitizeWorkerConfig(raw); } return out; } // --- The schema ------------------------------------------------------------- // Global, OPERATIONAL settings shared across every site this editor powers. // Per-site presentation (branding, social links, channel groups, membership) // lives in sites//site.json — see common/lib/site.ts. // // FIELD ORDER IS FILE ORDER: zod emits keys in the order they are declared, and // writeSettings writes what the schema emits, so reordering these reorders every // settings.json on its next save. export const siteSettingsSchema = z.object({ adminTitle: settingsField((v): string => typeof v === "string" && v.trim() ? v.trim() : DEFAULT_ADMIN_TITLE).describe( "Title for the EDITOR admin shell only (the editor manages all sites and so is not tied to any one site's branding). Public sites get their own titles from site.json.", ), maxTranscriptPageBytes: settingsField((v): number => clampPageBytes(v)).describe( "Target size (bytes) of one exported transcript page shard — the unit the export site fetches. Clamped into [TRANSCRIPT_PAGE_MIN_BYTES, TRANSCRIPT_PAGE_HARD_CAP_BYTES] (256 KiB – 20 MiB); default 8 MiB.", ), transcriptionApp: settingsField((v): string => typeof v === "string" && TRANSCRIPTION_APPS[v] ? v : DEFAULT_TRANSCRIPTION_APP_ID).describe( "Active transcription app id (key into TRANSCRIPTION_APPS, e.g. \"whisper-cpp\" or \"chough\"). Selected globally; see common/lib/transcriptionApps.ts.", ), transcriptionApps: settingsField((v): Record => sanitizeTranscriptionApps(v)).describe( "Per-app configuration, keyed by app id. Each app reads only its own block; a missing block means \"use the app's defaults\". DEPRECATED in favor of `workers` (each local worker carries its own config); kept one release to drive migration and allow rollback. See common/lib/workers.ts.", ), workers: workersSchema.describe( "Configured transcription workers (named processing slots). The scheduler distributes each video to the highest-priority free worker. A settings.json predating this field is migrated to a single enabled worker from the active app (see defaultWorkersFromApps). See common/lib/workers.ts.", ), cookiesFromBrowser: settingsField((v): string => (typeof v === "string" ? v.trim() : "")).describe( "Browser spec (e.g. \"firefox\", \"chrome:Default\") passed to `yt-dlp --cookies-from-browser`. WHEN it is passed is governed by `cookieMode` below. Empty string = no cookies configured. Per-channel override available (ChannelConfig.cookiesFromBrowser).", ), cookieMode: settingsField((v): CookieMode => (isCookieMode(v) ? v : DEFAULT_COOKIE_MODE)).describe( "How yt-dlp invocations use the configured cookies (see common/lib/cookiePolicy.ts): \"always\" passes them on every invocation, \"when-required\" (default; the historical behavior) only to retry an auth/age failure, \"defer\" never in normal runs — auth-gated videos are excluded from batches and collected into the per-channel \"Needs cookies\" bucket for a manual cookie run. Per-channel override available (ChannelConfig.cookieMode).", ), social: settingsField((v): SocialSettings => sanitizeSocial(v)).describe( "Per-platform settings of the social posts. Today two keys, both X's, both chosen in the X account session section of /settings: where the X fetchers' login comes from (`social.x.cookieSource`) and where X posts may appear (`social.x.visibility`). See common/social/xCookieSource.ts.", ), sleepBetweenDownloadsSeconds: settingsField((v): number => clampSleepBetweenDownloadsSeconds(v)).describe( "Pause (seconds) inserted between per-video yt-dlp invocations in managed batch downloads, and between two auto-download units on one platform (release 17; the lane ignored it before). yt-dlp's own `-t sleep` only paces requests within one invocation, so without this the managed loop hammers the source IP back-to-back. The adaptive pace above its base (see `pacing`) is added to it. 0 disables. Per-channel override available for batch downloads.", ), pacing: settingsField((v): PacingSettingsBlock => sanitizePacing(v)).describe( "How the download pace adapts to rate limits, per platform (release 17). Every yt-dlp spawn against a platform paces its requests (`--sleep-requests`) at the platform's adaptive pace, and the download lane waits sleepBetweenDownloadsSeconds plus the pace above its fixed value between units. A rate limit doubles the pace; clean units ease it back; a rate limit that outlasts the cooldown cap holds the platform to one probe at a time. The live pace, cooldowns and holds are in `.auto-queue/state.json`, shown on /operations/download. See common/jobs/platformBackoff.ts.", ), downloadFormat: settingsField((v): DownloadFormatPreset => isDownloadFormatPreset(v) ? v : "auto").describe( "Default yt-dlp `-f` download format for every channel that doesn't set its own (ChannelConfig.downloadFormat). \"auto\" picks per-source: `original` for Odysee (whose HLS rungs are CDN-truncated), `bestaudio/worst` elsewhere. See common/ytdlp/downloadFormat.ts.", ), sourceVideoQuality: settingsField((v): SourceVideoQuality => isSourceVideoQuality(v) ? v : DEFAULT_SOURCE_VIDEO_QUALITY).describe( "Quality of the source container a full persist keeps — \"Persist source video\", the whole-recording fetch (`full: true`) and \"Persist kept now\" — for every channel that doesn't set its own (ChannelConfig.sourceVideoQuality). \"original\" (default) = `bestvideo*+bestaudio/best`; \"video_720\" = ≤720p H.264/AAC mp4 for clip and editing work, falling back to 480p and then to anything (logged). A single persist can override it. See common/ytdlp/downloadFormat.ts.", ), minFreeDiskGB: settingsField((v): number => clampMinFreeDiskGB(v)).describe( "Minimum free disk space (GB) required on the transcripts data directory for downloads to run. When free space is below this floor, a download job is prevented from starting and a running batch stops launching new videos (the in-flight one finishes). 0 disables the gate. See common/lib/diskSpace.ts.", ), resumeMarginGB: settingsField((v): number => clampResumeMarginGB(v)).describe( "Extra headroom (GB) above minFreeDiskGB that a stopped pipeline must see before it resumes. Resuming at the same number we stopped at flaps — the first restarted download pushes free space back under the floor. This is the hysteresis margin, so \"resumed\" means the operator actually freed something rather than a scratch file being cleaned up. 0 disables the hysteresis (resume at the floor). See diskGate() in common/lib/diskSpace.ts.", ), parallelTranscriptions: settingsField((v): number => clampParallelTranscriptions(v)).describe( "Default number of videos transcribed in parallel when a \"Transcribe missing\" / bucket run doesn't specify its own concurrency. The per-run Concurrency input in the channel UI overrides this for a single run.", ), inlineTranscribeOnFallback: settingsField((v): boolean => v === true).describe( "When true, the no-subs fallback in the managed downloader runs whisper inline immediately after the audio download succeeds. When false (default), audio is left for the next \"Transcribe missing\" pass so a batch download finishes faster and whisper can parallelize.", ), skipLiveDownloads: settingsField((v): boolean => v !== false).describe( "When true (default), managed downloads skip videos that are currently live or scheduled/upcoming, decided from a metadata-only prefetch pass. Finished livestream VODs (was_live) are NOT skipped and download normally. A skip is recorded but not archived, so the next sync/download-missing retries the video once the stream ends. Per-channel override available (ChannelConfig.skipLiveDownloads).", ), verifyAvailabilityBeforeClean: settingsField((v): boolean => v !== false).describe( "Whether the transcribed-audio cleanup sweep checks each candidate is still available upstream before deleting its audio, pinning (do-not-clean) any video found permanently gone. The delete is irreversible and a gone video's audio is irreplaceable, so this defaults to true. Turn it off for an offline or URL-less setup, where the check can never resolve and cleanup would otherwise never delete anything. See verifyBeforeClean.ts.", ), buildArchives: settingsField((v): boolean => v !== false).describe( "Whether site builds generate downloadable transcript/live-chat archive zips (into public/archives, linked on the Downloads page). Global default; a site can opt out via site.json `archives: false`, and a single build can skip via the \"Skip archive zips\" build control. Opt-out: default true.", ), archiveStorage: settingsField((v): ArchiveStorageSettings => { const r = (v && typeof v === "object" ? v : {}) as Record; return { bucket: typeof r.bucket === "string" ? r.bucket.trim() : "", publicBaseUrl: typeof r.publicBaseUrl === "string" ? r.publicBaseUrl.trim() : "", }; }).describe( "Overflow object storage (Cloudflare R2) for archive zips that exceed the Pages per-file size cap (see Site.archiveMaxBytes). When both fields are set, an oversize archive is uploaded here on deploy — via `wrangler r2 object put`, keyed `/archives/` — instead of being dropped, and the Downloads page links to `/`. Blank/absent → no overflow, so oversize archives stay unavailable (\"Too large to host\").", ), reportDebouncePreset: settingsField((v): ReportDebouncePreset => isReportDebouncePreset(v) ? v : DEFAULT_REPORT_DEBOUNCE_PRESET).describe( "Debounce preset for the global snapshot scheduler: how long it waits after the last report-changing action before regenerating affected channel reports. See REPORT_DEBOUNCE_PRESETS. Default \"fast\" (~1s, no cap).", ), autoRefreshIntervalSeconds: settingsField((v): number => clampAutoRefreshIntervalSeconds(v)).describe( "How often (seconds) the editor UI passively re-fetches the current page's server-rendered data via router.refresh(), so sidebar badges and reports stay live without a manual reload. Mounted globally; pauses while the tab is hidden. 0 disables passive refresh entirely. See AUTO_REFRESH_INTERVAL_*.", ), syncScheduler: settingsField((v): SyncSchedulerSettings => sanitizeSyncScheduler(v)).describe( "Global configuration for the scheduled (cron-driven) channel sync system. The per-channel cadence lives on ChannelConfig.syncIntervalMinutes; this block holds the defaults and guard rails the scheduler applies across all channels. See common/jobs/syncScheduler.ts.", ), autoQueue: autoQueueSchema.describe( "Configuration for the automatic priority-queue runners (auto-transcribe / auto-download). Each holds a tree policy that decides which channel's video to process next, cross-channel, by priority/round-robin/weighted-fair rules. Independent of syncScheduler (which decides staleness, not work order). See common/jobs/autoQueuePolicy.ts.", ), channelPriority: channelPrioritySchema.describe( "THE OPERATOR-FACING PRIORITY MODEL: one tier per channel plus one corpus-wide focus selector. It is the SOURCE the four `autoQueue[lane].root` trees are compiled from (common/lib/channelPriority.ts), not a second mechanism beside them — and its `paused` tier is the one part that is not a tree shape, filtering the runner's channel list instead. An empty document (the default) is today's behaviour exactly: no focus, every channel normal, the stored trees stand.", ), socialLinks: settingsField((v): SocialLink[] => parseSocialLinks(v)).describe( "Default social links applied to every site that doesn't define its own. A site inherits these unless its site.json carries an explicit `socialLinks` array — see Site.socialLinks / resolveSocialLinks in common/lib/site.ts. The one presentation field that lives globally so a shared footer doesn't have to be repeated per site.", ), homepageUrl: settingsField((v): string => normalizeHomepageUrl(v)).describe( "Absolute public URL of the family hub (e.g. \"https://archilyzer-hub.pages.dev\"). The default for every site's `hubUrl` (a site's own wins): published as `hubUrl` in the site's public `/site.json` and `/corpus.json`, so the hub can tell its member sites from arbitrary added origins. No page links to it (the header's Hub link was removed in release 14). Empty = none published. Normalized to a trailing-slash-free http(s) URL.", ), savedVideoBackup: settingsField((v): SavedVideoBackupSettings => sanitizeSavedVideoBackup(v)).describe( "Backup configuration for the saved-video store (Phase 4 of the video-persistence feature). When enabled with a destination, the store is mirrored there (additively, no deletes) with a per-backup manifest, and the sync scheduler runs the backup on the configured cadence. See common/controller/backupSavedVideos.ts.", ), storage: settingsField((v): StorageSettings => sanitizeStorage(v)).describe( "Where a channel's downloaded media goes when it is relocated off the corpus disk. A DEFAULT ONLY: the relocate controller never reads it and always takes an explicit root, so this is the value the per-channel Storage panel prefills and the /channels bulk move falls back to. Blank = no default. See StorageSettings.", ), buildPipeline: settingsField((v): BuildPipelineSettings => sanitizeBuildPipeline(v)).describe( "The docker build runner's settings. A site's build is a publish stage (release 18): by default (`publish.runner: \"local\"`) every site builds in turn as a child of the editor, one stage at a time on the `publish` queue. With `publish.runner: \"docker\"` (or `archilyzer publish build all --runner docker`) every stale site builds in its own container — Dockerfile.build, the image and maxParallelBuilds below — on a Linux host whose container engine answers; it is refused inside a container. Deploys are their own stages. See PUBLISH.md. There is no mode switch; a `mode` key left in an older file is dropped on the next save.", ), digest: settingsField((v): DigestSettings => sanitizeDigest(v)).describe( "AI digest generation (chapters + topic tags over the existing transcripts). Local-first: the metered lane is off by default. See DigestSettings.", ), diarization: settingsField((v): DiarizationSettings => sanitizeDiarization(v)).describe( "Speaker diarization captured right after transcription, while the audio is still on disk. OFF by default. See DiarizationSettings.", ), backfill: settingsField((v): BackfillSettings => sanitizeBackfill(v)).describe( "The generic catch-up lane for derived data the existing corpus predates. OFF by default, and idle-only when on. See BackfillSettings.", ), attribution: settingsField((v): AttributionSettings => sanitizeAttribution(v)).describe( "Naming the speakers diarization found (or reconstructing them from the transcript when it found none). OFF by default. See AttributionSettings.", ), archiveOrg: settingsField((v): ArchiveOrgFetchSettings => sanitizeArchiveOrg(v)).describe( "How archive.org files are fetched (controller/archiveOrgDownload.ts). Over BitTorrent with aria2c when the item's torrent carries the file — archive.org is the torrent's web seed, so the swarm takes load off archive.org — then seeded for a while; otherwise, or when the torrent stalls, a direct download from archive.org. Either way the file is verified against archive.org's sha1/md5. No yt-dlp.", ), publish: settingsField((v): PublishSettings => sanitizePublish(v)).describe( "The publish LANE (release 18): a runner that, when the index is stale, updates it and then builds — and, where a site's own `publish.auto` (site.json) says so, deploys — what changed, one stage at a time on the `publish` queue. OFF by default; the manual stages (Publish now, Build, Deploy) work either way. See PublishSettings and common/publish/publishRunner.ts.", ), seeder: settingsField((v): SeederSettings => sanitizeSeeder(v)).describe( "The home seeder of last resort (release 21): `archilyzer seed` seeds the sites' playable torrents (`archilyzer media playable`), each one only while no other seeder has it, and only behind a VPN (the docker profile `seeder`, docker-compose.seeder.yml). Not configured until `sites` names one. See SeederSettings and common/controller/seeder.ts.", ), }); export type SiteSettings = z.infer; // The whole default settings object, without touching disk: the schema's answer // for an empty file. Not a second literal — a default that lived anywhere but in // the field's own coercion would be a second place to change it. export function defaults(): SiteSettings { return siteSettingsSchema.parse({}); } // Exported under this name too, because tests and e2e helpers already say it: // a caller that needs a settings-SHAPED value rather than the operator's actual // configuration builds one here without a settings.json. export function defaultSiteSettings(): SiteSettings { return defaults(); }