commit b15d2a0141a3d05a3267ed847e7ebd685e6704e0
parent b94055b379541a2a1af878782fd8d28fc1e6f836
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 23 Sep 2026 19:36:03 -0400
settings: settings.json.example and SETTINGS.md generated from the schema
one-core phase 3 slice 4a, commit 5. common/bin/settings-example.ts
(`--check` to verify; becomes `archilyzer settings example` in phase 4)
writes both files from common/lib/settingsDocs.ts, which renders them from
siteSettingsSchema: keys and order from its shape, defaults from
defaultSiteSettings(), prose from each field's .describe().
settingsDocs.test.ts asserts the committed bytes equal the generator's.
The example is the default object minus `workers` — spelling `workers: []`
would mean zero transcription slots, while an absent key synthesizes one
from transcriptionApp on read. The three dead keys the hand-written example
carried (transcribeBin, transcribeModel, transcribeArgs — the pre-multi-app
spelling, migrated on read since the app registry landed) are gone.
SETUP.md points at SETTINGS.md.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
6 files changed, 809 insertions(+), 16 deletions(-)
diff --git a/SETTINGS.md b/SETTINGS.md
@@ -0,0 +1,435 @@
+# settings.json keys
+
+<!-- GENERATED by common/bin/settings-example.ts from common/lib/settingsSchema.ts — do not edit by hand. -->
+
+Global operational settings shared by every site this editor powers, persisted to `settings.json` at the repo root (or `$SETTINGS_FILE`). Per-site presentation lives in `sites/<id>/site.json`. Every key is optional: a missing key reads as its default, an ill-typed one is coerced to its default or clamped, and an unknown one is dropped on the next save.
+
+Regenerate this file and `settings.json.example` with `pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts`.
+
+`settings.json.example` is the default object with one key left out, `workers`: a file that does not name it gets a worker list synthesized from `transcriptionApp` on read, where `workers: []` would mean no transcription at all.
+
+| Key | Default |
+|---|---|
+| [`adminTitle`](#admintitle) | `"Transcript Browser Admin"` |
+| [`maxTranscriptPageBytes`](#maxtranscriptpagebytes) | `8388608` |
+| [`transcriptionApp`](#transcriptionapp) | `"whisper-cpp"` |
+| [`transcriptionApps`](#transcriptionapps) | `{}` |
+| [`workers`](#workers) | `[]` |
+| [`cookiesFromBrowser`](#cookiesfrombrowser) | `""` |
+| [`cookieMode`](#cookiemode) | `"when-required"` |
+| [`sleepBetweenDownloadsSeconds`](#sleepbetweendownloadsseconds) | `10` |
+| [`downloadFormat`](#downloadformat) | `"auto"` |
+| [`minFreeDiskGB`](#minfreediskgb) | `5` |
+| [`resumeMarginGB`](#resumemargingb) | `2` |
+| [`parallelTranscriptions`](#paralleltranscriptions) | `2` |
+| [`inlineTranscribeOnFallback`](#inlinetranscribeonfallback) | `false` |
+| [`skipLiveDownloads`](#skiplivedownloads) | `true` |
+| [`verifyAvailabilityBeforeClean`](#verifyavailabilitybeforeclean) | `true` |
+| [`buildArchives`](#buildarchives) | `true` |
+| [`archiveStorage`](#archivestorage) | object — see below |
+| [`reportDebouncePreset`](#reportdebouncepreset) | `"fast"` |
+| [`autoRefreshIntervalSeconds`](#autorefreshintervalseconds) | `5` |
+| [`syncScheduler`](#syncscheduler) | object — see below |
+| [`autoQueue`](#autoqueue) | object — see below |
+| [`channelPriority`](#channelpriority) | object — see below |
+| [`socialLinks`](#sociallinks) | `[]` |
+| [`homepageUrl`](#homepageurl) | `""` |
+| [`savedVideoBackup`](#savedvideobackup) | object — see below |
+| [`storage`](#storage) | object — see below |
+| [`buildPipeline`](#buildpipeline) | object — see below |
+| [`digest`](#digest) | object — see below |
+| [`diarization`](#diarization) | object — see below |
+| [`backfill`](#backfill) | object — see below |
+| [`attribution`](#attribution) | object — see below |
+
+## `adminTitle`
+
+Title for the EDITOR admin shell only (the editor manages all sites and so is not tied to any one site's branding). Public sites get their own titles from site.json.
+
+Default: `"Transcript Browser Admin"`
+
+## `maxTranscriptPageBytes`
+
+Target size (bytes) of one exported transcript page shard — the unit the export site fetches. Clamped into [TRANSCRIPT_PAGE_MIN_BYTES, TRANSCRIPT_PAGE_HARD_CAP_BYTES] (256 KiB – 20 MiB); default 8 MiB.
+
+Default: `8388608`
+
+## `transcriptionApp`
+
+Active transcription app id (key into TRANSCRIPTION_APPS, e.g. "whisper-cpp" or "chough"). Selected globally; see common/lib/transcriptionApps.ts.
+
+Default: `"whisper-cpp"`
+
+## `transcriptionApps`
+
+Per-app configuration, keyed by app id. Each app reads only its own block; a missing block means "use the app's defaults". DEPRECATED in favor of `workers` (each local worker carries its own config); kept one release to drive migration and allow rollback. See common/lib/workers.ts.
+
+Default:
+
+```json
+{}
+```
+
+## `workers`
+
+Configured transcription workers (named processing slots). The scheduler distributes each video to the highest-priority free worker. A settings.json predating this field is migrated to a single enabled worker from the active app (see defaultWorkersFromApps). See common/lib/workers.ts.
+
+Default:
+
+```json
+[]
+```
+
+## `cookiesFromBrowser`
+
+Browser spec (e.g. "firefox", "chrome:Default") passed to `yt-dlp --cookies-from-browser`. WHEN it is passed is governed by `cookieMode` below. Empty string = no cookies configured. Per-channel override available (ChannelConfig.cookiesFromBrowser).
+
+Default: `""`
+
+## `cookieMode`
+
+How yt-dlp invocations use the configured cookies (see common/lib/cookiePolicy.ts): "always" passes them on every invocation, "when-required" (default; the historical behavior) only to retry an auth/age failure, "defer" never in normal runs — auth-gated videos are excluded from batches and collected into the per-channel "Needs cookies" bucket for a manual cookie run. Per-channel override available (ChannelConfig.cookieMode).
+
+Default: `"when-required"`
+
+## `sleepBetweenDownloadsSeconds`
+
+Pause (seconds) inserted between per-video yt-dlp invocations in managed batch downloads. yt-dlp's own `-t sleep` only paces requests within one invocation, so without this the managed loop hammers the source IP back-to-back. 0 disables. Per-channel override available.
+
+Default: `10`
+
+## `downloadFormat`
+
+Default yt-dlp `-f` download format for every channel that doesn't set its own (ChannelConfig.downloadFormat). "auto" picks per-source: `original` for Odysee (whose HLS rungs are CDN-truncated), `bestaudio/worst` elsewhere. See common/ytdlp/downloadFormat.ts.
+
+Default: `"auto"`
+
+## `minFreeDiskGB`
+
+Minimum free disk space (GB) required on the transcripts data directory for downloads to run. When free space is below this floor, a download job is prevented from starting and a running batch stops launching new videos (the in-flight one finishes). 0 disables the gate. See common/lib/diskSpace.ts.
+
+Default: `5`
+
+## `resumeMarginGB`
+
+Extra headroom (GB) above minFreeDiskGB that a stopped pipeline must see before it resumes. Resuming at the same number we stopped at flaps — the first restarted download pushes free space back under the floor. This is the hysteresis margin, so "resumed" means the operator actually freed something rather than a scratch file being cleaned up. 0 disables the hysteresis (resume at the floor). See diskGate() in common/lib/diskSpace.ts.
+
+Default: `2`
+
+## `parallelTranscriptions`
+
+Default number of videos transcribed in parallel when a "Transcribe missing" / bucket run doesn't specify its own concurrency. The per-run Concurrency input in the channel UI overrides this for a single run.
+
+Default: `2`
+
+## `inlineTranscribeOnFallback`
+
+When true, the no-subs fallback in the managed downloader runs whisper inline immediately after the audio download succeeds. When false (default), audio is left for the next "Transcribe missing" pass so a batch download finishes faster and whisper can parallelize.
+
+Default: `false`
+
+## `skipLiveDownloads`
+
+When true (default), managed downloads skip videos that are currently live or scheduled/upcoming, decided from a metadata-only prefetch pass. Finished livestream VODs (was_live) are NOT skipped and download normally. A skip is recorded but not archived, so the next sync/download-missing retries the video once the stream ends. Per-channel override available (ChannelConfig.skipLiveDownloads).
+
+Default: `true`
+
+## `verifyAvailabilityBeforeClean`
+
+Whether the transcribed-audio cleanup sweep checks each candidate is still available upstream before deleting its audio, pinning (do-not-clean) any video found permanently gone. The delete is irreversible and a gone video's audio is irreplaceable, so this defaults to true. Turn it off for an offline or URL-less setup, where the check can never resolve and cleanup would otherwise never delete anything. See verifyBeforeClean.ts.
+
+Default: `true`
+
+## `buildArchives`
+
+Whether site builds generate downloadable transcript/live-chat archive zips (into public/archives, linked on the Downloads page). Global default; a site can opt out via site.json `archives: false`, and a single build can skip via the "Skip archive zips" build control. Opt-out: default true.
+
+Default: `true`
+
+## `archiveStorage`
+
+Overflow object storage (Cloudflare R2) for archive zips that exceed the Pages per-file size cap (see Site.archiveMaxBytes). When both fields are set, an oversize archive is uploaded here on deploy — via `wrangler r2 object put`, keyed `<siteId>/archives/<file>` — instead of being dropped, and the Downloads page links to `<publicBaseUrl>/<key>`. Blank/absent → no overflow, so oversize archives stay unavailable ("Too large to host").
+
+Default:
+
+```json
+{
+ "bucket": "",
+ "publicBaseUrl": ""
+}
+```
+
+## `reportDebouncePreset`
+
+Debounce preset for the global snapshot scheduler: how long it waits after the last report-changing action before regenerating affected channel reports. See REPORT_DEBOUNCE_PRESETS. Default "fast" (~1s, no cap).
+
+Default: `"fast"`
+
+## `autoRefreshIntervalSeconds`
+
+How often (seconds) the editor UI passively re-fetches the current page's server-rendered data via router.refresh(), so sidebar badges and reports stay live without a manual reload. Mounted globally; pauses while the tab is hidden. 0 disables passive refresh entirely. See AUTO_REFRESH_INTERVAL_*.
+
+Default: `5`
+
+## `syncScheduler`
+
+Global configuration for the scheduled (cron-driven) channel sync system. The per-channel cadence lives on ChannelConfig.syncIntervalMinutes; this block holds the defaults and guard rails the scheduler applies across all channels. See common/jobs/syncScheduler.ts.
+
+Default:
+
+```json
+{
+ "enabled": false,
+ "defaultIntervalMinutes": 1440,
+ "maxConcurrentSyncs": 2,
+ "quietHoursStart": null,
+ "quietHoursEnd": null,
+ "backoffBaseMinutes": 30,
+ "backoffMaxMinutes": 1440,
+ "heartbeatSeconds": 0,
+ "keepLatestCheckIntervalMinutes": 1440,
+ "fullSweepIntervalMinutes": 1440,
+ "fullSweepConfirmMaxSuspects": 25,
+ "fullSweepShrinkGuardPercent": 10
+}
+```
+
+## `autoQueue`
+
+Configuration for the automatic priority-queue runners (auto-transcribe / auto-download). Each holds a tree policy that decides which channel's video to process next, cross-channel, by priority/round-robin/weighted-fair rules. Independent of syncScheduler (which decides staleness, not work order). See common/jobs/autoQueuePolicy.ts.
+
+Default:
+
+```json
+{
+ "transcription": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "listed",
+ "snoozeUntil": null,
+ "held": false,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": []
+ }
+ },
+ "download": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "listed",
+ "snoozeUntil": null,
+ "held": false,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": []
+ }
+ },
+ "digest": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "cheapest",
+ "snoozeUntil": null,
+ "held": false,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": [
+ {
+ "id": "all",
+ "match": {
+ "type": "all"
+ },
+ "weight": 1,
+ "maxWorkers": null
+ }
+ ]
+ }
+ },
+ "backfill": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "listed",
+ "snoozeUntil": null,
+ "held": true,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": [
+ {
+ "id": "all",
+ "match": {
+ "type": "all"
+ },
+ "weight": 1,
+ "maxWorkers": null
+ }
+ ]
+ }
+ }
+}
+```
+
+## `channelPriority`
+
+THE OPERATOR-FACING PRIORITY MODEL: one tier per channel plus one corpus-wide focus selector. It is the SOURCE the four `autoQueue[lane].root` trees are compiled from (common/lib/channelPriority.ts), not a second mechanism beside them — and its `paused` tier is the one part that is not a tree shape, filtering the runner's channel list instead. An empty document (the default) is today's behaviour exactly: no focus, every channel normal, the stored trees stand.
+
+Default:
+
+```json
+{
+ "focus": {
+ "kind": "none"
+ },
+ "channels": {}
+}
+```
+
+## `socialLinks`
+
+Default social links applied to every site that doesn't define its own. A site inherits these unless its site.json carries an explicit `socialLinks` array — see Site.socialLinks / resolveSocialLinks in common/lib/site.ts. The one presentation field that lives globally so a shared footer doesn't have to be repeated per site.
+
+Default:
+
+```json
+[]
+```
+
+## `homepageUrl`
+
+Absolute public URL of the family hub/homepage (e.g. "https://archilyzer.pages.dev"). Every export site links back to it ("the family" backlink) when set. Empty = no hub link rendered. Normalized to a trailing-slash-free http(s) URL.
+
+Default: `""`
+
+## `savedVideoBackup`
+
+Backup configuration for the saved-video store (Phase 4 of the video-persistence feature). When enabled with a destination, the store is mirrored there (additively, no deletes) with a per-backup manifest, and the sync scheduler runs the backup on the configured cadence. See common/controller/backupSavedVideos.ts.
+
+Default:
+
+```json
+{
+ "enabled": false,
+ "dest": "",
+ "intervalMinutes": 1440
+}
+```
+
+## `storage`
+
+Where a channel's downloaded media goes when it is relocated off the corpus disk. A DEFAULT ONLY: the relocate controller never reads it and always takes an explicit root, so this is the value the per-channel Storage panel prefills and the /channels bulk move falls back to. Blank = no default. See StorageSettings.
+
+Default:
+
+```json
+{
+ "locations": [],
+ "defaultLocationId": ""
+}
+```
+
+## `buildPipeline`
+
+How the static export is built: "basic" reuses the single export/ tree and serializes builds on one queue (the long-standing behavior); "docker" runs each site's build in an isolated container for safe parallelism. The Docker pipeline itself is a follow-up; this block persists the chosen mode plus the container/concurrency knobs the deploy page and the future orchestrator read.
+
+Default:
+
+```json
+{
+ "mode": "basic",
+ "maxParallelBuilds": 2,
+ "dockerImage": "yt-dlp-transcript-browser-build",
+ "dockerfile": "Dockerfile.build"
+}
+```
+
+## `digest`
+
+AI digest generation (chapters + topic tags over the existing transcripts). Local-first: the metered lane is off by default. See DigestSettings.
+
+Default:
+
+```json
+{
+ "remoteEnabled": false,
+ "longTailSeconds": 14400,
+ "localAppId": "ollama-direct",
+ "remoteAppId": "claude-code",
+ "apps": {},
+ "yieldToTranscription": true,
+ "yieldToCpuWorkers": false,
+ "spendCapUsd": 0,
+ "sections": [
+ "chapters"
+ ],
+ "timestampMode": "chunk-local",
+ "promptVariant": ""
+}
+```
+
+## `diarization`
+
+Speaker diarization captured right after transcription, while the audio is still on disk. OFF by default. See DiarizationSettings.
+
+Default:
+
+```json
+{
+ "enabled": false,
+ "inlineAfterTranscribe": false,
+ "threshold": 0.9,
+ "threads": 4,
+ "engine": "sherpa-onnx",
+ "backend": "vulkan",
+ "python": "python3",
+ "segModel": "",
+ "embModel": "",
+ "sortformerBin": "",
+ "sortformerModel": "",
+ "concurrency": 1,
+ "maxAudioHours": 0
+}
+```
+
+## `backfill`
+
+The generic catch-up lane for derived data the existing corpus predates. OFF by default, and idle-only when on. See BackfillSettings.
+
+Default:
+
+```json
+{
+ "concurrency": 1,
+ "allowRedownload": false
+}
+```
+
+## `attribution`
+
+Naming the speakers diarization found (or reconstructing them from the transcript when it found none). OFF by default. See AttributionSettings.
+
+Default:
+
+```json
+{
+ "enabled": false,
+ "appId": "ollama-direct",
+ "model": "",
+ "diarizedEnabled": false,
+ "textOnlyEnabled": false,
+ "promptVersion": 1
+}
+```
diff --git a/SETUP.md b/SETUP.md
@@ -252,8 +252,10 @@ channels.
## Configuration & environment variables
Most configuration now lives in the editor's **/settings** page, persisted to
-`settings.json` at the repo root (gitignored). `settings.json.example` is a minimal
-starting template. Settings are optional — a missing/partial `settings.json` falls
+`settings.json` at the repo root (gitignored). Every key, its default and what it
+does is in [SETTINGS.md](SETTINGS.md); `settings.json.example` is the defaults as a
+starting template. Both are generated from the settings schema
+(`common/lib/settingsSchema.ts`). Settings are optional — a missing/partial `settings.json` falls
back to built-in defaults, so the app runs out of the box.
Paths and binaries resolve through `getPaths()` in `common/lib/paths.ts`. Override
diff --git a/common/bin/settings-example.ts b/common/bin/settings-example.ts
@@ -0,0 +1,58 @@
+#!/usr/bin/env tsx
+// WRITE settings.json.example AND SETTINGS.md FROM THE SETTINGS SCHEMA.
+//
+// Usage (from the repo root):
+// pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts
+// pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts --check
+//
+// `--check` writes nothing and exits 1 if either committed file differs from
+// what the schema generates (the same claim common/lib/settingsDocs.test.ts
+// makes). Becomes `archilyzer settings example` in one-core phase 4.
+//
+// Reads no settings.json and writes no settings.json: both outputs are
+// functions of the schema alone.
+
+import { readFile, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import {
+ renderSettingsExample,
+ renderSettingsMarkdown,
+} from "../lib/settingsDocs";
+import { parseFlags } from "./_parseFlags";
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+const OUTPUTS: ReadonlyArray<[string, () => string]> = [
+ ["settings.json.example", renderSettingsExample],
+ ["SETTINGS.md", renderSettingsMarkdown],
+];
+
+async function main(): Promise<number> {
+ const flags = parseFlags(process.argv.slice(2));
+ const check = flags.check === "true";
+ let stale = 0;
+ for (const [name, render] of OUTPUTS) {
+ const file = path.join(REPO, name);
+ const want = render();
+ if (check) {
+ const have = await readFile(file, "utf8").catch(() => "");
+ if (have !== want) {
+ console.error(`${name} is stale — regenerate it`);
+ stale++;
+ }
+ continue;
+ }
+ await writeFile(file, want);
+ console.log(`wrote ${name}`);
+ }
+ return stale > 0 ? 1 : 0;
+}
+
+main().then(
+ (code) => process.exit(code),
+ (err) => {
+ console.error(err);
+ process.exit(1);
+ },
+);
diff --git a/common/lib/settingsDocs.test.ts b/common/lib/settingsDocs.test.ts
@@ -0,0 +1,34 @@
+import { readFileSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { renderSettingsExample, renderSettingsMarkdown } from "./settingsDocs";
+
+// Run with: node_modules/.bin/tsx --test common/lib/settingsDocs.test.ts
+//
+// settings.json.example and SETTINGS.md are GENERATED from the settings schema
+// (common/bin/settings-example.ts). This is what keeps them generated: a hand
+// edit to either file, or a schema change without a regenerate, fails here.
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+for (const [name, render] of [
+ ["settings.json.example", renderSettingsExample],
+ ["SETTINGS.md", renderSettingsMarkdown],
+] as const) {
+ test(`${name} is what the schema generates`, () => {
+ const committed = readFileSync(path.join(REPO, name), "utf8");
+ assert.equal(
+ committed,
+ render(),
+ `${name} is stale: run pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts`,
+ );
+ });
+}
+
+test("the example parses back to the defaults", async () => {
+ const { siteSettingsSchema, defaultSiteSettings } = await import("./settingsSchema");
+ const parsed = siteSettingsSchema.parse(JSON.parse(renderSettingsExample()));
+ assert.deepEqual(parsed, defaultSiteSettings());
+});
diff --git a/common/lib/settingsDocs.ts b/common/lib/settingsDocs.ts
@@ -0,0 +1,101 @@
+// THE TWO FILES GENERATED FROM THE SETTINGS SCHEMA: settings.json.example and
+// the key table SETTINGS.md, both at the repo root.
+//
+// Pure renderers — `common/bin/settings-example.ts` writes them, and
+// `settingsDocs.test.ts` asserts the committed files are byte-identical to what
+// these return, so neither can be edited by hand without the test failing.
+//
+// Everything comes from `siteSettingsSchema` (./settingsSchema.ts): the keys and
+// their order from its shape, the defaults from `defaultSiteSettings()`, the
+// prose from each field's `.describe()`. Changing a default or a description is
+// a schema edit followed by regenerating, never an edit here.
+
+import { defaultSiteSettings, siteSettingsSchema } from "./settingsSchema";
+
+// `workers` IS LEFT OUT OF THE EXAMPLE, and that is the one place the example
+// is not the literal default object. Its default is `[]`, and a settings.json
+// that SPELLS `workers: []` means "no transcription workers" — auto-transcribe
+// then does nothing, silently. A file that does not name the key gets a worker
+// list synthesized from `transcriptionApp` on read (see getSettings), which is
+// what a template copied to settings.json should give.
+export const EXAMPLE_OMITTED_KEYS = ["workers"] as const;
+
+export function renderSettingsExample(): string {
+ const d = defaultSiteSettings() as Record<string, unknown>;
+ for (const key of EXAMPLE_OMITTED_KEYS) delete d[key];
+ return JSON.stringify(d, null, 2) + "\n";
+}
+
+function isScalar(v: unknown): boolean {
+ return v === null || typeof v !== "object";
+}
+
+// The default as a table cell: a scalar inline, an empty container inline,
+// anything larger by reference to its section.
+function defaultCell(v: unknown): string {
+ if (isScalar(v)) return "`" + JSON.stringify(v) + "`";
+ const json = JSON.stringify(v);
+ if (json === "[]" || json === "{}") return "`" + json + "`";
+ return Array.isArray(v) ? "list — see below" : "object — see below";
+}
+
+export function renderSettingsMarkdown(): string {
+ const d = defaultSiteSettings() as Record<string, unknown>;
+ const shape = siteSettingsSchema.shape as Record<
+ string,
+ { description?: string }
+ >;
+ const keys = Object.keys(shape);
+ const out: string[] = [];
+ out.push("# settings.json keys");
+ out.push("");
+ out.push(
+ "<!-- GENERATED by common/bin/settings-example.ts from common/lib/settingsSchema.ts — do not edit by hand. -->",
+ );
+ out.push("");
+ out.push(
+ "Global operational settings shared by every site this editor powers, " +
+ "persisted to `settings.json` at the repo root (or `$SETTINGS_FILE`). " +
+ "Per-site presentation lives in `sites/<id>/site.json`. Every key is " +
+ "optional: a missing key reads as its default, an ill-typed one is " +
+ "coerced to its default or clamped, and an unknown one is dropped on the " +
+ "next save.",
+ );
+ out.push("");
+ out.push(
+ "Regenerate this file and `settings.json.example` with " +
+ "`pnpm --filter yt-dlp-transcript-common exec tsx bin/settings-example.ts`.",
+ );
+ out.push("");
+ out.push(
+ "`settings.json.example` is the default object with one key left out, " +
+ "`workers`: a file that does not name it gets a worker list synthesized " +
+ "from `transcriptionApp` on read, where `workers: []` would mean no " +
+ "transcription at all.",
+ );
+ out.push("");
+ out.push("| Key | Default |");
+ out.push("|---|---|");
+ for (const key of keys) {
+ out.push(`| [\`${key}\`](#${key.toLowerCase()}) | ${defaultCell(d[key])} |`);
+ }
+ out.push("");
+ for (const key of keys) {
+ out.push(`## \`${key}\``);
+ out.push("");
+ out.push(shape[key].description ?? "");
+ out.push("");
+ const v = d[key];
+ if (isScalar(v)) {
+ out.push(`Default: \`${JSON.stringify(v)}\``);
+ } else {
+ out.push("Default:");
+ out.push("");
+ out.push("```json");
+ out.push(JSON.stringify(v, null, 2));
+ out.push("```");
+ }
+ out.push("");
+ }
+ return out.join("\n");
+}
diff --git a/settings.json.example b/settings.json.example
@@ -1,19 +1,182 @@
{
"adminTitle": "Transcript Browser Admin",
"maxTranscriptPageBytes": 8388608,
- "transcribeBin": "whisper-cli",
- "transcribeModel": "~/whispercpp/whisper.cpp/models/ggml-base.en.bin",
- "transcribeArgs": [
- "-ojf",
- "-l",
- "en",
- "-m",
- "{model}",
- "-of",
- "{outputBase}",
- "{audioFile}"
- ],
- "cookiesFromBrowser": "firefox",
+ "transcriptionApp": "whisper-cpp",
+ "transcriptionApps": {},
+ "cookiesFromBrowser": "",
+ "cookieMode": "when-required",
"sleepBetweenDownloadsSeconds": 10,
- "inlineTranscribeOnFallback": false
+ "downloadFormat": "auto",
+ "minFreeDiskGB": 5,
+ "resumeMarginGB": 2,
+ "parallelTranscriptions": 2,
+ "inlineTranscribeOnFallback": false,
+ "skipLiveDownloads": true,
+ "verifyAvailabilityBeforeClean": true,
+ "buildArchives": true,
+ "archiveStorage": {
+ "bucket": "",
+ "publicBaseUrl": ""
+ },
+ "reportDebouncePreset": "fast",
+ "autoRefreshIntervalSeconds": 5,
+ "syncScheduler": {
+ "enabled": false,
+ "defaultIntervalMinutes": 1440,
+ "maxConcurrentSyncs": 2,
+ "quietHoursStart": null,
+ "quietHoursEnd": null,
+ "backoffBaseMinutes": 30,
+ "backoffMaxMinutes": 1440,
+ "heartbeatSeconds": 0,
+ "keepLatestCheckIntervalMinutes": 1440,
+ "fullSweepIntervalMinutes": 1440,
+ "fullSweepConfirmMaxSuspects": 25,
+ "fullSweepShrinkGuardPercent": 10
+ },
+ "autoQueue": {
+ "transcription": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "listed",
+ "snoozeUntil": null,
+ "held": false,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": []
+ }
+ },
+ "download": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "listed",
+ "snoozeUntil": null,
+ "held": false,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": []
+ }
+ },
+ "digest": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "cheapest",
+ "snoozeUntil": null,
+ "held": false,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": [
+ {
+ "id": "all",
+ "match": {
+ "type": "all"
+ },
+ "weight": 1,
+ "maxWorkers": null
+ }
+ ]
+ }
+ },
+ "backfill": {
+ "enabled": false,
+ "maxWorkers": null,
+ "replaceAutoSubs": false,
+ "order": "listed",
+ "snoozeUntil": null,
+ "held": true,
+ "root": {
+ "id": "root",
+ "mode": "strict",
+ "weight": 1,
+ "maxWorkers": null,
+ "children": [
+ {
+ "id": "all",
+ "match": {
+ "type": "all"
+ },
+ "weight": 1,
+ "maxWorkers": null
+ }
+ ]
+ }
+ }
+ },
+ "channelPriority": {
+ "focus": {
+ "kind": "none"
+ },
+ "channels": {}
+ },
+ "socialLinks": [],
+ "homepageUrl": "",
+ "savedVideoBackup": {
+ "enabled": false,
+ "dest": "",
+ "intervalMinutes": 1440
+ },
+ "storage": {
+ "locations": [],
+ "defaultLocationId": ""
+ },
+ "buildPipeline": {
+ "mode": "basic",
+ "maxParallelBuilds": 2,
+ "dockerImage": "yt-dlp-transcript-browser-build",
+ "dockerfile": "Dockerfile.build"
+ },
+ "digest": {
+ "remoteEnabled": false,
+ "longTailSeconds": 14400,
+ "localAppId": "ollama-direct",
+ "remoteAppId": "claude-code",
+ "apps": {},
+ "yieldToTranscription": true,
+ "yieldToCpuWorkers": false,
+ "spendCapUsd": 0,
+ "sections": [
+ "chapters"
+ ],
+ "timestampMode": "chunk-local",
+ "promptVariant": ""
+ },
+ "diarization": {
+ "enabled": false,
+ "inlineAfterTranscribe": false,
+ "threshold": 0.9,
+ "threads": 4,
+ "engine": "sherpa-onnx",
+ "backend": "vulkan",
+ "python": "python3",
+ "segModel": "",
+ "embModel": "",
+ "sortformerBin": "",
+ "sortformerModel": "",
+ "concurrency": 1,
+ "maxAudioHours": 0
+ },
+ "backfill": {
+ "concurrency": 1,
+ "allowRedownload": false
+ },
+ "attribution": {
+ "enabled": false,
+ "appId": "ollama-direct",
+ "model": "",
+ "diarizedEnabled": false,
+ "textOnlyEnabled": false,
+ "promptVersion": 1
+ }
}