commit 16b38668a3cb8675d409c043e29be9ce6877cb4f
parent 37ad056c9128834998fc6c454cd282c6cb3c6f71
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 24 Sep 2026 13:32:47 -0400
bin: file-schemas-docs — SITE.md and CHANNEL.md generated from the schemas
common/bin/file-schemas-docs.ts (+ --check), sibling of settings-example.ts,
renders through lib/fileSchemaDocs.ts with settingsDocs.ts's table renderer
(its helpers are now exported): SITE.md — every site.json key with its
default and the groups[] / channels[] / socialLinks[] / relatedSites[]
tables; CHANNEL.md — every config.json key, which are required / config /
sync state, the one-line smallest channel, and the downloadFilter /
audioCheck tables. fileSchemaDocs.test.ts pins the committed bytes. No
.example files.
Links: SETUP.md (configuration section and the SITES_DIR row), AGENTS.md
(the corpus table, and a paragraph naming the three schemas, the patcher
and sidecar()). plans/FACTS.md's stale SUB_FILE_RE anchor is corrected.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
9 files changed, 588 insertions(+), 14 deletions(-)
diff --git a/AGENTS.md b/AGENTS.md
@@ -193,8 +193,8 @@ apps*, not [DEPLOY_DOCKER.md](DEPLOY_DOCKER.md), which is about *building sites*
| Path | What it is |
|---|---|
-| `transcripts/channels/<slug>/` | One channel: `config.json`, `playlist`, `archive`, `snapshot.json`, and `data/<videoId>/` holding media, transcripts and sidecars. **`data/` may be an absolute SYMLINK** — see below. |
-| `transcripts/sites/<id>/site.json` | **Per-site config, including the public URL.** This is where deployed-site facts live — *not* under `channels/`. |
+| `transcripts/channels/<slug>/` | One channel: `config.json` (every key in [CHANNEL.md](CHANNEL.md)), `playlist`, `archive`, `snapshot.json`, and `data/<videoId>/` holding media, transcripts and sidecars. **`data/` may be an absolute SYMLINK** — see below. |
+| `transcripts/sites/<id>/site.json` | **Per-site config, including the public URL** (every key in [SITE.md](SITE.md)). This is where deployed-site facts live — *not* under `channels/`. |
| `transcripts/index.mdb` | The LMDB transcript index. Key-only range scans over its `byChannel` sub-DB are cheap; see `common/controller/recencyIndex.ts`. |
| `transcripts/saved-videos/` | Persisted source-video store. |
| `transcripts/search-aliases.json`, `duplicates*.json` | Corpus-wide curated data. |
@@ -220,6 +220,14 @@ new path without moving a byte. They migrated from the single `settings.storage.
**The public-URL key in `site.json` is `siteUrl`.** The editor form labels the field
"Public URL", so grepping for `publicUrl` finds the UI hint and misses the data.
+**The three file schemas are code, and their key tables are generated.** `settings.json`
+→ [SETTINGS.md](SETTINGS.md) (`common/lib/settingsSchema.ts`), `site.json` →
+[SITE.md](SITE.md) (`common/lib/siteSchema.ts`), a channel's `config.json` →
+[CHANNEL.md](CHANNEL.md) (`common/lib/channelConfigSchema.ts`). A channel's config is
+changed by `patchChannelConfig` (`common/controller/channels.ts`), never by spreading a
+config read earlier; a per-video sidecar is declared once with `sidecar()`
+(`common/lib/sidecar-server.ts`), which refuses a `transcript.<x>.<y>` name.
+
## A channel's `data/` may live on another drive
`channels/<slug>/data` can be an **absolute symlink** to `<root>/<slug>/data` on
diff --git a/CHANNEL.md b/CHANNEL.md
@@ -0,0 +1,68 @@
+# Channel config.json keys
+
+<!-- GENERATED by common/bin/file-schemas-docs.ts from the *_FIELD_DOCS records beside each type — do not edit by hand. -->
+
+One channel of the corpus, persisted to `transcripts/channels/<slug>/config.json`. The schema is `common/lib/channelConfigSchema.ts` over the coercions in `common/lib/channelConfig.ts`. Which sites expose a channel is `site.json`'s business — see [SITE.md](SITE.md); global settings are [SETTINGS.md](SETTINGS.md).
+
+`handling` is the one required key: a file without a valid one is not a channel. The smallest channel is `{ "handling": "youtube", "url": "https://www.youtube.com/@example" }`.
+
+Every other key is optional, and ABSENT MEANS INHERIT: an override that is not set takes the global setting of the same name. So an ill-typed or out-of-range value is not coerced to a default — it is DROPPED, as if the file did not spell it. Unknown keys (including the retired `excludeFromSync`, now a paused `sync` tier in the channel-priority document) are dropped by every read and every write.
+
+The three **sync state** keys are not configuration: the sync, sweep and download passes stamp them, the channel form never does, and they live in the same file on purpose. Every writer after creation PATCHES (`patchChannelConfig`): it re-reads the file at the moment it writes and changes only its own keys, so a stamp and a form save made at once in the editor both land.
+
+Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`.
+
+| Key | Kind | Description |
+|---|---|---|
+| `handling` | required | REQUIRED. `"youtube"` (fetch the platform's captions) or `"transcribe"` (download audio and transcribe it locally). A file without a valid `handling` is not a channel: it reads as null. |
+| `sourceKind` | config | What KIND of source this is: `"video"` (default — yt-dlp + transcription) or `"social"` (an account fetched into the posts corpus, skipped by the video scan). A separate axis from `handling`, so every binary handling branch stays binary. |
+| `postFetcher` | config | Social channels only: which social fetcher drives ingest (e.g. `"bluesky-atproto"`, `"x-gallery-dl"`). Absent = resolve by URL detection. Trimmed. |
+| `socialHandle` | config | Social channels only: the bare account handle (a leading "@" is stripped). Derived from `url` at creation but stored, so a later URL-format change upstream cannot silently re-point ingest at a different account. |
+| `platform` | config | The source platform (youtube, rumble, …). An unknown value is dropped. |
+| `name` | config | Display name. |
+| `url` | config | The channel / playlist / account URL syncs enumerate. Absent = the channel is never auto-synced. |
+| `audioFormat` | config | `"m4a"`, `"mp3"` or `"opus"`: the audio a transcribe-handling download keeps. |
+| `downloadFormat` | config | Per-channel override for the yt-dlp `-f` download format preset. Absent = inherit the global `downloadFormat`, which itself falls back to the per-source "auto" selector. Lets a channel whose source serves full-length audio only in its `original` format (e.g. Odysee) force it. |
+| `keepSourceVideo` | config | Keep the downloaded source video beside the audio. |
+| `keepLatest` | config | Keep-latest window: the newest N videos (by upload date) are protected from the Clean-audio sweep AND have their source video persisted to the saved-video store. 0 or absent = disabled; positives clamp to [1, 100000]. A kept video later found deleted at the source is pinned permanently via the do-not-clean marker. |
+| `extractionMode` | config | `"ytdlp"` (default — yt-dlp's own `-x --audio-format` postprocessor, no source container kept) or `"app"` (yt-dlp downloads the source container and the app runs ffmpeg). The keep-latest persistence rule forces `"app"` for the videos it persists. |
+| `savedVideosDir` | config | Per-channel override for the saved-video store root: this channel's persisted source videos live under `<savedVideosDir>/<slug>/<videoId>/`. Trimmed; blank = the global store. |
+| `dataDir` | config | Where this channel's media ACTUALLY lives when relocated to another drive: the absolute path `channels/<slug>/data` is a symlink to. Absent = in place. Written ONLY by the relocate / re-point jobs on success — a record of what is on disk, never free text, because a value that disagrees with the link is an "inconsistent" channel every guard refuses. |
+| `ytdlpExtraArgs` | config | Extra yt-dlp arguments, appended verbatim. Must be an array of strings or it is dropped. |
+| `subLangs` | config | yt-dlp `--sub-langs` value for caption downloads. |
+| `lastSyncedAt` | sync state | SYNC STATE. When the channel last synced (ISO time). Stamped by every sync, and by a social fetch; read by the scheduler's cadence gate. |
+| `lastFullDownloadAt` | sync state | SYNC STATE. When a full download pass last completed (ISO time). |
+| `lastFullSweepAt` | sync state | SYNC STATE. When this channel last paid for a sync FULL SWEEP — the deep pass that re-enumerates the whole listing to refresh `playlist` and flag videos that have left it. Stamped by the sweep; read by the cadence gate to decide whether the next sync sweeps or stays on the cheap newest-first paged walk. |
+| `excludeFromBuild` | config | Leave this channel out of every site build. |
+| `excludeFromCleanup` | config | Leave this channel's reclaimable bytes out of the aggregate "cleanable data" total on /cleanup and its badge. The per-channel cleanup sweeps stay available; only the running total changes. |
+| `syncIntervalMinutes` | config | Auto-sync cadence: the scheduler syncs this channel when `now - lastSyncedAt >= syncIntervalMinutes`. Absent = inherit the global default; 0 = auto-sync off (still manually syncable); positives clamp to [1, 44640] (~31 days). A missing `url`, or a `sync` tier of paused in the channel-priority document, also disables auto-sync. |
+| `fullSweepIntervalMinutes` | config | Full-sweep cadence: a sync upgrades itself to a full sweep when `now - lastFullSweepAt >= fullSweepIntervalMinutes`. Absent = inherit `syncScheduler.fullSweepIntervalMinutes`; 0 = never sweep (every sync is a paged walk); positives clamp to [1, 44640]. |
+| `skipLiveDownloads` | config | Per-channel override for the global `skipLiveDownloads`. Absent = inherit; false = allow downloading currently-live / upcoming videos. |
+| `downloadFilter` | config | Per-channel title/description download filter, matched against the per-video metadata prefetch. A declined video is SETTLED by a terminal download-outcome keyed on the filter's signature, not by an archive line, so changing either pattern re-evaluates every settled video on the next run. An object with no real rule is dropped (the filter is inert). See [`downloadFilter`](#downloadfilter). |
+| `cookiesFromBrowser` | config | Per-channel override of the global cookies-from-browser spec. Trimmed; blank = inherit. |
+| `cookieMode` | config | Per-channel override of the global cookie mode. Absent = inherit. |
+| `sleepBetweenDownloadsSeconds` | config | Per-channel override for the global pause between downloads. Absent = inherit; 0 = no sleep; floored and capped at 600. |
+| `audioCheck` | config | Opt-in audio-integrity checking for sources that intermittently serve corrupt audio mid-download (e.g. Odysee "original"): the managed downloader periodically validates the in-progress `.part` file and rolls back to the last known-good snapshot on corruption. transcribe-handling only. See [`audioCheck`](#audiocheck). |
+
+## `downloadFilter`
+
+#### `downloadFilter`
+
+| Key | Default | Description |
+|---|---|---|
+| `include` | absent | Case-insensitive regex SOURCE (no delimiters, no flags) a video's `title + "\n" + description` must match to be downloaded. Trimmed; blank = no include rule. |
+| `exclude` | absent | Case-insensitive regex source that rejects a matching video. Wins over `include`. Trimmed; blank = no exclude rule. |
+| `includeLivestreams` | absent | Opt every livestream VOD in, whatever it is called. A SECOND positive selector beside `include`, not a modifier of it — so `{ includeLivestreams: true }` alone rejects plain uploads and passes livestreams. Stored only when `true`. |
+| `rejectedLivestreams` | absent | What to do with a livestream the filter REJECTED: `"skip"` (default, and what every channel predating the field did) or `"chat-only"` — the video is not downloaded, its live chat is, and it joins the corpus as a chat track with no captions. Stored only when not `"skip"`, and only beside a real filter: with nothing to reject it names a decision that can never be taken. |
+
+## `audioCheck`
+
+#### `audioCheck`
+
+| Key | Default | Description |
+|---|---|---|
+| `enabled` | required | Turn the check on. Required: an `audioCheck` object without a boolean `enabled` is dropped whole. |
+| `intervalSeconds` | absent | Seconds between integrity probes of the in-progress `.part` file; clamped to [10, 600]. Absent = 60. The live cadence adapts (AIMD): a malformed checkpoint halves it toward the 10 s floor, clean ones step it back up. |
+| `maxRollbacks` | absent | Rollbacks to the last known-good snapshot before the download is given up; clamped to [1, 20]. Absent = 5. |
+| `copyTimeoutSeconds` | absent | Seconds allowed for the snapshot copy; clamped to [5, 120]. Absent = 30. |
+| `resumeDuringProbe` | absent | When false (default), yt-dlp stays SIGSTOPped across each probe, so it never downloads bytes a malformed verdict would discard and force a re-fetch — minimising HTTP 429 risk. True = the legacy behaviour: resume right after the snapshot copy and probe while the download keeps running. |
diff --git a/SETUP.md b/SETUP.md
@@ -255,7 +255,9 @@ Most configuration now lives in the editor's **/settings** page, persisted to
`settings.json` at the repo root (gitignored). Every key, its default and what it
does is in [SETTINGS.md](SETTINGS.md); `settings.json.example` is the defaults as a
starting template. Both are generated from the settings schema
-(`common/lib/settingsSchema.ts`). Settings are optional — a missing/partial `settings.json` falls
+(`common/lib/settingsSchema.ts`). The per-site `site.json` and the per-channel
+`config.json` have generated key tables of their own: [SITE.md](SITE.md) and
+[CHANNEL.md](CHANNEL.md). Settings are optional — a missing/partial `settings.json` falls
back to built-in defaults, so the app runs out of the box.
Paths and binaries resolve through `getPaths()` in `common/lib/paths.ts`. Override
@@ -265,7 +267,7 @@ any of them via environment variables before launching:
| --- | --- | --- |
| `TRANSCRIPTS_DIR` | `<repo>/transcripts` | Channels, archives, LMDB index, job logs. |
| `SAVED_VIDEOS_DIR` | `<TRANSCRIPTS_DIR>/saved-videos` | Persisted source-video store (can live on a separate disk). |
-| `SITES_DIR` | `<TRANSCRIPTS_DIR>/sites` | Per-site config (`sites/<id>/site.json`). |
+| `SITES_DIR` | `<TRANSCRIPTS_DIR>/sites` | Per-site config (`sites/<id>/site.json` — every key in [SITE.md](SITE.md)). |
| `EXPORT_PUBLIC_DIR` | `<repo>/export/public` | Where the index writes paginated JSON. |
| `SETTINGS_FILE` | `<repo>/settings.json` | Site-settings file. |
| `YTDLP_BIN` | `yt-dlp` (PATH) | Pipeline downloader. |
diff --git a/SITE.md b/SITE.md
@@ -0,0 +1,198 @@
+# site.json keys
+
+<!-- GENERATED by common/bin/file-schemas-docs.ts from the *_FIELD_DOCS records beside each type — do not edit by hand. -->
+
+One public site: its branding, its channel grouping and which channels it exposes, persisted to `transcripts/sites/<id>/site.json` (the directory under `$SITES_DIR` when that is set). The schema is `common/lib/siteSchema.ts`. Global operational settings are `settings.json` — see [SETTINGS.md](SETTINGS.md). The PUBLIC `/site.json` a built site serves is a different file (`common/lib/siteDescriptor.ts`).
+
+Every key is optional. A missing key reads as its default, an ill-typed one as its default (or is dropped, for the optional ones), and an unknown one is dropped on the next save. A save writes only the keys that differ from the default. It is REFUSED when there is no channel group, when `defaultGroupId` names no group, or when a social link's SVG is not safe to inline.
+
+Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`.
+
+| Key | Default |
+|---|---|
+| [`siteId`](#siteid) | the directory name |
+| [`siteTitle`](#sitetitle) | `"Transcript Browser"` |
+| [`siteDescription`](#sitedescription) | `"Browse and search video transcripts"` |
+| [`headerTitle`](#headertitle) | `"Transcript Browser"` |
+| [`homeTagline`](#hometagline) | `""` |
+| [`socialLinks`](#sociallinks) | absent |
+| [`groups`](#groups) | list — see below |
+| [`defaultGroupId`](#defaultgroupid) | `"default"` |
+| [`channels`](#channels) | `[]` |
+| [`cloudflareProject`](#cloudflareproject) | absent |
+| [`accent`](#accent) | absent |
+| [`siteUrl`](#siteurl) | absent |
+| [`relatedSites`](#relatedsites) | `[]` |
+| [`pwa`](#pwa) | `false` |
+| [`archives`](#archives) | `true` |
+| [`duplicates`](#duplicates) | `true` |
+| [`archiveMaxBytes`](#archivemaxbytes) | absent |
+| [`hubUrl`](#huburl) | absent |
+
+## `siteId`
+
+The site's id: a lowercase slug (`[a-z0-9][a-z0-9-]*`), and its directory name under `sites/`. The directory is authoritative — a read takes the id from the path, never from the file.
+
+## `siteTitle`
+
+The site's title (browser tab, manifest, headings).
+
+Default: `"Transcript Browser"`
+
+## `siteDescription`
+
+One-line description (meta description, manifest).
+
+Default: `"Browse and search video transcripts"`
+
+## `headerTitle`
+
+The title shown in the site header.
+
+Default: `"Transcript Browser"`
+
+## `homeTagline`
+
+Tagline under the home page title. Empty = none.
+
+Default: `""`
+
+## `socialLinks`
+
+Per-site social links. ABSENT means inherit the global default (`settings.json` `socialLinks`); an array — even an empty one — overrides it. Each link's SVG must be safe to inline or the save is refused.
+
+Default: absent
+
+#### `socialLinks[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `label` | Visible name, also the accessible label of the icon. |
+| `url` | Link target: http(s), mailto: or a site-relative path. |
+| `svg` | Inline SVG markup. Normalized on save (width/height stripped, fill="currentColor", aria-hidden) and rejected when unsafe (script, foreignObject, event handlers, javascript: URLs) or when it has no viewBox. |
+
+## `groups`
+
+Channel grouping layout for THIS site: the buckets the export UI renders channel checkboxes in, and which are selected by default. At least one is required on save; a file with none reads as one inline fallback group.
+
+#### `groups[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `id` | Group id: a lowercase slug (`[a-z0-9][a-z0-9-]*`), unique within the list. An entry with an invalid or repeated id is dropped. |
+| `name` | Header text. May be blank — the UI then shows no header (and falls back to the id in admin contexts) — but must be a string. Trimmed. |
+| `description` | Optional description under the header. Trimmed; blank = none. |
+| `selectedByDefault` | Whether this group's channels start checked in the export UI's channel filter. Only `true` counts. |
+| `order` | Optional explicit ordering hint (lower first); floored. Unordered groups sort after ordered ones, then by name. |
+| `accent` | Provenance accent, set only in hub mode where each group is a federated site (id = origin): that site's own accent, drawn as a swatch on the group header. Never read from a site.json — absent in single-site mode. |
+| `inline` | Render this group's channels as loose individual chips instead of one collapsible group chip. Stored only when `true`. |
+
+Default:
+
+```json
+[
+ {
+ "id": "default",
+ "name": "All channels",
+ "selectedByDefault": true,
+ "inline": true
+ }
+]
+```
+
+## `defaultGroupId`
+
+The group a channel falls into when its membership names none (or an unknown one). Must name a configured group on save; on read an unknown value resolves to the first group.
+
+Default: `"default"`
+
+## `channels`
+
+The channels this site exposes. A channel absent from this list is not built or deployed for this site even though its data exists in the pool.
+
+#### `channels[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `slug` | Channel slug (its directory name under `transcripts/channels/`). Blank and duplicate slugs are dropped. |
+| `groupId` | Group this channel belongs to WITHIN this site. The same channel can sit in different groups on different sites. A value naming no configured group is dropped on read and falls back to `defaultGroupId` at render time. |
+| `order` | Optional explicit ordering hint within the site (lower first); floored to an integer. |
+
+Default:
+
+```json
+[]
+```
+
+## `cloudflareProject`
+
+Cloudflare Pages project name this site deploys to (`wrangler pages deploy out --project-name <cloudflareProject>`). Trimmed; blank = none.
+
+Default: absent
+
+## `accent`
+
+Per-site brand accent, `"#rrggbb"`. Overrides the family brass on this site's public build. Absent = inherit the family brass. Any other spelling is dropped.
+
+Default: absent
+
+## `siteUrl`
+
+Absolute public URL of this site's deployment, e.g. `https://jeralyzer.pages.dev` (trimmed, trailing slashes removed; anything not absolute http(s) is dropped). Drives the cross-site footer: a site with no siteUrl is omitted from every other site's list.
+
+Default: absent
+
+## `relatedSites`
+
+Pulls specific siblings to the front of the footer's cross-site list, in named groups. Siblings not named here fall into a trailing "Other sites" group. Absent/empty = one flat list of every sibling.
+
+#### `relatedSites[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `label` | Optional muted heading shown above the group; omit for an unlabeled group. |
+| `siteIds` | Sibling site ids, in display order. Invalid and repeated ids are dropped, and a group left with none is dropped. Ids are resolved against the live pool at render time, so an id for a site that does not exist (yet) is harmless — it is skipped. |
+
+Default:
+
+```json
+[]
+```
+
+## `pwa`
+
+Whether this site ships an installable PWA (service worker + web manifest). Default false: a "dumb instance" that serves the CORS-enabled JSON federation contract but is not independently installable, so a visitor trusts only the hub PWA. Stored only when true.
+
+Default: `false`
+
+## `archives`
+
+Whether the site build generates downloadable transcript/live-chat archive zips (and links them on the Downloads page). Opt-OUT: absent/true = on, only an explicit `false` disables. Also gated by the global setting and a per-build flag.
+
+Default: `true`
+
+## `duplicates`
+
+Whether this site publishes the Duplicates page (and its header link). Opt-OUT: absent/true = on, only an explicit `false` hides it. Even when on, the page auto-hides when the site has no in-scope duplicate clusters.
+
+Default: `true`
+
+## `archiveMaxBytes`
+
+Per-site served-file size cap in bytes: any archive larger is dropped from what is served and flagged in the manifest, so a capped host (Cloudflare Pages: 25 MB) will not reject the deploy. 0 = no cap. Absent = the global default. Negative or non-numeric values are dropped.
+
+Default: absent
+
+## `hubUrl`
+
+Per-site override for the hub this site belongs under (the PWA it points visitors toward). Absent = the family default, `settings.json` `homepageUrl`. Surfaced on the public /site.json so a hub can tell member sites from arbitrary added origins.
+
+Default: absent
diff --git a/common/bin/file-schemas-docs.ts b/common/bin/file-schemas-docs.ts
@@ -0,0 +1,58 @@
+#!/usr/bin/env tsx
+// WRITE SITE.md AND CHANNEL.md FROM THE FILE SCHEMAS.
+//
+// Usage (from the repo root):
+// pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts
+// pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts --check
+//
+// `--check` writes nothing and exits 1 if either committed file differs from
+// what the schemas generate (the same claim common/lib/fileSchemaDocs.test.ts
+// makes). The sibling of settings-example.ts (SETTINGS.md).
+//
+// Reads no site.json and no config.json: both outputs are functions of the
+// schemas' docs records alone.
+
+import { readFile, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import {
+ renderChannelMarkdown,
+ renderSiteMarkdown,
+} from "../lib/fileSchemaDocs";
+import { parseFlags } from "./_parseFlags";
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+const FILE_SCHEMA_OUTPUTS: ReadonlyArray<[string, () => string]> = [
+ ["SITE.md", renderSiteMarkdown],
+ ["CHANNEL.md", renderChannelMarkdown],
+];
+
+async function main(): Promise<number> {
+ const flags = parseFlags(process.argv.slice(2));
+ const check = flags.check === "true";
+ let stale = 0;
+ for (const [name, render] of FILE_SCHEMA_OUTPUTS) {
+ const file = path.join(REPO, name);
+ const want = render();
+ if (check) {
+ const have = await readFile(file, "utf8").catch(() => "");
+ if (have !== want) {
+ console.error(`${name} is stale — regenerate it`);
+ stale++;
+ }
+ continue;
+ }
+ await writeFile(file, want);
+ console.log(`wrote ${name}`);
+ }
+ return stale > 0 ? 1 : 0;
+}
+
+main().then(
+ (code) => process.exit(code),
+ (err) => {
+ console.error(err);
+ process.exit(1);
+ },
+);
diff --git a/common/lib/fileSchemaDocs.test.ts b/common/lib/fileSchemaDocs.test.ts
@@ -0,0 +1,44 @@
+import { readFileSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { renderChannelMarkdown, renderSiteMarkdown } from "./fileSchemaDocs";
+import { CHANNEL_CONFIG_KEYS, parseChannelConfig } from "./channelConfig";
+import { SITE_KEYS } from "./siteSchema";
+
+// SITE.md and CHANNEL.md are GENERATED from the file schemas
+// (common/bin/file-schemas-docs.ts). This is what keeps them generated: a hand
+// edit to either file, or a schema change without a regenerate, fails here.
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+for (const [name, render] of [
+ ["SITE.md", renderSiteMarkdown],
+ ["CHANNEL.md", renderChannelMarkdown],
+] as const) {
+ test(`${name} is what the schema generates`, () => {
+ const committed = readFileSync(path.join(REPO, name), "utf8");
+ assert.equal(
+ committed,
+ render(),
+ `${name} is stale: run pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`,
+ );
+ });
+}
+
+test("every key has a row", () => {
+ const site = renderSiteMarkdown();
+ for (const key of SITE_KEYS) assert.ok(site.includes(`| [\`${key}\`]`), key);
+ const channel = renderChannelMarkdown();
+ for (const key of CHANNEL_CONFIG_KEYS) assert.ok(channel.includes(`| \`${key}\` |`), key);
+});
+
+test("CHANNEL.md's smallest channel is a channel", () => {
+ const m = /`(\{ "handling"[^`]*\})`/.exec(renderChannelMarkdown());
+ assert.ok(m, "the one-line example is present");
+ assert.deepEqual(parseChannelConfig(JSON.parse(m![1])), {
+ handling: "youtube",
+ url: "https://www.youtube.com/@example",
+ });
+});
diff --git a/common/lib/fileSchemaDocs.ts b/common/lib/fileSchemaDocs.ts
@@ -0,0 +1,192 @@
+// THE TWO KEY TABLES GENERATED FROM THE FILE SCHEMAS: SITE.md (a site's
+// `site.json`) and CHANNEL.md (a channel's `config.json`), both at the repo
+// root.
+//
+// one-core phase 3 slice 4b; the sibling of settingsDocs.ts (SETTINGS.md), and
+// rendered with its table renderer. Pure renderers — common/bin/
+// file-schemas-docs.ts writes the files, and fileSchemaDocs.test.ts asserts the
+// committed bytes are what these return, so neither can be edited by hand.
+//
+// Everything comes from the schemas' docs records: the keys and their order
+// from SITE_FIELD_DOCS / CHANNEL_CONFIG_FIELD_DOCS, the defaults (site.json only
+// — a channel's optional keys have no default, absent means inherit) from
+// `parseSite(id, {})`, the nested tables from the records beside each type.
+//
+// No `.example` files: a site needs an id (its directory), and the smallest
+// channel is one line, spelled out in CHANNEL.md.
+
+import { CHANNEL_GROUP_FIELD_DOCS } from "./channelGroups";
+import {
+ AUDIO_CHECK_FIELD_DOCS,
+ CHANNEL_CONFIG_FIELD_DOCS,
+ CHANNEL_SYNC_STATE_KEYS,
+ DOWNLOAD_FILTER_FIELD_DOCS,
+} from "./channelConfig";
+import { SOCIAL_LINK_FIELD_DOCS } from "./settingsSchema";
+import {
+ RELATED_SITE_GROUP_FIELD_DOCS,
+ SITE_CHANNEL_MEMBERSHIP_FIELD_DOCS,
+ SITE_FIELD_DOCS,
+ parseSite,
+ type Site,
+} from "./siteSchema";
+import { cell, defaultCell, isScalar, renderTable, type KeyTable } from "./settingsDocs";
+
+const GENERATED =
+ "<!-- GENERATED by common/bin/file-schemas-docs.ts from the *_FIELD_DOCS records beside each type — do not edit by hand. -->";
+
+const REGENERATE =
+ "Regenerate this file with " +
+ "`pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`.";
+
+// The default as a cell, where `undefined` — a key parseSite emits but a file
+// need not spell — is "absent".
+function siteDefaultCell(v: unknown): string {
+ return v === undefined ? "absent" : defaultCell(v);
+}
+
+const SITE_NESTED: Partial<Record<keyof Site, KeyTable[]>> = {
+ socialLinks: [{ path: "socialLinks[]", docs: SOCIAL_LINK_FIELD_DOCS }],
+ groups: [{ path: "groups[]", docs: CHANNEL_GROUP_FIELD_DOCS }],
+ channels: [{ path: "channels[]", docs: SITE_CHANNEL_MEMBERSHIP_FIELD_DOCS }],
+ relatedSites: [{ path: "relatedSites[]", docs: RELATED_SITE_GROUP_FIELD_DOCS }],
+};
+
+export function renderSiteMarkdown(): string {
+ const d = parseSite("<id>", {}) as Record<string, unknown>;
+ const keys = Object.keys(SITE_FIELD_DOCS) as Array<keyof Site>;
+ const out: string[] = [];
+ out.push("# site.json keys");
+ out.push("");
+ out.push(GENERATED);
+ out.push("");
+ out.push(
+ "One public site: its branding, its channel grouping and which channels " +
+ "it exposes, persisted to `transcripts/sites/<id>/site.json` (the " +
+ "directory under `$SITES_DIR` when that is set). The schema is " +
+ "`common/lib/siteSchema.ts`. Global operational settings are " +
+ "`settings.json` — see [SETTINGS.md](SETTINGS.md). The PUBLIC " +
+ "`/site.json` a built site serves is a different file " +
+ "(`common/lib/siteDescriptor.ts`).",
+ );
+ out.push("");
+ out.push(
+ "Every key is optional. A missing key reads as its default, an ill-typed " +
+ "one as its default (or is dropped, for the optional ones), and an " +
+ "unknown one is dropped on the next save. A save writes only the keys " +
+ "that differ from the default. It is REFUSED when there is no channel " +
+ "group, when `defaultGroupId` names no group, or when a social link's " +
+ "SVG is not safe to inline.",
+ );
+ out.push("");
+ out.push(REGENERATE);
+ out.push("");
+ out.push("| Key | Default |");
+ out.push("|---|---|");
+ for (const key of keys) {
+ const def = key === "siteId" ? "the directory name" : siteDefaultCell(d[key]);
+ out.push(`| [\`${key}\`](#${key.toLowerCase()}) | ${def} |`);
+ }
+ out.push("");
+ for (const key of keys) {
+ out.push(`## \`${key}\``);
+ out.push("");
+ out.push(SITE_FIELD_DOCS[key]);
+ out.push("");
+ const v = d[key];
+ if (key === "siteId") continue;
+ if (v === undefined || isScalar(v)) {
+ out.push(`Default: ${siteDefaultCell(v)}`);
+ out.push("");
+ for (const table of SITE_NESTED[key] ?? []) renderTable(out, table);
+ } else {
+ for (const table of SITE_NESTED[key] ?? []) renderTable(out, table);
+ out.push("Default:");
+ out.push("");
+ out.push("```json");
+ out.push(JSON.stringify(v, null, 2));
+ out.push("```");
+ out.push("");
+ }
+ }
+ return out.join("\n");
+}
+
+// A nested block's keys are, like the top level's, absent unless spelled —
+// except `audioCheck.enabled`, without which the block is dropped.
+const CHANNEL_NESTED: Partial<Record<string, KeyTable[]>> = {
+ downloadFilter: [
+ { path: "downloadFilter", docs: DOWNLOAD_FILTER_FIELD_DOCS, defaults: () => "absent" },
+ ],
+ audioCheck: [
+ {
+ path: "audioCheck",
+ docs: AUDIO_CHECK_FIELD_DOCS,
+ defaults: (key) => (key === "enabled" ? "required" : "absent"),
+ },
+ ],
+};
+
+function channelKind(key: string): string {
+ if (key === "handling") return "required";
+ if ((CHANNEL_SYNC_STATE_KEYS as readonly string[]).includes(key)) return "sync state";
+ return "config";
+}
+
+export function renderChannelMarkdown(): string {
+ const keys = Object.keys(CHANNEL_CONFIG_FIELD_DOCS);
+ const out: string[] = [];
+ out.push("# Channel config.json keys");
+ out.push("");
+ out.push(GENERATED);
+ out.push("");
+ out.push(
+ "One channel of the corpus, persisted to " +
+ "`transcripts/channels/<slug>/config.json`. The schema is " +
+ "`common/lib/channelConfigSchema.ts` over the coercions in " +
+ "`common/lib/channelConfig.ts`. Which sites expose a channel is " +
+ "`site.json`'s business — see [SITE.md](SITE.md); global settings are " +
+ "[SETTINGS.md](SETTINGS.md).",
+ );
+ out.push("");
+ out.push(
+ "`handling` is the one required key: a file without a valid one is not a " +
+ "channel. The smallest channel is " +
+ '`{ "handling": "youtube", "url": "https://www.youtube.com/@example" }`.',
+ );
+ out.push("");
+ out.push(
+ "Every other key is optional, and ABSENT MEANS INHERIT: an override that " +
+ "is not set takes the global setting of the same name. So an ill-typed " +
+ "or out-of-range value is not coerced to a default — it is DROPPED, as " +
+ "if the file did not spell it. Unknown keys (including the retired " +
+ "`excludeFromSync`, now a paused `sync` tier in the channel-priority " +
+ "document) are dropped by every read and every write.",
+ );
+ out.push("");
+ out.push(
+ "The three **sync state** keys are not configuration: the sync, sweep and " +
+ "download passes stamp them, the channel form never does, and they live " +
+ "in the same file on purpose. Every writer after creation PATCHES " +
+ "(`patchChannelConfig`): it re-reads the file at the moment it writes " +
+ "and changes only its own keys, so a stamp and a form save made at once " +
+ "in the editor both land.",
+ );
+ out.push("");
+ out.push(REGENERATE);
+ out.push("");
+ out.push("| Key | Kind | Description |");
+ out.push("|---|---|---|");
+ for (const key of keys) {
+ const docs = (CHANNEL_CONFIG_FIELD_DOCS as Record<string, string>)[key];
+ const nested = CHANNEL_NESTED[key] ? ` See [\`${key}\`](#${key.toLowerCase()}).` : "";
+ out.push(`| \`${key}\` | ${channelKind(key)} | ${cell(docs)}${nested} |`);
+ }
+ out.push("");
+ for (const key of Object.keys(CHANNEL_NESTED)) {
+ out.push(`## \`${key}\``);
+ out.push("");
+ for (const table of CHANNEL_NESTED[key] ?? []) renderTable(out, table);
+ }
+ return out.join("\n");
+}
diff --git a/common/lib/settingsDocs.ts b/common/lib/settingsDocs.ts
@@ -64,13 +64,13 @@ export function renderSettingsExample(): string {
return JSON.stringify(d, null, 2) + "\n";
}
-function isScalar(v: unknown): boolean {
+export function isScalar(v: unknown): boolean {
return v === null || typeof v !== "object";
}
// The default as a table cell: a scalar inline, an empty container inline,
// anything larger by reference to its section.
-function defaultCell(v: unknown): string {
+export function defaultCell(v: unknown): string {
if (isScalar(v)) return "`" + JSON.stringify(v) + "`";
const json = JSON.stringify(v);
if (json === "[]" || json === "{}") return "`" + json + "`";
@@ -78,20 +78,20 @@ function defaultCell(v: unknown): string {
}
// A description inside a table cell: one line, pipes escaped, paragraphs kept.
-function cell(text: string): string {
+export function cell(text: string): string {
return text.replace(/\|/g, "\\|").replace(/\n\n/g, "<br><br>").replace(/\n/g, " ");
}
// One nested key table. `defaults(key)` answers the Default column; a table of
// per-entry fields (list items, map values, tree nodes) has no defaults — each
// entry spells its own — and says so.
-type KeyTable = {
+export type KeyTable = {
path: string;
docs: Readonly<Record<string, string>>;
defaults?: (key: string) => string;
};
-function fromObject(obj: unknown): (key: string) => string {
+export function fromObject(obj: unknown): (key: string) => string {
const r = (obj ?? {}) as Record<string, unknown>;
return (key) => (key in r ? defaultCell(r[key]) : "absent");
}
@@ -208,7 +208,8 @@ export function blockTables(d: SiteSettings): Partial<Record<keyof SiteSettings,
};
}
-function renderTable(out: string[], table: KeyTable): void {
+// Shared with lib/fileSchemaDocs.ts (SITE.md, CHANNEL.md).
+export function renderTable(out: string[], table: KeyTable): void {
out.push(`#### \`${table.path}\``);
out.push("");
if (table.defaults) {
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -131,10 +131,13 @@ source tree).
(`DisplaySummary` records), documented in `common/lib/corpus.ts:35-36` and served by
`common/components/summariesCache.ts`. Nothing to do with AI summaries. Use **`digest`**.
-**Never name a sidecar `transcript.<x>.<y>`.** `common/lib/videoStatus.ts:137` —
-`const SUB_FILE_RE = /^transcript\.([^.]+)\.([^.]+)$/;` treats any such file as a subtitle
-track. The exclusion list in `readSubTracks` (`:238-256`) is hardcoded. Use
-`ai-digest.json`, `diarization.json`, `attribution.json`.
+**Never name a sidecar `transcript.<x>.<y>`.** `common/lib/videoStatus.ts:88` —
+`export const SUB_FILE_RE = /^transcript\.([^.]+)\.([^.]+)$/;` treats any such file as a
+subtitle track. The exclusion list in `readSubTracks` is hardcoded. Use `ai-digest.json`,
+`diarization.json`, `attribution.json`. Since one-core phase 3 slice 4b (2026-09-24) a
+sidecar is declared with `sidecar(filename, schema)` (`common/lib/sidecar-server.ts`), which
+THROWS at declaration — module load — on a name `SUB_FILE_RE` matches;
+`sidecar-server.test.ts` enumerates `SIDECAR_FILENAMES`.
**`report` already means three things**: `reportDebouncePreset` in settings, the
`refresh-report` job kind, and `/ask`'s `ReportPanel`. Use `feedback` / `review` instead.