commit edd9fb3bd831490e1d32a045d137150399feb4da
parent a9f5961e3182e4cb65d78f6fe5152e37970402db
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 24 Sep 2026 16:41:11 -0400
Merge one-core/phase-3-s4b — the rest of the schemas: site.json, channel config.json, sidecars
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
54 files changed, 3349 insertions(+), 1131 deletions(-)
diff --git a/AGENTS.md b/AGENTS.md
@@ -193,8 +193,8 @@ apps*, not [DEPLOY_DOCKER.md](DEPLOY_DOCKER.md), which is about *building sites*
| Path | What it is |
|---|---|
-| `transcripts/channels/<slug>/` | One channel: `config.json`, `playlist`, `archive`, `snapshot.json`, and `data/<videoId>/` holding media, transcripts and sidecars. **`data/` may be an absolute SYMLINK** — see below. |
-| `transcripts/sites/<id>/site.json` | **Per-site config, including the public URL.** This is where deployed-site facts live — *not* under `channels/`. |
+| `transcripts/channels/<slug>/` | One channel: `config.json` (every key in [CHANNEL.md](CHANNEL.md)), `playlist`, `archive`, `snapshot.json`, and `data/<videoId>/` holding media, transcripts and sidecars. **`data/` may be an absolute SYMLINK** — see below. |
+| `transcripts/sites/<id>/site.json` | **Per-site config, including the public URL** (every key in [SITE.md](SITE.md)). This is where deployed-site facts live — *not* under `channels/`. |
| `transcripts/index.mdb` | The LMDB transcript index. Key-only range scans over its `byChannel` sub-DB are cheap; see `common/controller/recencyIndex.ts`. |
| `transcripts/saved-videos/` | Persisted source-video store. |
| `transcripts/search-aliases.json`, `duplicates*.json` | Corpus-wide curated data. |
@@ -220,6 +220,14 @@ new path without moving a byte. They migrated from the single `settings.storage.
**The public-URL key in `site.json` is `siteUrl`.** The editor form labels the field
"Public URL", so grepping for `publicUrl` finds the UI hint and misses the data.
+**The three file schemas are code, and their key tables are generated.** `settings.json`
+→ [SETTINGS.md](SETTINGS.md) (`common/lib/settingsSchema.ts`), `site.json` →
+[SITE.md](SITE.md) (`common/lib/siteSchema.ts`), a channel's `config.json` →
+[CHANNEL.md](CHANNEL.md) (`common/lib/channelConfigSchema.ts`). A channel's config is
+changed by `patchChannelConfig` (`common/controller/channels.ts`), never by spreading a
+config read earlier; a per-video sidecar is declared once with `sidecar()`
+(`common/lib/sidecar-server.ts`), which refuses a `transcript.<x>.<y>` name.
+
## A channel's `data/` may live on another drive
`channels/<slug>/data` can be an **absolute symlink** to `<root>/<slug>/data` on
diff --git a/CHANNEL.md b/CHANNEL.md
@@ -0,0 +1,64 @@
+# Channel config.json keys
+
+<!-- GENERATED by common/bin/file-schemas-docs.ts from the *_FIELD_DOCS records beside each type — do not edit by hand. -->
+
+One channel of the corpus, persisted to `transcripts/channels/<slug>/config.json`. The schema is `common/lib/channelConfigSchema.ts` over the coercions in `common/lib/channelConfig.ts`. Which sites expose a channel is `site.json`'s business — see [SITE.md](SITE.md); global settings are [SETTINGS.md](SETTINGS.md).
+
+`handling` is the one required key: a file without a valid one is not a channel. The smallest channel is `{ "handling": "youtube", "url": "https://www.youtube.com/@example" }`.
+
+Every other key is optional and has NO default of its own: an absent key means whatever its description says — for the per-channel overrides, inherit the global setting of the same name; for `name`, `url`, `dataDir`, `subLangs` and the sync-state stamps, simply unset. So an ill-typed or out-of-range value is not coerced — it is DROPPED, as if the file did not spell it. Unknown keys (including the retired `excludeFromSync`, now a paused `sync` tier in the channel-priority document) are dropped by every read and every write.
+
+The three **sync state** keys are not configuration: the sync, sweep and download passes stamp them, the channel form never does, and they live in the same file on purpose. Writers after creation PATCH (`patchChannelConfig`): each re-reads the file at the moment it writes and changes only its own keys, so a stamp and a form save made at once in the editor both land. The two exceptions write a whole config, and only when there is no readable file to patch: a media move and a channel rename record `dataDir` from their own copy of the config.
+
+Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`.
+
+| Key | Kind | Description |
+|---|---|---|
+| `handling` | required | REQUIRED. `"youtube"` (fetch the platform's captions) or `"transcribe"` (download audio and transcribe it locally). A file without a valid `handling` is not a channel: it reads as null. |
+| `sourceKind` | config | What KIND of source this is: `"video"` (default — yt-dlp + transcription) or `"social"` (an account fetched into the posts corpus, skipped by the video scan). A separate axis from `handling`, so every binary handling branch stays binary. |
+| `postFetcher` | config | Social channels only: which social fetcher drives ingest (e.g. `"bluesky-atproto"`, `"x-gallery-dl"`). Absent = resolve by URL detection. Trimmed. |
+| `socialHandle` | config | Social channels only: the bare account handle (a leading "@" is stripped). Derived from `url` at creation but stored, so a later URL-format change upstream cannot silently re-point ingest at a different account. |
+| `platform` | config | The source platform (youtube, rumble, …). An unknown value is dropped. |
+| `name` | config | Display name. |
+| `url` | config | The channel / playlist / account URL syncs enumerate. Absent = the channel is never auto-synced. |
+| `audioFormat` | config | `"m4a"`, `"mp3"` or `"opus"`: the audio a transcribe-handling download keeps. |
+| `downloadFormat` | config | Per-channel override for the yt-dlp `-f` download format preset. Absent = inherit the global `downloadFormat`, which itself falls back to the per-source "auto" selector. Lets a channel whose source serves full-length audio only in its `original` format (e.g. Odysee) force it. |
+| `keepSourceVideo` | config | Keep the downloaded source video beside the audio. |
+| `keepLatest` | config | Keep-latest window: the newest N videos (by upload date) are protected from the Clean-audio sweep AND have their source video persisted to the saved-video store. 0 or absent = disabled; positives clamp to [1, 100000]. A kept video later found deleted at the source is pinned permanently via the do-not-clean marker. |
+| `extractionMode` | config | `"ytdlp"` (default — yt-dlp's own `-x --audio-format` postprocessor, no source container kept) or `"app"` (yt-dlp downloads the source container and the app runs ffmpeg). The keep-latest persistence rule forces `"app"` for the videos it persists. |
+| `savedVideosDir` | config | Per-channel override for the saved-video store root: this channel's persisted source videos live under `<savedVideosDir>/<slug>/<videoId>/`. Trimmed; blank = the global store. |
+| `dataDir` | config | Where this channel's media ACTUALLY lives when relocated to another drive: the absolute path `channels/<slug>/data` is a symlink to. Absent = in place. Written ONLY by the relocate / re-point jobs on success — a record of what is on disk, never free text, because a value that disagrees with the link is an "inconsistent" channel every guard refuses. |
+| `ytdlpExtraArgs` | config | Extra yt-dlp arguments, appended verbatim. Must be an array of strings or it is dropped. |
+| `subLangs` | config | yt-dlp `--sub-langs` value for caption downloads. |
+| `lastSyncedAt` | sync state | SYNC STATE. When the channel last synced (ISO time). Stamped by every sync, and by a social fetch; read by the scheduler's cadence gate. |
+| `lastFullDownloadAt` | sync state | SYNC STATE. When a full download pass last completed (ISO time). |
+| `lastFullSweepAt` | sync state | SYNC STATE. When this channel last paid for a sync FULL SWEEP — the deep pass that re-enumerates the whole listing to refresh `playlist` and flag videos that have left it. Stamped by the sweep; read by the cadence gate to decide whether the next sync sweeps or stays on the cheap newest-first paged walk. |
+| `excludeFromBuild` | config | Leave this channel out of every site build. |
+| `excludeFromCleanup` | config | Leave this channel's reclaimable bytes out of the aggregate "cleanable data" total on /cleanup and its badge. The per-channel cleanup sweeps stay available; only the running total changes. |
+| `syncIntervalMinutes` | config | Auto-sync cadence: the scheduler syncs this channel when `now - lastSyncedAt >= syncIntervalMinutes`. Absent = inherit the global default; 0 = auto-sync off (still manually syncable); positives clamp to [1, 44640] (~31 days). A missing `url`, or a `sync` tier of paused in the channel-priority document, also disables auto-sync. |
+| `fullSweepIntervalMinutes` | config | Full-sweep cadence: a sync upgrades itself to a full sweep when `now - lastFullSweepAt >= fullSweepIntervalMinutes`. Absent = inherit `syncScheduler.fullSweepIntervalMinutes`; 0 = never sweep (every sync is a paged walk); positives clamp to [1, 44640]. |
+| `skipLiveDownloads` | config | Per-channel override for the global `skipLiveDownloads`. Absent = inherit; false = allow downloading currently-live / upcoming videos. |
+| `downloadFilter` | config | Per-channel title/description download filter, matched against the per-video metadata prefetch. A declined video is SETTLED by a terminal download-outcome keyed on the filter's signature, not by an archive line, so changing either pattern re-evaluates every settled video on the next run. An object with no real rule is dropped (the filter is inert). See [`downloadFilter`](#downloadfilter). |
+| `cookiesFromBrowser` | config | Per-channel override of the global cookies-from-browser spec. Trimmed; blank = inherit. |
+| `cookieMode` | config | Per-channel override of the global cookie mode. Absent = inherit. |
+| `sleepBetweenDownloadsSeconds` | config | Per-channel override for the global pause between downloads. Absent = inherit; 0 = no sleep; floored and capped at 600. |
+| `audioCheck` | config | Opt-in audio-integrity checking for sources that intermittently serve corrupt audio mid-download (e.g. Odysee "original"): the managed downloader periodically validates the in-progress `.part` file and rolls back to the last known-good snapshot on corruption. transcribe-handling only. See [`audioCheck`](#audiocheck). |
+
+#### `downloadFilter`
+
+| Key | Default | Description |
+|---|---|---|
+| `include` | absent | Case-insensitive regex SOURCE (no delimiters, no flags) a video's `title + "\n" + description` must match to be downloaded. Trimmed; blank = no include rule. |
+| `exclude` | absent | Case-insensitive regex source that rejects a matching video. Wins over `include`. Trimmed; blank = no exclude rule. |
+| `includeLivestreams` | absent | Opt every livestream VOD in, whatever it is called. A SECOND positive selector beside `include`, not a modifier of it — so `{ includeLivestreams: true }` alone rejects plain uploads and passes livestreams. Stored only when `true`. |
+| `rejectedLivestreams` | absent | What to do with a livestream the filter REJECTED: `"skip"` (default, and what every channel predating the field did) or `"chat-only"` — the video is not downloaded, its live chat is, and it joins the corpus as a chat track with no captions. Stored only when not `"skip"`, and only beside a real filter: with nothing to reject it names a decision that can never be taken. |
+
+#### `audioCheck`
+
+| Key | Default | Description |
+|---|---|---|
+| `enabled` | required | Turn the check on. Required: an `audioCheck` object without a boolean `enabled` is dropped whole. |
+| `intervalSeconds` | absent | Seconds between integrity probes of the in-progress `.part` file; clamped to [10, 600]. Absent = 60. The live cadence adapts (AIMD): a malformed checkpoint halves it toward the 10 s floor, clean ones step it back up. |
+| `maxRollbacks` | absent | Rollbacks to the last known-good snapshot before the download is given up; clamped to [1, 20]. Absent = 5. |
+| `copyTimeoutSeconds` | absent | Seconds allowed for the snapshot copy; clamped to [5, 120]. Absent = 30. |
+| `resumeDuringProbe` | absent | When false (default), yt-dlp stays SIGSTOPped across each probe, so it never downloads bytes a malformed verdict would discard and force a re-fetch — minimising HTTP 429 risk. True = the legacy behaviour: resume right after the snapshot copy and probe while the download keeps running. |
diff --git a/SETUP.md b/SETUP.md
@@ -255,7 +255,9 @@ Most configuration now lives in the editor's **/settings** page, persisted to
`settings.json` at the repo root (gitignored). Every key, its default and what it
does is in [SETTINGS.md](SETTINGS.md); `settings.json.example` is the defaults as a
starting template. Both are generated from the settings schema
-(`common/lib/settingsSchema.ts`). Settings are optional — a missing/partial `settings.json` falls
+(`common/lib/settingsSchema.ts`). The per-site `site.json` and the per-channel
+`config.json` have generated key tables of their own: [SITE.md](SITE.md) and
+[CHANNEL.md](CHANNEL.md). Settings are optional — a missing/partial `settings.json` falls
back to built-in defaults, so the app runs out of the box.
Paths and binaries resolve through `getPaths()` in `common/lib/paths.ts`. Override
@@ -265,7 +267,7 @@ any of them via environment variables before launching:
| --- | --- | --- |
| `TRANSCRIPTS_DIR` | `<repo>/transcripts` | Channels, archives, LMDB index, job logs. |
| `SAVED_VIDEOS_DIR` | `<TRANSCRIPTS_DIR>/saved-videos` | Persisted source-video store (can live on a separate disk). |
-| `SITES_DIR` | `<TRANSCRIPTS_DIR>/sites` | Per-site config (`sites/<id>/site.json`). |
+| `SITES_DIR` | `<TRANSCRIPTS_DIR>/sites` | Per-site config (`sites/<id>/site.json` — every key in [SITE.md](SITE.md)). |
| `EXPORT_PUBLIC_DIR` | `<repo>/export/public` | Where the index writes paginated JSON. |
| `SETTINGS_FILE` | `<repo>/settings.json` | Site-settings file. |
| `YTDLP_BIN` | `yt-dlp` (PATH) | Pipeline downloader. |
diff --git a/SITE.md b/SITE.md
@@ -0,0 +1,198 @@
+# site.json keys
+
+<!-- GENERATED by common/bin/file-schemas-docs.ts from the *_FIELD_DOCS records beside each type — do not edit by hand. -->
+
+One public site: its branding, its channel grouping and which channels it exposes, persisted to `transcripts/sites/<id>/site.json` (the directory under `$SITES_DIR` when that is set). The schema is `common/lib/siteSchema.ts`. Global operational settings are `settings.json` — see [SETTINGS.md](SETTINGS.md). The PUBLIC `/site.json` a built site serves is a different file (`common/lib/siteDescriptor.ts`).
+
+Every key is optional on read. A missing key reads as its default, an ill-typed one as its default (or is dropped, for the optional ones), and an unknown one is dropped on the next save. A save ALWAYS writes `siteId`, `siteTitle`, `siteDescription`, `headerTitle`, `homeTagline`, `groups`, `defaultGroupId` and `channels`; every other key is written only when it differs from its default (`socialLinks` whenever it is an array, even an empty one). A save is REFUSED when there is no channel group, when `defaultGroupId` names no group, or when a social link's SVG is not safe to inline.
+
+Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`.
+
+| Key | Default |
+|---|---|
+| [`siteId`](#siteid) | the directory name |
+| [`siteTitle`](#sitetitle) | `"Transcript Browser"` |
+| [`siteDescription`](#sitedescription) | `"Browse and search video transcripts"` |
+| [`headerTitle`](#headertitle) | `"Transcript Browser"` |
+| [`homeTagline`](#hometagline) | `""` |
+| [`socialLinks`](#sociallinks) | absent |
+| [`groups`](#groups) | list — see below |
+| [`defaultGroupId`](#defaultgroupid) | `"default"` |
+| [`channels`](#channels) | `[]` |
+| [`cloudflareProject`](#cloudflareproject) | absent |
+| [`accent`](#accent) | absent |
+| [`siteUrl`](#siteurl) | absent |
+| [`relatedSites`](#relatedsites) | `[]` |
+| [`pwa`](#pwa) | `false` |
+| [`archives`](#archives) | `true` |
+| [`duplicates`](#duplicates) | `true` |
+| [`archiveMaxBytes`](#archivemaxbytes) | absent |
+| [`hubUrl`](#huburl) | absent |
+
+## `siteId`
+
+The site's id: a lowercase slug (`[a-z0-9][a-z0-9-]*`), and its directory name under `sites/`. The directory is authoritative — a read takes the id from the path, never from the file.
+
+## `siteTitle`
+
+The site's title (browser tab, manifest, headings).
+
+Default: `"Transcript Browser"`
+
+## `siteDescription`
+
+One-line description (meta description, manifest).
+
+Default: `"Browse and search video transcripts"`
+
+## `headerTitle`
+
+The title shown in the site header.
+
+Default: `"Transcript Browser"`
+
+## `homeTagline`
+
+Tagline under the home page title. Empty = none.
+
+Default: `""`
+
+## `socialLinks`
+
+Per-site social links. ABSENT means inherit the global default (`settings.json` `socialLinks`); an array — even an empty one — overrides it. Each link's SVG must be safe to inline or the save is refused.
+
+Default: absent
+
+#### `socialLinks[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `label` | Visible name, also the accessible label of the icon. |
+| `url` | Link target: http(s), mailto: or a site-relative path. |
+| `svg` | Inline SVG markup. Normalized on save (width/height stripped, fill="currentColor", aria-hidden) and rejected when unsafe (script, foreignObject, event handlers, javascript: URLs) or when it has no viewBox. |
+
+## `groups`
+
+Channel grouping layout for THIS site: the buckets the export UI renders channel checkboxes in, and which are selected by default. At least one is required on save; a file with none reads as one inline fallback group.
+
+#### `groups[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `id` | Group id: a lowercase slug (`[a-z0-9][a-z0-9-]*`), unique within the list. An entry with an invalid or repeated id is dropped. |
+| `name` | Header text. May be blank — the UI then shows no header (and falls back to the id in admin contexts) — but must be a string. Trimmed. |
+| `description` | Optional description under the header. Trimmed; blank = none. |
+| `selectedByDefault` | Whether this group's channels start checked in the export UI's channel filter. Only `true` counts. |
+| `order` | Optional explicit ordering hint (lower first); floored. Unordered groups sort after ordered ones, then by name. |
+| `accent` | Provenance accent, set only in hub mode where each group is a federated site (id = origin): that site's own accent, drawn as a swatch on the group header. Never read from a site.json — absent in single-site mode. |
+| `inline` | Render this group's channels as loose individual chips instead of one collapsible group chip. Stored only when `true`. |
+
+Default:
+
+```json
+[
+ {
+ "id": "default",
+ "name": "All channels",
+ "selectedByDefault": true,
+ "inline": true
+ }
+]
+```
+
+## `defaultGroupId`
+
+The group a channel falls into when its membership names none (or an unknown one). Must name a configured group on save; on read an unknown value resolves to the first group.
+
+Default: `"default"`
+
+## `channels`
+
+The channels this site exposes. A channel absent from this list is not built or deployed for this site even though its data exists in the pool.
+
+#### `channels[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `slug` | Channel slug (its directory name under `transcripts/channels/`). Blank and duplicate slugs are dropped. |
+| `groupId` | Group this channel belongs to WITHIN this site. The same channel can sit in different groups on different sites. A value naming no configured group is dropped on read and falls back to `defaultGroupId` at render time. |
+| `order` | Optional explicit ordering hint within the site (lower first); floored to an integer. |
+
+Default:
+
+```json
+[]
+```
+
+## `cloudflareProject`
+
+Cloudflare Pages project name this site deploys to (`wrangler pages deploy out --project-name <cloudflareProject>`). Trimmed; blank = none.
+
+Default: absent
+
+## `accent`
+
+Per-site brand accent, `"#rrggbb"`. Overrides the family brass on this site's public build. Absent = inherit the family brass. Any other spelling is dropped.
+
+Default: absent
+
+## `siteUrl`
+
+Absolute public URL of this site's deployment, e.g. `https://jeralyzer.pages.dev` (trimmed, trailing slashes removed; anything not absolute http(s) is dropped). Drives the cross-site footer: a site with no siteUrl is omitted from every other site's list.
+
+Default: absent
+
+## `relatedSites`
+
+Pulls specific siblings to the front of the footer's cross-site list, in named groups. Siblings not named here fall into a trailing "Other sites" group. Absent/empty = one flat list of every sibling.
+
+#### `relatedSites[]`
+
+Per entry — each entry spells its own values.
+
+| Key | Description |
+|---|---|
+| `label` | Optional muted heading shown above the group; omit for an unlabeled group. |
+| `siteIds` | Sibling site ids, in display order. Invalid and repeated ids are dropped, and a group left with none is dropped. Ids are resolved against the live pool at render time, so an id for a site that does not exist (yet) is harmless — it is skipped. |
+
+Default:
+
+```json
+[]
+```
+
+## `pwa`
+
+Whether this site ships an installable PWA (service worker + web manifest). Default false: a "dumb instance" that serves the CORS-enabled JSON federation contract but is not independently installable, so a visitor trusts only the hub PWA. Stored only when true.
+
+Default: `false`
+
+## `archives`
+
+Whether the site build generates downloadable transcript/live-chat archive zips (and links them on the Downloads page). Opt-OUT: absent/true = on, only an explicit `false` disables. Also gated by the global setting and a per-build flag.
+
+Default: `true`
+
+## `duplicates`
+
+Whether this site publishes the Duplicates page (and its header link). Opt-OUT: absent/true = on, only an explicit `false` hides it. Even when on, the page auto-hides when the site has no in-scope duplicate clusters.
+
+Default: `true`
+
+## `archiveMaxBytes`
+
+Per-site served-file size cap in bytes: any archive larger is dropped from what is served and flagged in the manifest, so a capped host (Cloudflare Pages: 25 MB) will not reject the deploy. 0 = no cap. Absent = the global default. Negative or non-numeric values are dropped.
+
+Default: absent
+
+## `hubUrl`
+
+Per-site override for the hub this site belongs under (the PWA it points visitors toward). Absent = the family default, `settings.json` `homepageUrl`. Surfaced on the public /site.json so a hub can tell member sites from arbitrary added origins.
+
+Default: absent
diff --git a/common/bin/compose-homepage.ts b/common/bin/compose-homepage.ts
@@ -12,8 +12,9 @@
// prebuild chains it), same as the export pipeline.
import path from "node:path";
-import { mkdir, readFile, rename, writeFile } from "node:fs/promises";
+import { mkdir, readFile } from "node:fs/promises";
import { getPaths } from "../lib/paths";
+import { writeJsonAtomic as writeJsonAtomicShared } from "../lib/jsonFile-server";
import { buildStats } from "../controller/buildStats";
import { listSites } from "../lib/site";
import {
@@ -35,10 +36,9 @@ function homepagePublicDir(monorepoRoot: string): string {
);
}
-async function writeJsonAtomic(filePath: string, value: unknown): Promise<void> {
- const tmp = `${filePath}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(value));
- await rename(tmp, filePath);
+// Compact, no trailing newline — the homepage files' historical bytes.
+function writeJsonAtomic(filePath: string, value: unknown): Promise<void> {
+ return writeJsonAtomicShared(filePath, value, { indent: 0, newline: false });
}
// Read the whole-pool stats dataset back from the pages buildStats just wrote, so
diff --git a/common/bin/file-schemas-docs.ts b/common/bin/file-schemas-docs.ts
@@ -0,0 +1,58 @@
+#!/usr/bin/env tsx
+// WRITE SITE.md AND CHANNEL.md FROM THE FILE SCHEMAS.
+//
+// Usage (from the repo root):
+// pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts
+// pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts --check
+//
+// `--check` writes nothing and exits 1 if either committed file differs from
+// what the schemas generate (the same claim common/lib/fileSchemaDocs.test.ts
+// makes). The sibling of settings-example.ts (SETTINGS.md).
+//
+// Reads no site.json and no config.json: both outputs are functions of the
+// schemas' docs records alone.
+
+import { readFile, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import {
+ renderChannelMarkdown,
+ renderSiteMarkdown,
+} from "../lib/fileSchemaDocs";
+import { parseFlags } from "./_parseFlags";
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+const FILE_SCHEMA_OUTPUTS: ReadonlyArray<[string, () => string]> = [
+ ["SITE.md", renderSiteMarkdown],
+ ["CHANNEL.md", renderChannelMarkdown],
+];
+
+async function main(): Promise<number> {
+ const flags = parseFlags(process.argv.slice(2));
+ const check = flags.check === "true";
+ let stale = 0;
+ for (const [name, render] of FILE_SCHEMA_OUTPUTS) {
+ const file = path.join(REPO, name);
+ const want = render();
+ if (check) {
+ const have = await readFile(file, "utf8").catch(() => "");
+ if (have !== want) {
+ console.error(`${name} is stale — regenerate it`);
+ stale++;
+ }
+ continue;
+ }
+ await writeFile(file, want);
+ console.log(`wrote ${name}`);
+ }
+ return stale > 0 ? 1 : 0;
+}
+
+main().then(
+ (code) => process.exit(code),
+ (err) => {
+ console.error(err);
+ process.exit(1);
+ },
+);
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -24,6 +24,7 @@ import { createWriteStream } from "node:fs";
import type { Dirent, WriteStream } from "node:fs";
import { once } from "node:events";
import { open } from "lmdb";
+import { writeJsonAtomic as writeJsonAtomicShared } from "../lib/jsonFile-server";
import { parseVtt, type Cue } from "../lib/vtt";
import { parseTranscriptJson } from "../lib/whisper";
import { parseLiveChat } from "../lib/liveChat";
@@ -72,10 +73,10 @@ import {
siteSubsDir,
} from "../lib/site";
import {
- parseChannelConfig,
type ChannelConfig,
type ChannelHandling,
} from "../lib/channelConfig";
+import { readChannelConfigFile } from "./channels";
import { resolveChannelGroupId } from "../lib/channelGroups";
import type { Paths } from "../lib/paths";
import {
@@ -273,17 +274,6 @@ async function exists(p: string): Promise<boolean> {
}
}
-async function readChannelConfigFile(
- dir: string,
-): Promise<ChannelConfig | null> {
- try {
- const raw = await readFile(path.join(dir, "config.json"), "utf8");
- return parseChannelConfig(JSON.parse(raw));
- } catch {
- return null;
- }
-}
-
async function scanSource(
channelsDir: string,
log: (msg: string) => void,
@@ -303,7 +293,7 @@ async function scanSource(
for (const ch of channelEntries) {
if (!ch.isDirectory()) continue;
const channelDir = path.join(channelsDir, ch.name);
- const cfg = await readChannelConfigFile(channelDir);
+ const cfg = await readChannelConfigFile(path.join(channelDir, "config.json"));
if (!cfg) {
log(`Skipping channel ${ch.name}: missing or invalid config.json`);
continue;
@@ -410,13 +400,9 @@ function indexKeysEqual(a: IndexKey, b: IndexKey): boolean {
return a[0] === b[0] && a[1] === b[1] && a[2] === b[2];
}
-async function writeJsonAtomic(
- filePath: string,
- value: unknown,
-): Promise<void> {
- const tmp = `${filePath}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(value));
- await rename(tmp, filePath);
+// Compact, no trailing newline — the export pages' historical bytes.
+function writeJsonAtomic(filePath: string, value: unknown): Promise<void> {
+ return writeJsonAtomicShared(filePath, value, { indent: 0, newline: false });
}
export type BuildIndexResult = {
diff --git a/common/controller/buildStats.ts b/common/controller/buildStats.ts
@@ -23,6 +23,7 @@ import {
} from "node:fs/promises";
import type { Dirent } from "node:fs";
import { open } from "lmdb";
+import { writeJsonAtomic as writeJsonAtomicShared } from "../lib/jsonFile-server";
import type { Cue } from "../lib/vtt";
import { transcriptCoverage } from "../lib/transcriptCoverage";
import {
@@ -34,7 +35,8 @@ import type { VideoState } from "../lib/availability";
import { loadDownloadOutcome } from "../lib/downloadOutcome-server";
import { loadTranscribeOutcome } from "../lib/transcribeOutcome-server";
import type { VideoStatus } from "../lib/stats";
-import { parseChannelConfig, type ChannelConfig } from "../lib/channelConfig";
+import type { ChannelConfig } from "../lib/channelConfig";
+import { readChannelConfigFile } from "./channels";
import type { Paths } from "../lib/paths";
import { listSites, siteStatsDir } from "../lib/site";
import {
@@ -138,17 +140,6 @@ async function resolveAcquisitionDates(
return { downloadedDate, transcribedDate };
}
-async function readChannelConfigFile(
- dir: string,
-): Promise<ChannelConfig | null> {
- try {
- const raw = await readFile(path.join(dir, "config.json"), "utf8");
- return parseChannelConfig(JSON.parse(raw));
- } catch {
- return null;
- }
-}
-
async function scanSource(
channelsDir: string,
log: (msg: string) => void,
@@ -164,7 +155,7 @@ async function scanSource(
for (const ch of channelEntries) {
if (!ch.isDirectory()) continue;
const channelDir = path.join(channelsDir, ch.name);
- const cfg = await readChannelConfigFile(channelDir);
+ const cfg = await readChannelConfigFile(path.join(channelDir, "config.json"));
if (!cfg) {
log(`Skipping channel ${ch.name}: missing or invalid config.json`);
continue;
@@ -243,10 +234,9 @@ async function writePages(
return { pageCount: pages.length, pagesWritten: pages.length };
}
-async function writeJsonAtomic(filePath: string, value: unknown): Promise<void> {
- const tmp = `${filePath}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(value));
- await rename(tmp, filePath);
+// Compact, no trailing newline — the stats manifests' historical bytes.
+function writeJsonAtomic(filePath: string, value: unknown): Promise<void> {
+ return writeJsonAtomicShared(filePath, value, { indent: 0, newline: false });
}
export async function buildStats({
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -1,6 +1,7 @@
+import { writeJsonAtomic } from "../lib/jsonFile-server";
import path from "node:path";
import type { Dirent } from "node:fs";
-import { readdir, readFile, rename, stat, writeFile } from "node:fs/promises";
+import { readdir, readFile, stat } from "node:fs/promises";
import pLimit from "p-limit";
import { readArchive } from "../lib/archive";
import {
@@ -1621,7 +1622,5 @@ async function writeChannelSnapshot(
snapshot: ChannelSnapshot,
): Promise<void> {
const file = snapshotPath(paths, slug);
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(snapshot, null, 2) + "\n");
- await rename(tmp, file);
+ await writeJsonAtomic(file, snapshot);
}
diff --git a/common/controller/channels.test.ts b/common/controller/channels.test.ts
@@ -3,6 +3,7 @@ import assert from "node:assert/strict";
import {
mkdir,
mkdtemp,
+ readFile,
rm,
stat,
symlink,
@@ -16,7 +17,14 @@ import {
relocatedDataDir,
RELOCATION_MARKER_FILENAME,
} from "../lib/channelMedia";
-import { channelExists, deleteChannel, writeChannelConfig } from "./channels";
+import {
+ channelConfigPath,
+ channelExists,
+ deleteChannel,
+ patchChannelConfig,
+ readChannelConfig,
+ writeChannelConfig,
+} from "./channels";
// Run with:
// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/channels.test.ts
@@ -130,3 +138,73 @@ test("an unmounted target does not make the channel undeletable", async () => {
assert.equal(await channelExists(paths, "alpha"), false);
});
});
+
+// ---------------------------------------------------------------------------
+// One reader, one strict writer, one patcher (one-core phase 3 slice 4b).
+// ---------------------------------------------------------------------------
+
+test("patchChannelConfig sets one key and leaves the rest", async () => {
+ await withPaths(async (paths) => {
+ await writeChannelConfig(paths, "a", { ...config, lastSyncedAt: "t0", keepLatest: 3 });
+ const out = await patchChannelConfig(paths, "a", { lastSyncedAt: "t1" });
+ assert.deepEqual(out, { handling: "youtube", name: "A channel", keepLatest: 3, lastSyncedAt: "t1" });
+ assert.deepEqual(await readChannelConfig(paths, "a"), out);
+ });
+});
+
+test("patchChannelConfig unsets before it assigns", async () => {
+ await withPaths(async (paths) => {
+ await writeChannelConfig(paths, "a", { ...config, dataDir: "/x", keepLatest: 3 });
+ await patchChannelConfig(paths, "a", {}, { unset: ["dataDir"] });
+ assert.deepEqual(await readChannelConfig(paths, "a"), { ...config, keepLatest: 3 });
+ // unset + set of the same key = set.
+ await patchChannelConfig(paths, "a", { keepLatest: 5 }, { unset: ["keepLatest"] });
+ assert.equal((await readChannelConfig(paths, "a"))?.keepLatest, 5);
+ });
+});
+
+test("patchChannelConfig on a missing or non-channel file writes nothing and answers null", async () => {
+ await withPaths(async (paths) => {
+ assert.equal(await patchChannelConfig(paths, "ghost", { lastSyncedAt: "t" }), null);
+ await assert.rejects(stat(channelConfigPath(paths, "ghost")));
+ await mkdir(path.join(paths.channelsDir, "junk"), { recursive: true });
+ await writeFile(channelConfigPath(paths, "junk"), "{ torn");
+ assert.equal(await patchChannelConfig(paths, "junk", { lastSyncedAt: "t" }), null);
+ assert.equal(await readFile(channelConfigPath(paths, "junk"), "utf8"), "{ torn");
+ });
+});
+
+test("writeChannelConfig is strict: a non-channel throws, unknown keys are dropped", async () => {
+ await withPaths(async (paths) => {
+ await assert.rejects(
+ writeChannelConfig(paths, "a", { handling: "x" } as unknown as ChannelConfig),
+ /not a channel config/,
+ );
+ await assert.rejects(stat(channelConfigPath(paths, "a")));
+ await writeChannelConfig(paths, "a", {
+ ...config,
+ bogus: 1,
+ excludeFromSync: true,
+ keepLatest: -4,
+ } as unknown as ChannelConfig);
+ assert.equal(
+ await readFile(channelConfigPath(paths, "a"), "utf8"),
+ JSON.stringify(config, null, 2) + "\n",
+ );
+ });
+});
+
+test("two concurrent patches to one channel both land", async () => {
+ await withPaths(async (paths) => {
+ await writeChannelConfig(paths, "a", config);
+ await Promise.all([
+ patchChannelConfig(paths, "a", { lastSyncedAt: "sync" }),
+ patchChannelConfig(paths, "a", { keepLatest: 7 }),
+ patchChannelConfig(paths, "a", { lastFullSweepAt: "sweep" }),
+ ]);
+ const out = await readChannelConfig(paths, "a");
+ assert.equal(out?.lastSyncedAt, "sync");
+ assert.equal(out?.keepLatest, 7);
+ assert.equal(out?.lastFullSweepAt, "sweep");
+ });
+});
diff --git a/common/controller/channels.ts b/common/controller/channels.ts
@@ -1,10 +1,13 @@
import path from "node:path";
-import { readdir, readFile, writeFile, rename, rm, stat, mkdir } from "node:fs/promises";
+import { readdir, readFile, rm, stat } from "node:fs/promises";
import type { Dirent } from "node:fs";
+import type { ChannelConfig } from "../lib/channelConfig";
+import { channelConfigSchema } from "../lib/channelConfigSchema";
import {
- parseChannelConfig,
- type ChannelConfig,
-} from "../lib/channelConfig";
+ readJsonFile,
+ withJsonFileLock,
+ writeJsonAtomic,
+} from "../lib/jsonFile-server";
import type { Paths } from "../lib/paths";
import { mapConcurrent } from "../lib/concurrency";
import {
@@ -146,19 +149,31 @@ export async function countPlaylist(p: string): Promise<number | null> {
return count;
}
+// THE CHANNEL config.json READER, WRITER AND PATCHER (one-core phase 3 slice
+// 4b). The shape is lib/channelConfigSchema.ts; everything that reads or
+// writes a config.json on the server goes through these three, except the two
+// places that read the RAW file on purpose (lib/channelMedia.ts's `dataDir`
+// guard, and the legacy `group` / `excludeFromSync` migrations, which need keys
+// the schema no longer names).
+
+export function channelConfigPath(paths: Paths, slug: string): string {
+ return path.join(paths.channelsDir, slug, "config.json");
+}
+
+// One config.json by path: the parsed config, or null when the file is absent,
+// unreadable, not JSON, or not a channel. Never throws.
+export async function readChannelConfigFile(
+ file: string,
+): Promise<ChannelConfig | null> {
+ const read = await readJsonFile(file);
+ return read.ok ? channelConfigSchema.parse(read.value) : null;
+}
+
export async function readChannelConfig(
paths: Paths,
slug: string,
): Promise<ChannelConfig | null> {
- try {
- const raw = await readFile(
- path.join(paths.channelsDir, slug, "config.json"),
- "utf8",
- );
- return parseChannelConfig(JSON.parse(raw));
- } catch {
- return null;
- }
+ return readChannelConfigFile(channelConfigPath(paths, slug));
}
export async function channelExists(
@@ -450,17 +465,56 @@ export async function listChannelStatsFromSnapshots(
}));
}
+// WRITE STAYS STRICT, like writeSettings: the config is parsed on the way out
+// — which is what drops an unknown or invalid key — and a value that is not a
+// channel at all (no valid `handling`) THROWS rather than writing a file every
+// reader would then treat as absent. Every caller passes a parsed config, so
+// this only ever fires on a bug. The e2e fixtures seed their configs raw, on
+// purpose, and are unaffected. Returns what it wrote.
export async function writeChannelConfig(
paths: Paths,
slug: string,
config: ChannelConfig,
-): Promise<void> {
- const dir = path.join(paths.channelsDir, slug);
- await mkdir(dir, { recursive: true });
- const file = path.join(dir, "config.json");
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(config, null, 2) + "\n");
- await rename(tmp, file);
+): Promise<ChannelConfig> {
+ const parsed = channelConfigSchema.parse(config);
+ if (!parsed) {
+ throw new Error(
+ `Refusing to write channels/${slug}/config.json: not a channel config ` +
+ `(handling ${JSON.stringify((config as { handling?: unknown })?.handling)})`,
+ );
+ }
+ await writeJsonAtomic(channelConfigPath(paths, slug), parsed, { mkdir: true });
+ return parsed;
+}
+
+export type PatchChannelConfigOptions = {
+ // Keys to delete before the patch is applied — how a caller says "clear
+ // this override" (a key absent from `patch` is left alone).
+ unset?: ReadonlyArray<keyof ChannelConfig>;
+};
+
+// READ-MODIFY-WRITE ONE CHANNEL'S CONFIG, and the only way to change some keys
+// of it: re-read the file NOW (not a copy the caller read earlier — that was
+// the stale-spread bug, where a fetch that took minutes wrote back the config
+// it had read at its start over every edit made meanwhile), delete `unset`,
+// assign `patch`, write strictly. A channel with no readable config is not
+// created: the answer is null and nothing is written (the old sync-stamp rule).
+// Two patches to one channel in this process serialise, so both land.
+export async function patchChannelConfig(
+ paths: Paths,
+ slug: string,
+ patch: Partial<ChannelConfig>,
+ opts: PatchChannelConfigOptions = {},
+): Promise<ChannelConfig | null> {
+ const file = channelConfigPath(paths, slug);
+ return withJsonFileLock(file, async () => {
+ const current = await readChannelConfigFile(file);
+ if (!current) return null;
+ const next: ChannelConfig = { ...current };
+ for (const key of opts.unset ?? []) delete next[key];
+ Object.assign(next, patch);
+ return writeChannelConfig(paths, slug, next);
+ });
}
export async function createChannel(
diff --git a/common/controller/fetchPosts.ts b/common/controller/fetchPosts.ts
@@ -7,7 +7,10 @@
// pages writes nothing new.
import path from "node:path";
-import { readChannelConfig, writeChannelConfig } from "./channels";
+import {
+ patchChannelConfig,
+ readChannelConfig,
+} from "./channels";
import type { Paths } from "../lib/paths";
import { isSocialChannel } from "../lib/channelConfig";
import {
@@ -180,8 +183,11 @@ export async function fetchPosts(
// it keys only off url / syncIntervalMinutes / lastSyncedAt, plus the
// channel-priority document's `sync` tier (`isChannelPaused(model, slug,
// "sync")`), which is where the retired `excludeFromSync` flag went.
- await writeChannelConfig(paths, slug, {
- ...config,
+ //
+ // A PATCH, re-read at write time. It used to spread the `config` this fetch
+ // read at its START, so any edit made during a fetch (minutes, for a long
+ // history) was silently reverted when the stamp landed.
+ await patchChannelConfig(paths, slug, {
lastSyncedAt: new Date().toISOString(),
});
diff --git a/common/controller/relocateChannelMedia.ts b/common/controller/relocateChannelMedia.ts
@@ -48,7 +48,11 @@ import {
type RelocationMarker,
type RelocationPhase,
} from "../lib/channelMedia";
-import { readChannelConfig, writeChannelConfig } from "./channels";
+import {
+ patchChannelConfig,
+ readChannelConfig,
+ writeChannelConfig,
+} from "./channels";
// MOVE A CHANNEL'S MEDIA TO ANOTHER DRIVE, AND BACK.
//
@@ -757,9 +761,17 @@ async function moveOut(args: {
// Written only now, on success: config.dataDir is a record of what is on
// disk, never an intention. Skipped when it already says so, so a rerun
// does not rewrite a file it agrees with.
- const fresh = (await readChannelConfig(paths, slug)) ?? args.config;
- if (fresh.dataDir?.trim() !== target) {
- await writeChannelConfig(paths, slug, { ...fresh, dataDir: target });
+ // No readable config.json (it vanished mid-move, before the read or
+ // between the read and the patch): the job's own copy is the best record
+ // there is, and the swap has already happened — so write that, never
+ // nothing.
+ const fresh = await readChannelConfig(paths, slug);
+ const patched =
+ fresh && fresh.dataDir?.trim() !== target
+ ? await patchChannelConfig(paths, slug, { dataDir: target })
+ : fresh;
+ if (!patched) {
+ await writeChannelConfig(paths, slug, { ...args.config, dataDir: target });
}
log(`Swapped: ${dataDir} -> ${target}`);
await writeMarker(paths, slug, {
@@ -975,8 +987,7 @@ async function moveBack(args: {
const fresh = await readChannelConfig(paths, slug);
if (fresh?.dataDir !== undefined) {
- const { dataDir: _dropped, ...rest } = fresh;
- await writeChannelConfig(paths, slug, rest);
+ await patchChannelConfig(paths, slug, {}, { unset: ["dataDir"] });
}
await writeMarker(paths, slug, {
target,
diff --git a/common/controller/renameChannel.ts b/common/controller/renameChannel.ts
@@ -5,7 +5,7 @@ import type { ChannelConfig } from "../lib/channelConfig";
import {
channelExists,
isValidChannelSlug,
- readChannelConfig,
+ patchChannelConfig,
writeChannelConfig,
} from "./channels";
import { savedVideoRoot } from "../lib/savedVideo";
@@ -166,11 +166,14 @@ export async function renameChannel(
relinked = true;
// Re-read: the channel dir has already moved, so this is the file that
// will actually be on disk afterwards.
- const fresh = (await readChannelConfig(paths, newSlug)) ?? config;
- await writeChannelConfig(paths, newSlug, {
- ...fresh,
+ const patched = await patchChannelConfig(paths, newSlug, {
dataDir: newTarget,
});
+ if (!patched) {
+ // No readable config.json at the new slug: write the one this rename
+ // started from, so the moved data is not left unrecorded.
+ await writeChannelConfig(paths, newSlug, { ...config, dataDir: newTarget });
+ }
} catch (err) {
// The link goes back too, and before the directory under it moves: a
// failure between the symlink and the config write left `data/` pointing
diff --git a/common/controller/storageLocations.ts b/common/controller/storageLocations.ts
@@ -33,8 +33,8 @@ import {
import type { ChannelConfig } from "../lib/channelConfig";
import {
listChannelConfigs,
+ patchChannelConfig,
readChannelConfig,
- writeChannelConfig,
} from "./channels";
import { relocationRootProblem } from "./relocateChannelMedia";
@@ -630,8 +630,9 @@ async function rollbackChannel(
entry: ChannelLedgerEntry,
): Promise<void> {
if (entry.configWritten && entry.oldConfig) {
- await writeChannelConfig(paths, entry.slug, {
- ...entry.oldConfig,
+ // Only the field this job changed goes back; anything else edited since
+ // stays. (It used to rewrite the whole config it found at the start.)
+ await patchChannelConfig(paths, entry.slug, {
dataDir: entry.oldTarget,
}).catch(() => {});
}
@@ -725,10 +726,15 @@ export async function repointStorageLocation(opts: {
await symlink(newTarget, link);
entry.relinked = true;
}
- await writeChannelConfig(opts.paths, slug, {
- ...(fresh as ChannelConfig),
+ // A null patch means config.json vanished or became unreadable after
+ // preflight listed the channel: fail this step, so the ledger rolls
+ // the link back, rather than leave a link no dataDir records.
+ const written = await patchChannelConfig(opts.paths, slug, {
dataDir: newTarget,
});
+ if (!written) {
+ throw new Error(`channels/${slug}/config.json is missing or unreadable`);
+ }
entry.configWritten = true;
} catch (err) {
const step = resuming
diff --git a/common/lib/aliasesStore.ts b/common/lib/aliasesStore.ts
@@ -7,8 +7,8 @@
// ships useful suggestions out of the box; a missing per-site file = no
// overrides. See common/lib/searchAliases.ts for the pure model + merge.
-import path from "node:path";
-import { readFileSync, writeFileSync, renameSync, mkdirSync } from "node:fs";
+import { readFileSync } from "node:fs";
+import { writeJsonAtomicSync } from "./jsonFile-server";
import type { Paths } from "./paths";
import { siteAliasesFile } from "./site";
import {
@@ -19,11 +19,9 @@ import {
type SearchAlias,
} from "./searchAliases";
+// Indented, no trailing newline — these files' historical bytes.
function writeJsonAtomic(filePath: string, value: unknown): void {
- mkdirSync(path.dirname(filePath), { recursive: true });
- const tmp = `${filePath}.tmp-${process.pid}`;
- writeFileSync(tmp, JSON.stringify(value, null, 2));
- renameSync(tmp, filePath);
+ writeJsonAtomicSync(filePath, value, { newline: false, mkdir: true });
}
// Global dictionary. Absent/unreadable → the seeded defaults (a fresh install
diff --git a/common/lib/attribution-server.ts b/common/lib/attribution-server.ts
@@ -1,48 +1,37 @@
-import path from "node:path";
-import { readFile, rename, writeFile } from "node:fs/promises";
import { ATTRIBUTION_FILENAME, type AttributionRecord } from "./attribution";
+import { sidecar, sidecarField } from "./sidecar-server";
-export function attributionPath(videoDir: string): string {
- return path.join(videoDir, ATTRIBUTION_FILENAME);
-}
-
-export async function loadAttribution(
- videoDir: string,
-): Promise<AttributionRecord | null> {
- try {
- const raw = await readFile(attributionPath(videoDir), "utf8");
- const parsed = JSON.parse(raw) as Partial<AttributionRecord>;
- // Validate the SHAPE, not just the parse. A half-written sidecar must read
- // as absent everywhere — the same rule diarization-server.ts's
- // hasDiarization encodes, and here it matters for a second reason: a
- // malformed file that read as "present" would also read as a DOWNGRADE
- // guard, and could block the diarized lane from ever writing a real record.
- if (
- typeof parsed?.videoId === "string" &&
- typeof parsed.generatedAt === "string" &&
- Array.isArray(parsed.speakers) &&
- Array.isArray(parsed.segments) &&
- !!parsed.provenance &&
- typeof parsed.provenance.method === "string"
- ) {
- return parsed as AttributionRecord;
- }
- return null;
- } catch {
- return null;
+// Validate the SHAPE, not just the parse. A half-written sidecar must read as
+// absent everywhere — the same rule diarization-server.ts's hasDiarization
+// encodes, and here it matters for a second reason: a malformed file that read
+// as "present" would also read as a DOWNGRADE guard, and could block the
+// diarized lane from ever writing a real record.
+export function coerceAttribution(value: unknown): AttributionRecord | null {
+ const parsed = value as Partial<AttributionRecord> | null;
+ if (
+ typeof parsed?.videoId === "string" &&
+ typeof parsed.generatedAt === "string" &&
+ Array.isArray(parsed.speakers) &&
+ Array.isArray(parsed.segments) &&
+ !!parsed.provenance &&
+ typeof parsed.provenance.method === "string"
+ ) {
+ return parsed as AttributionRecord;
}
+ return null;
}
-// tmp + rename, so a crash mid-write leaves the previous record rather than a
-// truncated one. Identical to writeDiarization; the atomicity is what lets
-// loadAttribution treat "parsed but wrong shape" as a real anomaly rather than
-// the expected state of a file being written.
-export async function writeAttribution(
- videoDir: string,
- record: AttributionRecord,
-): Promise<void> {
- const file = attributionPath(videoDir);
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(record) + "\n");
- await rename(tmp, file);
-}
+// Compact (one line + "\n"), as it always was. The write is atomic (tmp +
+// rename), which is what lets loadAttribution treat "parsed but wrong shape"
+// as a real anomaly rather than the expected state of a file being written.
+export const attributionSidecar = sidecar(
+ ATTRIBUTION_FILENAME,
+ sidecarField(coerceAttribution),
+ { indent: 0 },
+);
+
+export const {
+ path: attributionPath,
+ load: loadAttribution,
+ write: writeAttribution,
+} = attributionSidecar;
diff --git a/common/lib/availability-server.ts b/common/lib/availability-server.ts
@@ -1,5 +1,3 @@
-import path from "node:path";
-import { readFile, rename, writeFile } from "node:fs/promises";
import {
AVAILABILITY_FILENAME,
AVAILABILITY_VALUES,
@@ -10,43 +8,31 @@ import {
type AvailabilityHistorySource,
type AvailabilityRecord,
} from "./availability";
-import {
- DOWNLOAD_OUTCOME_FILENAME,
- type DownloadOutcomeRecord,
-} from "./downloadOutcome";
-
-export function availabilityPath(videoDir: string): string {
- return path.join(videoDir, AVAILABILITY_FILENAME);
-}
+import { loadDownloadOutcome } from "./downloadOutcome-server";
+import { sidecar, sidecarField } from "./sidecar-server";
-export async function loadAvailability(
- videoDir: string,
-): Promise<AvailabilityRecord | null> {
- try {
- const raw = await readFile(availabilityPath(videoDir), "utf8");
- const parsed = JSON.parse(raw) as Partial<AvailabilityRecord>;
- if (
- typeof parsed?.availability === "string" &&
- (AVAILABILITY_VALUES as string[]).includes(parsed.availability) &&
- typeof parsed.checkedAt === "string"
- ) {
- return parsed as AvailabilityRecord;
- }
- return null;
- } catch {
- return null;
+export function coerceAvailability(value: unknown): AvailabilityRecord | null {
+ const parsed = value as Partial<AvailabilityRecord> | null;
+ if (
+ typeof parsed?.availability === "string" &&
+ (AVAILABILITY_VALUES as string[]).includes(parsed.availability) &&
+ typeof parsed.checkedAt === "string"
+ ) {
+ return parsed as AvailabilityRecord;
}
+ return null;
}
-export async function writeAvailability(
- videoDir: string,
- record: AvailabilityRecord,
-): Promise<void> {
- const file = availabilityPath(videoDir);
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(record, null, 2) + "\n");
- await rename(tmp, file);
-}
+export const availabilitySidecar = sidecar(
+ AVAILABILITY_FILENAME,
+ sidecarField(coerceAvailability),
+);
+
+export const {
+ path: availabilityPath,
+ load: loadAvailability,
+ write: writeAvailability,
+} = availabilitySidecar;
// Append-on-change writer that all availability persistence funnels through.
// Records an observation in the video's history[] only when the availability
@@ -171,24 +157,18 @@ export async function resolveMaybeMissingState(
return "maybe_missing";
}
+// The download-outcome side of resolveEffectiveAvailability: the LAST
+// attempt's availabilityClass, when there is one. Reads through
+// loadDownloadOutcome, so a download-outcome.json that fails that shape check
+// no longer contributes an availability (slice 4b record, behaviour changes).
async function loadDownloadOutcomeAvailability(
videoDir: string,
): Promise<{ availability: Availability; finishedAt: string | null } | null> {
- try {
- const raw = await readFile(
- path.join(videoDir, DOWNLOAD_OUTCOME_FILENAME),
- "utf8",
- );
- const parsed = JSON.parse(raw) as Partial<DownloadOutcomeRecord>;
- const attempts = parsed.attempts;
- if (!Array.isArray(attempts) || attempts.length === 0) return null;
- const last = attempts[attempts.length - 1];
- if (!last?.availabilityClass) return null;
- return {
- availability: last.availabilityClass,
- finishedAt: typeof parsed.finishedAt === "string" ? parsed.finishedAt : null,
- };
- } catch {
- return null;
- }
+ const outcome = await loadDownloadOutcome(videoDir);
+ const last = outcome?.attempts.at(-1);
+ if (!outcome || !last?.availabilityClass) return null;
+ return {
+ availability: last.availabilityClass,
+ finishedAt: typeof outcome.finishedAt === "string" ? outcome.finishedAt : null,
+ };
}
diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts
@@ -4,6 +4,7 @@ import {
type DownloadFormatPreset,
} from "../ytdlp/downloadFormat";
import { isCookieMode, type CookieMode } from "./cookiePolicy";
+import type { FieldDocs } from "./fieldDocs";
export type ChannelHandling = "youtube" | "transcribe";
@@ -43,44 +44,51 @@ export const EXTRACTION_MODE_VALUES: ReadonlyArray<ExtractionMode> = [
"app",
];
+
+// Each field is documented in AUDIO_CHECK_FIELD_DOCS below.
export type AudioCheckConfig = {
enabled: boolean;
intervalSeconds?: number;
maxRollbacks?: number;
copyTimeoutSeconds?: number;
- // When false (default), yt-dlp stays SIGSTOPped across each integrity
- // probe, so it never downloads bytes that a malformed verdict would
- // discard and force a re-fetch — minimising HTTP 429 risk. Set true for
- // the legacy behavior: resume immediately after the snapshot copy and
- // probe concurrently while the download keeps running.
resumeDuringProbe?: boolean;
};
+export const AUDIO_CHECK_FIELD_DOCS: FieldDocs<AudioCheckConfig> = {
+ enabled:
+ "Turn the check on. Required: an `audioCheck` object without a boolean `enabled` is dropped whole.",
+ intervalSeconds:
+ "Seconds between integrity probes of the in-progress `.part` file; clamped to [10, 600]. Absent = 60. The live cadence adapts (AIMD): a malformed checkpoint halves it toward the 10 s floor, clean ones step it back up.",
+ maxRollbacks:
+ "Rollbacks to the last known-good snapshot before the download is given up; clamped to [1, 20]. Absent = 5.",
+ copyTimeoutSeconds:
+ "Seconds allowed for the snapshot copy; clamped to [5, 120]. Absent = 30.",
+ resumeDuringProbe:
+ "When false (default), yt-dlp stays SIGSTOPped across each probe, so it never downloads bytes a malformed verdict would discard and force a re-fetch — minimising HTTP 429 risk. True = the legacy behaviour: resume right after the snapshot copy and probe while the download keeps running.",
+};
+
// Per-channel download filter patterns. Regex SOURCES, always compiled with
// the "i" flag; an unparseable pattern makes the filter inert rather than
-// failing a download (the editor form refuses to save one).
+// failing a download (the editor form refuses to save one). Each field is
+// documented in DOWNLOAD_FILTER_FIELD_DOCS below.
export type DownloadFilterConfig = {
include?: string;
exclude?: string;
- // Opt every livestream VOD in, regardless of what it is called. A SECOND
- // positive selector beside `include`, not a modifier of it — which is why
- // `{ includeLivestreams: true }` alone rejects plain uploads and passes
- // livestreams. See titleFilterRejects.
includeLivestreams?: boolean;
- // WHAT TO DO WITH A LIVESTREAM THE FILTER REJECTED. Absent = "skip", which
- // is every channel that predates this field and is byte-for-byte what they
- // did before it existed.
- //
- // "chat-only" is the middle answer that did not exist: a multi-hour stream
- // whose TITLE says nothing about the subject is usually not worth its audio,
- // but its live chat is text, it is small, and it is the only record of what
- // the room said. So the video is not downloaded, its chat is, and it joins
- // the corpus as a chat track with no captions — NOT as a downloaded video.
- // See classifyAgainstFilter for why this is a third VERDICT rather than a
- // flag read at download time.
rejectedLivestreams?: RejectedLivestreamMode;
};
+export const DOWNLOAD_FILTER_FIELD_DOCS: FieldDocs<DownloadFilterConfig> = {
+ include:
+ "Case-insensitive regex SOURCE (no delimiters, no flags) a video's `title + \"\\n\" + description` must match to be downloaded. Trimmed; blank = no include rule.",
+ exclude:
+ "Case-insensitive regex source that rejects a matching video. Wins over `include`. Trimmed; blank = no exclude rule.",
+ includeLivestreams:
+ "Opt every livestream VOD in, whatever it is called. A SECOND positive selector beside `include`, not a modifier of it — so `{ includeLivestreams: true }` alone rejects plain uploads and passes livestreams. Stored only when `true`.",
+ rejectedLivestreams:
+ "What to do with a livestream the filter REJECTED: `\"skip\"` (default, and what every channel predating the field did) or `\"chat-only\"` — the video is not downloaded, its live chat is, and it joins the corpus as a chat track with no captions. Stored only when not `\"skip\"`, and only beside a real filter: with nothing to reject it names a decision that can never be taken.",
+};
+
// Absent behaves as "skip". Spelled as a union rather than a boolean because a
// third answer (keeping the audio at a lower quality, say) is a plausible next
// one and a boolean would have to be migrated to make room for it.
@@ -92,138 +100,119 @@ export function isRejectedLivestreamMode(
return v === "skip" || v === "chat-only";
}
+// A channel's `config.json`. Each field is documented in
+// CHANNEL_CONFIG_FIELD_DOCS below (rendered into CHANNEL.md); the keys that are
+// SYNC STATE rather than configuration are CHANNEL_SYNC_STATE_KEYS.
+//
+// `excludeFromSync` IS GONE. It said "stop syncing, keep everything else",
+// which is exactly what `channelPriority`'s per-operation override map says —
+// `{tier:<base>, overrides:{sync:"paused"}}` — and the model can say it for
+// every operation, not just this one. The parser drops the key rather than
+// carrying it, so an un-migrated config.json still parses: the migration
+// (common/bin/migrate-channel-priority.ts) reads it from the RAW file, not
+// through this parser, precisely because this parser no longer knows the word.
export type ChannelConfig = {
handling: ChannelHandling;
- // Omitted = "video" (every channel that predates the posts corpus).
sourceKind?: ChannelSourceKind;
- // Social channels only: which SocialFetcher drives ingest (see
- // common/social/fetchers.ts), e.g. "bluesky-atproto" / "x-gallery-dl".
- // Omitted = resolve by URL detection.
postFetcher?: string;
- // Social channels only: the bare account handle, without "@" or URL wrapper.
- // Derived from `url` at creation time but stored so a later URL-format change
- // upstream can't silently re-point ingest at a different account.
socialHandle?: string;
platform?: Platform;
name?: string;
url?: string;
audioFormat?: AudioFormat;
- // Per-channel override for the yt-dlp `-f` download format (see
- // common/ytdlp/downloadFormat.ts). Omitted -> inherit the global default
- // (SiteSettings.downloadFormat), which itself falls back to the per-source
- // "auto" selector. Lets a channel whose source only serves full-length audio
- // in its `original` format (e.g. Odysee) force it regardless of the global.
downloadFormat?: DownloadFormatPreset;
keepSourceVideo?: boolean;
- // Keep-latest retention/persistence window. The newest N videos (by upload
- // date) are protected from the Clean-audio sweep AND have their source video
- // persisted to the saved-video store (see common/controller/keptVideos.ts and
- // the per-download persistence rule in downloadOneManaged.ts). Semantics:
- // undefined / 0 -> disabled
- // > 0 -> keep newest N (clamped to KEEP_LATEST_MAX)
- // A kept video later found deleted-from-source is pinned permanently via the
- // do-not-clean marker so it survives even after it rolls out of the window.
keepLatest?: number;
- // Audio-extraction strategy for transcribe-handling downloads (see
- // ExtractionMode above). Omitted/unknown -> "ytdlp" (legacy behavior). The
- // keep-latest persistence rule forces "app" for the videos it persists,
- // regardless of this setting.
extractionMode?: ExtractionMode;
- // Per-channel override for the saved-video store root (global default is
- // paths.savedVideosDir / SAVED_VIDEOS_DIR). When set, this channel's persisted
- // source videos live under <savedVideosDir>/<slug>/<videoId>/. Lets a single
- // channel's large videos land on a different disk than the rest. Resolved by
- // savedVideoDir() in common/lib/savedVideo.ts. Empty/whitespace = use global.
savedVideosDir?: string;
- // Where this channel's downloaded media ACTUALLY lives, when it has been
- // relocated to another drive: the absolute path `channels/<slug>/data` is a
- // symlink to. Blank/absent = in place. Written ONLY by the relocate job on
- // success (common/controller/relocateChannelMedia.ts) — it is a record of
- // what is on disk, never a free-text field, because a value that disagrees
- // with the link is an "inconsistent" channel that every guard refuses. See
- // common/lib/channelMedia.ts.
dataDir?: string;
ytdlpExtraArgs?: string[];
subLangs?: string;
lastSyncedAt?: string;
lastFullDownloadAt?: string;
- // When this channel last paid for a sync FULL SWEEP — the deep pass that
- // re-enumerates the whole listing to refresh `playlist` and flag videos that
- // have left it. Stamped by the sweep itself; read by the cadence gate in
- // common/jobs/deepSync.ts to decide whether the next sync sweeps or stays on
- // the cheap newest-first paged walk.
lastFullSweepAt?: string;
excludeFromBuild?: boolean;
- // `excludeFromSync` IS GONE. It said "stop syncing, keep everything else",
- // which is exactly what `channelPriority`'s per-operation override map says
- // — `{tier:<base>, overrides:{sync:"paused"}}` — and the model can say it
- // for every operation, not just this one. The sanitizer below drops the key
- // rather than carrying it, so an un-migrated config.json still parses: the
- // migration (common/bin/migrate-channel-priority.ts) reads it from the RAW
- // file, not through this parser, precisely because this parser no longer
- // knows the word.
- // Opt this channel OUT of the aggregate "cleanable data" total shown on the
- // /cleanup page and its sidebar badge. The per-channel cleanup sweeps remain
- // fully available; this flag only removes the channel's reclaimable bytes from
- // the running total (e.g. keep a finicky-to-redownload channel's audio around
- // for now without it nagging in the badge). Toggled from the cleanup ledger.
excludeFromCleanup?: boolean;
- // Auto-sync cadence for the scheduled sync system (common/jobs/syncScheduler.ts).
- // The cron-driven scheduler syncs this channel when
- // `now - lastSyncedAt >= syncIntervalMinutes`. Semantics:
- // undefined -> inherit the global SiteSettings default interval
- // 0 -> auto-sync disabled for this channel (still manually syncable)
- // > 0 -> sync this often (clamped to [SYNC_INTERVAL_MIN/MAX_MINUTES])
- // A missing `url` also disables auto-sync, as does a `sync` tier of `paused`
- // in the channel-priority document (common/lib/channelPriority.ts).
syncIntervalMinutes?: number;
- // Full-sweep cadence for this channel (see common/jobs/deepSync.ts). A sync
- // upgrades itself to a full sweep when
- // `now - lastFullSweepAt >= fullSweepIntervalMinutes`. Same semantics as
- // syncIntervalMinutes:
- // undefined -> inherit the global syncScheduler.fullSweepIntervalMinutes
- // 0 -> never sweep this channel (every sync is a paged walk)
- // > 0 -> sweep this often (clamped to [SYNC_INTERVAL_MIN/MAX_MINUTES])
- // A sweep is one full enumeration, so long cadences (daily to weekly) are the
- // norm — it is much more expensive than one 50-entry sync page.
fullSweepIntervalMinutes?: number;
- // Per-channel override for the global setting of the same name. When
- // omitted, the global SiteSettings value is used. 0 disables the sleep
- // for this channel.
- sleepBetweenDownloadsSeconds?: number;
- // Per-channel override for the global skipLiveDownloads setting. When
- // omitted, the global SiteSettings value is used. Set false to allow this
- // channel to download currently-live/upcoming videos.
skipLiveDownloads?: boolean;
- // Per-channel title/description download filter. Both patterns are
- // case-insensitive regex SOURCES (no delimiters, no flags) matched against
- // `title + "\n" + description` from the per-video metadata prefetch — the
- // only text a download filter ever gets to see. `exclude` wins over
- // `include`. A video the filter declines is SETTLED by a terminal
- // download-outcome keyed on the filter's signature (see
- // downloadFilterSignature), NOT by an archive line: an archive line would
- // mean "downloaded" to verifyTranscripts and the missingFromArchive bucket,
- // and would still leave the id in the artifact-based undownloadedIds. The
- // signature is what makes the settlement self-expiring — change either
- // pattern and every settled video is re-evaluated on the next run.
- // Omitted/empty = no filter (the registry entry is inert).
downloadFilter?: DownloadFilterConfig;
- // Per-channel override of the global cookies-from-browser browser spec
- // (SiteSettings.cookiesFromBrowser). Omitted/empty = inherit the global
- // value. See common/lib/cookiePolicy.ts.
cookiesFromBrowser?: string;
- // Per-channel override of the global cookie mode
- // (SiteSettings.cookieMode). Omitted = inherit. See
- // common/lib/cookiePolicy.ts for the mode semantics.
cookieMode?: CookieMode;
- // Opt-in audio-integrity checking for sources that intermittently serve
- // corrupt audio mid-download (e.g. Odysee "original" format). When
- // enabled, the managed downloader periodically validates the in-progress
- // .part file via ffmpeg and rolls back to the last known-good snapshot
- // on corruption. transcribe-handling only.
+ sleepBetweenDownloadsSeconds?: number;
audioCheck?: AudioCheckConfig;
};
+// In the order parseChannelConfig emits (and CHANNEL.md lists) them.
+export const CHANNEL_CONFIG_FIELD_DOCS: FieldDocs<ChannelConfig> = {
+ handling:
+ 'REQUIRED. `"youtube"` (fetch the platform\'s captions) or `"transcribe"` (download audio and transcribe it locally). A file without a valid `handling` is not a channel: it reads as null.',
+ sourceKind:
+ 'What KIND of source this is: `"video"` (default — yt-dlp + transcription) or `"social"` (an account fetched into the posts corpus, skipped by the video scan). A separate axis from `handling`, so every binary handling branch stays binary.',
+ postFetcher:
+ 'Social channels only: which social fetcher drives ingest (e.g. `"bluesky-atproto"`, `"x-gallery-dl"`). Absent = resolve by URL detection. Trimmed.',
+ socialHandle:
+ 'Social channels only: the bare account handle (a leading "@" is stripped). Derived from `url` at creation but stored, so a later URL-format change upstream cannot silently re-point ingest at a different account.',
+ platform: "The source platform (youtube, rumble, …). An unknown value is dropped.",
+ name: "Display name.",
+ url: "The channel / playlist / account URL syncs enumerate. Absent = the channel is never auto-synced.",
+ audioFormat: '`"m4a"`, `"mp3"` or `"opus"`: the audio a transcribe-handling download keeps.',
+ downloadFormat:
+ "Per-channel override for the yt-dlp `-f` download format preset. Absent = inherit the global `downloadFormat`, which itself falls back to the per-source \"auto\" selector. Lets a channel whose source serves full-length audio only in its `original` format (e.g. Odysee) force it.",
+ keepSourceVideo: "Keep the downloaded source video beside the audio.",
+ keepLatest:
+ "Keep-latest window: the newest N videos (by upload date) are protected from the Clean-audio sweep AND have their source video persisted to the saved-video store. 0 or absent = disabled; positives clamp to [1, 100000]. A kept video later found deleted at the source is pinned permanently via the do-not-clean marker.",
+ extractionMode:
+ '`"ytdlp"` (default — yt-dlp\'s own `-x --audio-format` postprocessor, no source container kept) or `"app"` (yt-dlp downloads the source container and the app runs ffmpeg). The keep-latest persistence rule forces `"app"` for the videos it persists.',
+ savedVideosDir:
+ "Per-channel override for the saved-video store root: this channel's persisted source videos live under `<savedVideosDir>/<slug>/<videoId>/`. Trimmed; blank = the global store.",
+ dataDir:
+ "Where this channel's media ACTUALLY lives when relocated to another drive: the absolute path `channels/<slug>/data` is a symlink to. Absent = in place. Written ONLY by the relocate / re-point jobs on success — a record of what is on disk, never free text, because a value that disagrees with the link is an \"inconsistent\" channel every guard refuses.",
+ ytdlpExtraArgs: "Extra yt-dlp arguments, appended verbatim. Must be an array of strings or it is dropped.",
+ subLangs: "yt-dlp `--sub-langs` value for caption downloads.",
+ lastSyncedAt:
+ "SYNC STATE. When the channel last synced (ISO time). Stamped by every sync, and by a social fetch; read by the scheduler's cadence gate.",
+ lastFullDownloadAt: "SYNC STATE. When a full download pass last completed (ISO time).",
+ lastFullSweepAt:
+ "SYNC STATE. When this channel last paid for a sync FULL SWEEP — the deep pass that re-enumerates the whole listing to refresh `playlist` and flag videos that have left it. Stamped by the sweep; read by the cadence gate to decide whether the next sync sweeps or stays on the cheap newest-first paged walk.",
+ excludeFromBuild: "Leave this channel out of every site build.",
+ excludeFromCleanup:
+ "Leave this channel's reclaimable bytes out of the aggregate \"cleanable data\" total on /cleanup and its badge. The per-channel cleanup sweeps stay available; only the running total changes.",
+ syncIntervalMinutes:
+ "Auto-sync cadence: the scheduler syncs this channel when `now - lastSyncedAt >= syncIntervalMinutes`. Absent = inherit the global default; 0 = auto-sync off (still manually syncable); positives clamp to [1, 44640] (~31 days). A missing `url`, or a `sync` tier of paused in the channel-priority document, also disables auto-sync.",
+ fullSweepIntervalMinutes:
+ "Full-sweep cadence: a sync upgrades itself to a full sweep when `now - lastFullSweepAt >= fullSweepIntervalMinutes`. Absent = inherit `syncScheduler.fullSweepIntervalMinutes`; 0 = never sweep (every sync is a paged walk); positives clamp to [1, 44640].",
+ skipLiveDownloads:
+ "Per-channel override for the global `skipLiveDownloads`. Absent = inherit; false = allow downloading currently-live / upcoming videos.",
+ downloadFilter:
+ "Per-channel title/description download filter, matched against the per-video metadata prefetch. A declined video is SETTLED by a terminal download-outcome keyed on the filter's signature, not by an archive line, so changing either pattern re-evaluates every settled video on the next run. An object with no real rule is dropped (the filter is inert).",
+ cookiesFromBrowser:
+ "Per-channel override of the global cookies-from-browser spec. Trimmed; blank = inherit.",
+ cookieMode: "Per-channel override of the global cookie mode. Absent = inherit.",
+ sleepBetweenDownloadsSeconds:
+ "Per-channel override for the global pause between downloads. Absent = inherit; 0 = no sleep; floored and capped at 600.",
+ audioCheck:
+ "Opt-in audio-integrity checking for sources that intermittently serve corrupt audio mid-download (e.g. Odysee \"original\"): the managed downloader periodically validates the in-progress `.part` file and rolls back to the last known-good snapshot on corruption. transcribe-handling only.",
+};
+
+// Every key a config.json may carry, in emission order. The unknown-key oracle:
+// a key not listed here is dropped by every read and every write.
+export const CHANNEL_CONFIG_KEYS = Object.keys(
+ CHANNEL_CONFIG_FIELD_DOCS,
+) as ReadonlyArray<keyof ChannelConfig>;
+
+// The keys that are MUTABLE SYNC STATE rather than configuration: stamped by
+// the sync / sweep / download passes, never by the channel form. They live in
+// the same file on purpose (the file is not split); a form save preserves them
+// because it unsets only the form's own fields.
+export const CHANNEL_SYNC_STATE_KEYS = [
+ "lastSyncedAt",
+ "lastFullDownloadAt",
+ "lastFullSweepAt",
+] as const satisfies ReadonlyArray<keyof ChannelConfig>;
+
+export type ChannelSyncStateKey = (typeof CHANNEL_SYNC_STATE_KEYS)[number];
+
export const CHANNEL_SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS = 600;
// Bounds for a channel's auto-sync interval. A nonzero value is clamped into
@@ -254,6 +243,7 @@ export const AUDIO_CHECK_COPY_TIMEOUT_DEFAULT_SECONDS = 30;
export const AUDIO_CHECK_COPY_TIMEOUT_MIN_SECONDS = 5;
export const AUDIO_CHECK_COPY_TIMEOUT_MAX_SECONDS = 120;
+
function clampInt(
value: number,
min: number,
@@ -273,192 +263,152 @@ export const AUDIO_FORMAT_VALUES: ReadonlyArray<AudioFormat> = [
"opus",
];
-export function parseChannelConfig(raw: unknown): ChannelConfig | null {
- if (!raw || typeof raw !== "object") return null;
- const r = raw as Record<string, unknown>;
- if (r.handling !== "youtube" && r.handling !== "transcribe") return null;
- const config: ChannelConfig = { handling: r.handling };
- if (r.sourceKind === "video" || r.sourceKind === "social") {
- config.sourceKind = r.sourceKind;
- }
- if (typeof r.postFetcher === "string" && r.postFetcher.trim()) {
- config.postFetcher = r.postFetcher.trim();
- }
- if (typeof r.socialHandle === "string" && r.socialHandle.trim()) {
- config.socialHandle = r.socialHandle.trim().replace(/^@/, "");
- }
- if (
- typeof r.platform === "string" &&
- PLATFORM_VALUES.includes(r.platform as Platform)
- ) {
- config.platform = r.platform as Platform;
- }
- if (typeof r.name === "string") config.name = r.name;
- if (typeof r.url === "string") config.url = r.url;
- if (
- r.audioFormat === "m4a" ||
- r.audioFormat === "mp3" ||
- r.audioFormat === "opus"
- ) {
- config.audioFormat = r.audioFormat;
- }
- if (isDownloadFormatPreset(r.downloadFormat)) {
- config.downloadFormat = r.downloadFormat;
- }
- if (typeof r.keepSourceVideo === "boolean") {
- config.keepSourceVideo = r.keepSourceVideo;
- }
- if (
- typeof r.keepLatest === "number" &&
- Number.isFinite(r.keepLatest) &&
- r.keepLatest >= 0
- ) {
- // 0 is the "disabled" sentinel preserved as-is; positives clamp to the cap.
- config.keepLatest =
- r.keepLatest === 0 ? 0 : clampInt(r.keepLatest, 1, KEEP_LATEST_MAX);
- }
- if (r.extractionMode === "ytdlp" || r.extractionMode === "app") {
- config.extractionMode = r.extractionMode;
- }
- if (typeof r.savedVideosDir === "string" && r.savedVideosDir.trim() !== "") {
- config.savedVideosDir = r.savedVideosDir.trim();
- }
- if (typeof r.dataDir === "string" && r.dataDir.trim() !== "") {
- config.dataDir = r.dataDir.trim();
- }
- if (
- Array.isArray(r.ytdlpExtraArgs) &&
- r.ytdlpExtraArgs.every((x) => typeof x === "string")
- ) {
- config.ytdlpExtraArgs = r.ytdlpExtraArgs as string[];
- }
- if (typeof r.subLangs === "string") config.subLangs = r.subLangs;
- if (typeof r.lastSyncedAt === "string") config.lastSyncedAt = r.lastSyncedAt;
- if (typeof r.lastFullDownloadAt === "string") {
- config.lastFullDownloadAt = r.lastFullDownloadAt;
- }
- if (typeof r.lastFullSweepAt === "string") {
- config.lastFullSweepAt = r.lastFullSweepAt;
- }
- if (typeof r.excludeFromBuild === "boolean") {
- config.excludeFromBuild = r.excludeFromBuild;
- }
- if (typeof r.excludeFromCleanup === "boolean") {
- config.excludeFromCleanup = r.excludeFromCleanup;
- }
- if (
- typeof r.syncIntervalMinutes === "number" &&
- Number.isFinite(r.syncIntervalMinutes) &&
- r.syncIntervalMinutes >= 0
- ) {
- // 0 is a sentinel ("auto-sync disabled") preserved as-is; any other value
- // is clamped into the supported window.
- config.syncIntervalMinutes =
- r.syncIntervalMinutes === 0
- ? 0
- : clampInt(
- r.syncIntervalMinutes,
- SYNC_INTERVAL_MIN_MINUTES,
- SYNC_INTERVAL_MAX_MINUTES,
- );
- }
- if (
- typeof r.fullSweepIntervalMinutes === "number" &&
- Number.isFinite(r.fullSweepIntervalMinutes) &&
- r.fullSweepIntervalMinutes >= 0
- ) {
- // Same shape as syncIntervalMinutes above: 0 is the "never sweep" sentinel.
- config.fullSweepIntervalMinutes =
- r.fullSweepIntervalMinutes === 0
- ? 0
- : clampInt(
- r.fullSweepIntervalMinutes,
- SYNC_INTERVAL_MIN_MINUTES,
- SYNC_INTERVAL_MAX_MINUTES,
- );
- }
- if (typeof r.skipLiveDownloads === "boolean") {
- config.skipLiveDownloads = r.skipLiveDownloads;
- }
- // Download filter. Blank/whitespace patterns are dropped rather than stored,
- // so an all-blank object leaves the key absent and the filter inert — the
- // same "empty means inherit-off" rule the cookie overrides use.
- if (r.downloadFilter && typeof r.downloadFilter === "object") {
- const df = r.downloadFilter as Record<string, unknown>;
- const include = typeof df.include === "string" ? df.include.trim() : "";
- const exclude = typeof df.exclude === "string" ? df.exclude.trim() : "";
- const includeLivestreams = df.includeLivestreams === true;
- // ONLY ALONGSIDE A REAL FILTER, and only when it is not the default. The
- // mode says what to do with a REJECTED livestream, so with nothing to
- // reject it names a decision that can never be taken — storing it would put
- // a setting on the Configure form that does nothing and explains nothing.
- const rejectedLivestreams = isRejectedLivestreamMode(df.rejectedLivestreams)
- ? df.rejectedLivestreams
- : "skip";
- if (include || exclude || includeLivestreams) {
- config.downloadFilter = {
- ...(include ? { include } : {}),
- ...(exclude ? { exclude } : {}),
- ...(includeLivestreams ? { includeLivestreams: true } : {}),
- ...(rejectedLivestreams !== "skip" ? { rejectedLivestreams } : {}),
- };
- }
- }
- if (typeof r.cookiesFromBrowser === "string" && r.cookiesFromBrowser.trim()) {
- config.cookiesFromBrowser = r.cookiesFromBrowser.trim();
+// Only an object whose `handling` is valid is a channel. Everything else —
+// `null`, an array, a number, `{}` — reads as "not a channel".
+export function isChannelShape(raw: unknown): raw is Record<string, unknown> {
+ if (!raw || typeof raw !== "object") return false;
+ const h = (raw as Record<string, unknown>).handling;
+ return h === "youtube" || h === "transcribe";
+}
+
+const isFiniteNumber = (v: unknown): v is number =>
+ typeof v === "number" && Number.isFinite(v);
+const trimmedNonBlank = (v: unknown): string | undefined =>
+ typeof v === "string" && v.trim() ? v.trim() : undefined;
+const bool = (v: unknown): boolean | undefined =>
+ typeof v === "boolean" ? v : undefined;
+const str = (v: unknown): string | undefined =>
+ typeof v === "string" ? v : undefined;
+
+// 0 is a sentinel ("disabled") preserved as-is; any other non-negative value
+// is clamped into [min, max].
+const zeroOrClamped = (min: number, max: number) => (v: unknown): number | undefined =>
+ isFiniteNumber(v) && v >= 0 ? (v === 0 ? 0 : clampInt(v, min, max)) : undefined;
+
+export function coerceDownloadFilter(v: unknown): DownloadFilterConfig | undefined {
+ // Blank/whitespace patterns are dropped rather than stored, so an all-blank
+ // object leaves the key absent and the filter inert — the same "empty means
+ // inherit-off" rule the cookie overrides use.
+ if (!v || typeof v !== "object") return undefined;
+ const df = v as Record<string, unknown>;
+ const include = typeof df.include === "string" ? df.include.trim() : "";
+ const exclude = typeof df.exclude === "string" ? df.exclude.trim() : "";
+ const includeLivestreams = df.includeLivestreams === true;
+ // ONLY ALONGSIDE A REAL FILTER, and only when it is not the default. The
+ // mode says what to do with a REJECTED livestream, so with nothing to reject
+ // it names a decision that can never be taken — storing it would put a
+ // setting on the Configure form that does nothing and explains nothing.
+ const rejectedLivestreams = isRejectedLivestreamMode(df.rejectedLivestreams)
+ ? df.rejectedLivestreams
+ : "skip";
+ if (!include && !exclude && !includeLivestreams) return undefined;
+ return {
+ ...(include ? { include } : {}),
+ ...(exclude ? { exclude } : {}),
+ ...(includeLivestreams ? { includeLivestreams: true } : {}),
+ ...(rejectedLivestreams !== "skip" ? { rejectedLivestreams } : {}),
+ };
+}
+
+export function coerceAudioCheck(v: unknown): AudioCheckConfig | undefined {
+ if (!v || typeof v !== "object") return undefined;
+ const a = v as Record<string, unknown>;
+ if (typeof a.enabled !== "boolean") return undefined;
+ const ac: AudioCheckConfig = { enabled: a.enabled };
+ if (isFiniteNumber(a.intervalSeconds)) {
+ ac.intervalSeconds = clampInt(
+ a.intervalSeconds,
+ AUDIO_CHECK_INTERVAL_MIN_SECONDS,
+ AUDIO_CHECK_INTERVAL_MAX_SECONDS,
+ );
}
- if (isCookieMode(r.cookieMode)) {
- config.cookieMode = r.cookieMode;
+ if (isFiniteNumber(a.maxRollbacks)) {
+ ac.maxRollbacks = clampInt(
+ a.maxRollbacks,
+ AUDIO_CHECK_MAX_ROLLBACKS_MIN,
+ AUDIO_CHECK_MAX_ROLLBACKS_MAX,
+ );
}
- if (
- typeof r.sleepBetweenDownloadsSeconds === "number" &&
- Number.isFinite(r.sleepBetweenDownloadsSeconds) &&
- r.sleepBetweenDownloadsSeconds >= 0
- ) {
- config.sleepBetweenDownloadsSeconds = Math.min(
- Math.floor(r.sleepBetweenDownloadsSeconds),
- CHANNEL_SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS,
+ if (isFiniteNumber(a.copyTimeoutSeconds)) {
+ ac.copyTimeoutSeconds = clampInt(
+ a.copyTimeoutSeconds,
+ AUDIO_CHECK_COPY_TIMEOUT_MIN_SECONDS,
+ AUDIO_CHECK_COPY_TIMEOUT_MAX_SECONDS,
);
}
- if (r.audioCheck && typeof r.audioCheck === "object") {
- const a = r.audioCheck as Record<string, unknown>;
- if (typeof a.enabled === "boolean") {
- const ac: AudioCheckConfig = { enabled: a.enabled };
- if (
- typeof a.intervalSeconds === "number" &&
- Number.isFinite(a.intervalSeconds)
- ) {
- ac.intervalSeconds = clampInt(
- a.intervalSeconds,
- AUDIO_CHECK_INTERVAL_MIN_SECONDS,
- AUDIO_CHECK_INTERVAL_MAX_SECONDS,
- );
- }
- if (
- typeof a.maxRollbacks === "number" &&
- Number.isFinite(a.maxRollbacks)
- ) {
- ac.maxRollbacks = clampInt(
- a.maxRollbacks,
- AUDIO_CHECK_MAX_ROLLBACKS_MIN,
- AUDIO_CHECK_MAX_ROLLBACKS_MAX,
- );
- }
- if (
- typeof a.copyTimeoutSeconds === "number" &&
- Number.isFinite(a.copyTimeoutSeconds)
- ) {
- ac.copyTimeoutSeconds = clampInt(
- a.copyTimeoutSeconds,
- AUDIO_CHECK_COPY_TIMEOUT_MIN_SECONDS,
- AUDIO_CHECK_COPY_TIMEOUT_MAX_SECONDS,
- );
- }
- if (typeof a.resumeDuringProbe === "boolean") {
- ac.resumeDuringProbe = a.resumeDuringProbe;
- }
- config.audioCheck = ac;
- }
+ if (typeof a.resumeDuringProbe === "boolean") {
+ ac.resumeDuringProbe = a.resumeDuringProbe;
+ }
+ return ac;
+}
+
+// THE ONE SET OF COERCIONS for a config.json field: raw value in, the legal
+// value — or `undefined`, meaning "omit the key" — out. Each is total over
+// `unknown`. parseChannelConfig (below, client-safe) and channelConfigSchema
+// (lib/channelConfigSchema.ts, server-only zod) both compose exactly these, so
+// the two cannot disagree. Keyed and ordered like CHANNEL_CONFIG_FIELD_DOCS.
+export const CHANNEL_CONFIG_COERCIONS: {
+ readonly [K in keyof ChannelConfig]-?: (v: unknown) => ChannelConfig[K] | undefined;
+} = {
+ handling: (v) => (v === "youtube" || v === "transcribe" ? v : undefined),
+ sourceKind: (v) => (v === "video" || v === "social" ? v : undefined),
+ postFetcher: trimmedNonBlank,
+ socialHandle: (v) =>
+ typeof v === "string" && v.trim() ? v.trim().replace(/^@/, "") : undefined,
+ platform: (v) =>
+ typeof v === "string" && PLATFORM_VALUES.includes(v as Platform)
+ ? (v as Platform)
+ : undefined,
+ name: str,
+ url: str,
+ audioFormat: (v) => (v === "m4a" || v === "mp3" || v === "opus" ? v : undefined),
+ downloadFormat: (v) => (isDownloadFormatPreset(v) ? v : undefined),
+ keepSourceVideo: bool,
+ keepLatest: zeroOrClamped(1, KEEP_LATEST_MAX),
+ extractionMode: (v) => (v === "ytdlp" || v === "app" ? v : undefined),
+ savedVideosDir: trimmedNonBlank,
+ dataDir: trimmedNonBlank,
+ ytdlpExtraArgs: (v) =>
+ Array.isArray(v) && v.every((x) => typeof x === "string")
+ ? (v as string[])
+ : undefined,
+ subLangs: str,
+ lastSyncedAt: str,
+ lastFullDownloadAt: str,
+ lastFullSweepAt: str,
+ excludeFromBuild: bool,
+ excludeFromCleanup: bool,
+ syncIntervalMinutes: zeroOrClamped(SYNC_INTERVAL_MIN_MINUTES, SYNC_INTERVAL_MAX_MINUTES),
+ fullSweepIntervalMinutes: zeroOrClamped(
+ SYNC_INTERVAL_MIN_MINUTES,
+ SYNC_INTERVAL_MAX_MINUTES,
+ ),
+ skipLiveDownloads: bool,
+ downloadFilter: coerceDownloadFilter,
+ cookiesFromBrowser: trimmedNonBlank,
+ cookieMode: (v) => (isCookieMode(v) ? v : undefined),
+ sleepBetweenDownloadsSeconds: (v) =>
+ isFiniteNumber(v) && v >= 0
+ ? Math.min(Math.floor(v), CHANNEL_SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS)
+ : undefined,
+ audioCheck: coerceAudioCheck,
+};
+
+// A channel's config.json as a ChannelConfig, or null when it is not a channel
+// (see isChannelShape). An ill-typed or out-of-range OPTIONAL key is OMITTED —
+// never defaulted — so `"k" in config` means "the file set it", which the
+// inherit-the-global rule for every override depends on. Unknown keys (and the
+// retired `excludeFromSync`) are dropped.
+//
+// Client-safe (no zod): the channel form and five other `"use client"` modules
+// import this file. The server reads and writes through
+// lib/channelConfigSchema.ts, which composes the same coercions.
+export function parseChannelConfig(raw: unknown): ChannelConfig | null {
+ if (!isChannelShape(raw)) return null;
+ const out: Partial<Record<keyof ChannelConfig, unknown>> = {};
+ for (const key of CHANNEL_CONFIG_KEYS) {
+ const value = CHANNEL_CONFIG_COERCIONS[key](raw[key]);
+ if (value !== undefined) out[key] = value;
}
- return config;
+ return out as ChannelConfig;
}
diff --git a/common/lib/channelConfigSchema.test.ts b/common/lib/channelConfigSchema.test.ts
@@ -0,0 +1,158 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ AUDIO_CHECK_COPY_TIMEOUT_MAX_SECONDS,
+ AUDIO_CHECK_INTERVAL_MIN_SECONDS,
+ AUDIO_CHECK_MAX_ROLLBACKS_MAX,
+ CHANNEL_CONFIG_COERCIONS,
+ CHANNEL_CONFIG_FIELD_DOCS,
+ CHANNEL_CONFIG_KEYS,
+ CHANNEL_SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS,
+ CHANNEL_SYNC_STATE_KEYS,
+ KEEP_LATEST_MAX,
+ SYNC_INTERVAL_MAX_MINUTES,
+ parseChannelConfig,
+ type ChannelConfig,
+} from "./channelConfig";
+import type { z } from "zod";
+import {
+ channelConfigObjectSchema,
+ channelConfigSchema,
+} from "./channelConfigSchema";
+
+const parse = (raw: unknown) => channelConfigSchema.parse(raw);
+
+// The zod object's output names exactly ChannelConfig's keys, and fits it —
+// `handling` aside: its coercion may answer undefined, and it is isChannelShape,
+// run first, that guarantees it.
+type Out = z.output<typeof channelConfigObjectSchema>;
+const sameKeys: [keyof Out] extends [keyof ChannelConfig]
+ ? [keyof ChannelConfig] extends [keyof Out]
+ ? true
+ : false
+ : false = true;
+const fits: Omit<Out, "handling"> extends Omit<ChannelConfig, "handling">
+ ? true
+ : false = true;
+
+test("one key list: docs = coercions = schema shape, sync-state keys inside it", () => {
+ assert.deepEqual(Object.keys(CHANNEL_CONFIG_COERCIONS), [...CHANNEL_CONFIG_KEYS]);
+ assert.deepEqual(Object.keys(channelConfigObjectSchema.shape), [...CHANNEL_CONFIG_KEYS]);
+ assert.deepEqual(Object.keys(CHANNEL_CONFIG_FIELD_DOCS), [...CHANNEL_CONFIG_KEYS]);
+ assert.equal(CHANNEL_CONFIG_KEYS.length, 29);
+ assert.equal(sameKeys, true);
+ assert.equal(fits, true);
+ for (const k of CHANNEL_SYNC_STATE_KEYS) assert.ok(CHANNEL_CONFIG_KEYS.includes(k), k);
+});
+
+test("not a channel → null", () => {
+ for (const raw of [null, undefined, 3, "youtube", [], {}, { handling: "x" }, { handling: null }]) {
+ assert.equal(parse(raw), null, JSON.stringify(raw));
+ assert.equal(parseChannelConfig(raw), null, JSON.stringify(raw));
+ }
+});
+
+test("an invalid optional key is OMITTED, not defaulted", () => {
+ const cfg = parse({
+ handling: "youtube",
+ keepLatest: -1,
+ platform: "nope",
+ postFetcher: " ",
+ audioCheck: { enabled: "yes" },
+ downloadFilter: { include: " ", rejectedLivestreams: "chat-only" },
+ ytdlpExtraArgs: ["a", 1],
+ syncIntervalMinutes: "60",
+ })!;
+ assert.deepEqual(cfg, { handling: "youtube" });
+ for (const k of ["keepLatest", "platform", "postFetcher", "audioCheck", "downloadFilter", "ytdlpExtraArgs", "syncIntervalMinutes"]) {
+ assert.equal(k in cfg, false, k);
+ }
+});
+
+test("clamps: 0 is preserved as the disabled sentinel, positives clamp, fractions floor", () => {
+ const cfg = parse({
+ handling: "transcribe",
+ keepLatest: 0,
+ syncIntervalMinutes: 0,
+ fullSweepIntervalMinutes: 0,
+ sleepBetweenDownloadsSeconds: 0,
+ })!;
+ assert.equal(cfg.keepLatest, 0);
+ assert.equal(cfg.syncIntervalMinutes, 0);
+ assert.equal(cfg.fullSweepIntervalMinutes, 0);
+ assert.equal(cfg.sleepBetweenDownloadsSeconds, 0);
+ const big = parse({
+ handling: "transcribe",
+ keepLatest: 1e9,
+ syncIntervalMinutes: 0.5,
+ fullSweepIntervalMinutes: 1e9,
+ sleepBetweenDownloadsSeconds: 1e9,
+ audioCheck: { enabled: true, intervalSeconds: 1, maxRollbacks: 1e9, copyTimeoutSeconds: 1e9 },
+ })!;
+ assert.equal(big.keepLatest, KEEP_LATEST_MAX);
+ assert.equal(big.syncIntervalMinutes, 1);
+ assert.equal(big.fullSweepIntervalMinutes, SYNC_INTERVAL_MAX_MINUTES);
+ assert.equal(big.sleepBetweenDownloadsSeconds, CHANNEL_SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS);
+ assert.deepEqual(big.audioCheck, {
+ enabled: true,
+ intervalSeconds: AUDIO_CHECK_INTERVAL_MIN_SECONDS,
+ maxRollbacks: AUDIO_CHECK_MAX_ROLLBACKS_MAX,
+ copyTimeoutSeconds: AUDIO_CHECK_COPY_TIMEOUT_MAX_SECONDS,
+ });
+ assert.equal(parse({ handling: "youtube", keepLatest: 2.9 })!.keepLatest, 2);
+});
+
+test("trims and normalises", () => {
+ const cfg = parse({
+ handling: "youtube",
+ socialHandle: " @me ",
+ savedVideosDir: " /x ",
+ dataDir: " /d ",
+ cookiesFromBrowser: " firefox ",
+ downloadFilter: { include: " a ", exclude: "", includeLivestreams: true, rejectedLivestreams: "skip" },
+ })!;
+ assert.equal(cfg.socialHandle, "me");
+ assert.equal(cfg.savedVideosDir, "/x");
+ assert.equal(cfg.dataDir, "/d");
+ assert.equal(cfg.cookiesFromBrowser, "firefox");
+ assert.deepEqual(cfg.downloadFilter, { include: "a", includeLivestreams: true });
+});
+
+test("excludeFromSync and unknown keys are dropped; every emitted key is declared", () => {
+ const cfg = parse({ handling: "youtube", excludeFromSync: true, bogus: 1, name: "n" })!;
+ assert.deepEqual(cfg, { handling: "youtube", name: "n" });
+ const full = parse({
+ handling: "transcribe",
+ sourceKind: "social",
+ name: "n",
+ url: "u",
+ lastSyncedAt: "t",
+ audioCheck: { enabled: false },
+ })!;
+ for (const k of Object.keys(full)) {
+ assert.ok((CHANNEL_CONFIG_KEYS as readonly string[]).includes(k), k);
+ }
+});
+
+test("the zod schema and the client-safe parser agree, key order included", () => {
+ const vals: unknown[] = [undefined, null, 0, -1, 2.5, 1e9, "", " x ", "@h", true, false, [], ["a"], {}, { enabled: true, intervalSeconds: 5 }, { include: "a" }];
+ const pools: Record<string, unknown[]> = {
+ handling: ["youtube", "transcribe"],
+ platform: ["youtube", "rumble", "nope"],
+ cookieMode: ["always", "never", "nope"],
+ };
+ let seed = 7;
+ const rand = () => ((seed = (seed * 1103515245 + 12345) % 2 ** 31) / 2 ** 31);
+ for (let i = 0; i < 3000; i++) {
+ const raw: Record<string, unknown> = {};
+ for (const k of [...CHANNEL_CONFIG_KEYS, "excludeFromSync", "bogus"]) {
+ if (rand() < 0.5) continue;
+ const pool = pools[k] && rand() < 0.8 ? pools[k] : vals;
+ raw[k] = pool[Math.floor(rand() * pool.length)];
+ }
+ const a = parseChannelConfig(raw);
+ const b = parse(raw);
+ assert.deepEqual(b, a, JSON.stringify(raw));
+ assert.deepEqual(Object.keys(b ?? {}), Object.keys(a ?? {}));
+ }
+});
diff --git a/common/lib/channelConfigSchema.ts b/common/lib/channelConfigSchema.ts
@@ -0,0 +1,90 @@
+// THE CHANNEL config.json SCHEMA, as zod — the server's reader and writer.
+//
+// one-core phase 3 slice 4b, on slice 4a's pattern: every field is
+// `settingsField(coerce)`, and every coerce is the entry for that key in
+// CHANNEL_CONFIG_COERCIONS (lib/channelConfig.ts) — the ONE set of coercions.
+// The client-safe `parseChannelConfig` in that file composes the same functions
+// without zod; this module is the same list as a zod object, `.describe()`d
+// from CHANNEL_CONFIG_FIELD_DOCS, for the server paths
+// (controller/channels.ts: readChannelConfigFile, writeChannelConfig).
+//
+// WHY TWO COMPOSITIONS AND NOT ONE. channelConfig.ts is imported as VALUES by
+// six `"use client"` modules; a zod import there would ship zod to the browser.
+// So the coercions live there, zod-free, and this server-only module wraps
+// them (slice 4b record, deviation 2). channelConfigSchema.test.ts pins that
+// the two agree.
+//
+// THE OMIT RULE. A config's optional keys are OMITTED when invalid, never
+// defaulted: `"keepLatest" in config` means the file set it. zod emits a key
+// whose transform returned `undefined` when that key was PRESENT in the input,
+// so `stripUndefined` runs after the object parse. The nested objects
+// (`audioCheck`, `downloadFilter`) are built by their coercions with no
+// undefined members, so the strip is one level deep.
+//
+// NOT A CHANNEL → null. The object parse runs only for a value
+// `isChannelShape` accepts (an object with a valid `handling`).
+
+import { z } from "zod";
+import {
+ CHANNEL_CONFIG_COERCIONS,
+ CHANNEL_CONFIG_FIELD_DOCS,
+ isChannelShape,
+ type ChannelConfig,
+} from "./channelConfig";
+import { settingsField } from "./settingsFieldSchemas";
+
+function field<K extends keyof ChannelConfig>(key: K) {
+ // The cast is the correlated-union case TS cannot follow through a `-?`
+ // mapped type; the map's own declaration is what checks each entry.
+ const coerce = CHANNEL_CONFIG_COERCIONS[key] as (
+ v: unknown,
+ ) => ChannelConfig[K] | undefined;
+ return settingsField(coerce).describe(CHANNEL_CONFIG_FIELD_DOCS[key]);
+}
+
+export const channelConfigObjectSchema = z.object({
+ handling: field("handling"),
+ sourceKind: field("sourceKind"),
+ postFetcher: field("postFetcher"),
+ socialHandle: field("socialHandle"),
+ platform: field("platform"),
+ name: field("name"),
+ url: field("url"),
+ audioFormat: field("audioFormat"),
+ downloadFormat: field("downloadFormat"),
+ keepSourceVideo: field("keepSourceVideo"),
+ keepLatest: field("keepLatest"),
+ extractionMode: field("extractionMode"),
+ savedVideosDir: field("savedVideosDir"),
+ dataDir: field("dataDir"),
+ ytdlpExtraArgs: field("ytdlpExtraArgs"),
+ subLangs: field("subLangs"),
+ lastSyncedAt: field("lastSyncedAt"),
+ lastFullDownloadAt: field("lastFullDownloadAt"),
+ lastFullSweepAt: field("lastFullSweepAt"),
+ excludeFromBuild: field("excludeFromBuild"),
+ excludeFromCleanup: field("excludeFromCleanup"),
+ syncIntervalMinutes: field("syncIntervalMinutes"),
+ fullSweepIntervalMinutes: field("fullSweepIntervalMinutes"),
+ skipLiveDownloads: field("skipLiveDownloads"),
+ downloadFilter: field("downloadFilter"),
+ cookiesFromBrowser: field("cookiesFromBrowser"),
+ cookieMode: field("cookieMode"),
+ sleepBetweenDownloadsSeconds: field("sleepBetweenDownloadsSeconds"),
+ audioCheck: field("audioCheck"),
+});
+
+// Drop every own key whose value is `undefined` (the OMIT rule, above).
+export function stripUndefined<T extends object>(obj: T): T {
+ const out: Record<string, unknown> = {};
+ for (const [k, v] of Object.entries(obj)) if (v !== undefined) out[k] = v;
+ return out as T;
+}
+
+export const channelConfigSchema: z.ZodType<ChannelConfig | null, unknown> =
+ z.unknown().transform((raw): ChannelConfig | null =>
+ isChannelShape(raw)
+ ? (stripUndefined(channelConfigObjectSchema.parse(raw)) as ChannelConfig)
+ : null,
+ );
+
diff --git a/common/lib/channelGroups.ts b/common/lib/channelGroups.ts
@@ -3,23 +3,33 @@
// on each ChannelConfig (group?: string) — group definitions themselves live
// in SiteSettings.groups so they can be edited centrally without touching
// every channel config when a description or default-selected value changes.
+//
+// Pure, no runtime imports: `"use client"` code imports these.
+import type { FieldDocs } from "./fieldDocs";
+
+// Each field is documented in CHANNEL_GROUP_FIELD_DOCS below.
export type ChannelGroup = {
id: string;
name: string;
description?: string;
selectedByDefault: boolean;
order?: number;
- // Provenance accent, set only in hub mode where each group is a federated
- // site (id = origin): the site's own accent, rendered as a swatch on the
- // group header so origin is legible in the channel filter. Absent in
- // single-site mode.
accent?: string;
- // Render this group's channels as loose individual chips instead of a
- // collapsible group chip; absent = false.
inline?: boolean;
};
+export const CHANNEL_GROUP_FIELD_DOCS: FieldDocs<ChannelGroup> = {
+ id: "Group id: a lowercase slug (`[a-z0-9][a-z0-9-]*`), unique within the list. An entry with an invalid or repeated id is dropped.",
+ name: 'Header text. May be blank — the UI then shows no header (and falls back to the id in admin contexts) — but must be a string. Trimmed.',
+ description: "Optional description under the header. Trimmed; blank = none.",
+ selectedByDefault: "Whether this group's channels start checked in the export UI's channel filter. Only `true` counts.",
+ order: "Optional explicit ordering hint (lower first); floored. Unordered groups sort after ordered ones, then by name.",
+ accent:
+ "Provenance accent, set only in hub mode where each group is a federated site (id = origin): that site's own accent, drawn as a swatch on the group header. Never read from a site.json — absent in single-site mode.",
+ inline: "Render this group's channels as loose individual chips instead of one collapsible group chip. Stored only when `true`.",
+};
+
export const DEFAULT_GROUP_FALLBACK_ID = "default";
// Synthesized fallback used when settings has no groups configured at all,
diff --git a/common/lib/chartsStore.ts b/common/lib/chartsStore.ts
@@ -4,7 +4,8 @@
// bundle as /chart-templates.json for the static viewer to fetch.
import path from "node:path";
-import { readFileSync, writeFileSync, renameSync, mkdirSync } from "node:fs";
+import { readFileSync } from "node:fs";
+import { writeJsonAtomicSync } from "./jsonFile-server";
import type { Paths } from "./paths";
import { siteChartTemplatesFile, siteIndexDir } from "./site";
import {
@@ -45,11 +46,9 @@ export function readTemplates(paths: Paths, siteId: string): ChartTemplates {
}
}
+// Indented, no trailing newline — these files' historical bytes.
function writeJsonAtomic(filePath: string, value: unknown): void {
- mkdirSync(path.dirname(filePath), { recursive: true });
- const tmp = `${filePath}.tmp-${process.pid}`;
- writeFileSync(tmp, JSON.stringify(value, null, 2));
- renameSync(tmp, filePath);
+ writeJsonAtomicSync(filePath, value, { newline: false, mkdir: true });
}
export function writeTemplates(
diff --git a/common/lib/clipWindow-server.ts b/common/lib/clipWindow-server.ts
@@ -2,8 +2,9 @@
// layout and for why a `clips/` subdirectory is invisible to every video-dir
// enumerator in the repo.
+import { writeJsonAtomic } from "./jsonFile-server";
import path from "node:path";
-import { mkdir, readFile, readdir, rename, stat, writeFile } from "node:fs/promises";
+import { mkdir, readFile, readdir, stat } from "node:fs/promises";
import {
clipsDirFor,
clipWindowFile,
@@ -94,9 +95,7 @@ export async function writeClipProvenance(
const dir = clipsDirFor(videoDir);
await mkdir(dir, { recursive: true });
const file = path.join(dir, clipWindowSidecar(from, to));
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(provenance, null, 2) + "\n");
- await rename(tmp, file);
+ await writeJsonAtomic(file, provenance);
return file;
}
diff --git a/common/lib/curatedTagsStore.ts b/common/lib/curatedTagsStore.ts
@@ -42,8 +42,8 @@
// touches. A per-video loop would be N read-modify-writes and could interleave
// with another writer.
-import path from "node:path";
-import { readFileSync, writeFileSync, renameSync, mkdirSync } from "node:fs";
+import { readFileSync } from "node:fs";
+import { writeJsonAtomicSync } from "./jsonFile-server";
import type { Paths } from "./paths";
import { siteTagsFile } from "./site";
import {
@@ -57,11 +57,9 @@ import {
type CuratedTagsConfig,
} from "./curatedTags";
+// Indented, no trailing newline — these files' historical bytes.
function writeJsonAtomic(filePath: string, value: unknown): void {
- mkdirSync(path.dirname(filePath), { recursive: true });
- const tmp = `${filePath}.tmp-${process.pid}`;
- writeFileSync(tmp, JSON.stringify(value, null, 2));
- renameSync(tmp, filePath);
+ writeJsonAtomicSync(filePath, value, { newline: false, mkdir: true });
}
export function emptyTagsConfig(): CuratedTagsConfig {
diff --git a/common/lib/diarization-server.ts b/common/lib/diarization-server.ts
@@ -1,33 +1,34 @@
-import path from "node:path";
-import { readFile, rename, writeFile } from "node:fs/promises";
import {
DIARIZATION_FILENAME,
type DiarizationRecord,
} from "./diarization";
+import { sidecar, sidecarField } from "./sidecar-server";
-export function diarizationPath(videoDir: string): string {
- return path.join(videoDir, DIARIZATION_FILENAME);
-}
-
-export async function loadDiarization(
- videoDir: string,
-): Promise<DiarizationRecord | null> {
- try {
- const raw = await readFile(diarizationPath(videoDir), "utf8");
- const parsed = JSON.parse(raw) as Partial<DiarizationRecord>;
- if (
- typeof parsed?.videoId === "string" &&
- typeof parsed.generatedAt === "string" &&
- Array.isArray(parsed.turns)
- ) {
- return parsed as DiarizationRecord;
- }
- return null;
- } catch {
- return null;
+export function coerceDiarization(value: unknown): DiarizationRecord | null {
+ const parsed = value as Partial<DiarizationRecord> | null;
+ if (
+ typeof parsed?.videoId === "string" &&
+ typeof parsed.generatedAt === "string" &&
+ Array.isArray(parsed.turns)
+ ) {
+ return parsed as DiarizationRecord;
}
+ return null;
}
+// Compact (one line + "\n"), as it always was.
+export const diarizationSidecar = sidecar(
+ DIARIZATION_FILENAME,
+ sidecarField(coerceDiarization),
+ { indent: 0 },
+);
+
+export const {
+ path: diarizationPath,
+ load: loadDiarization,
+ write: writeDiarization,
+} = diarizationSidecar;
+
// Presence of a VALID sidecar. This is the predicate the cleanup guard keys
// off, so it deliberately treats a malformed file as absent: a half-written
// diarization.json must not be what convinces the sweep it is safe to delete
@@ -35,13 +36,3 @@ export async function loadDiarization(
export async function hasDiarization(videoDir: string): Promise<boolean> {
return (await loadDiarization(videoDir)) !== null;
}
-
-export async function writeDiarization(
- videoDir: string,
- record: DiarizationRecord,
-): Promise<void> {
- const file = diarizationPath(videoDir);
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(record) + "\n");
- await rename(tmp, file);
-}
diff --git a/common/lib/diarization.test.ts b/common/lib/diarization.test.ts
@@ -1,4 +1,5 @@
import { test } from "node:test";
+import { SUB_FILE_RE } from "./videoStatus";
import assert from "node:assert/strict";
import path from "node:path";
import os from "node:os";
@@ -37,12 +38,13 @@ function record(over: Partial<DiarizationRecord> = {}): DiarizationRecord {
}
test("the sidecar filename is not claimed by SUB_FILE_RE as a subtitle track", () => {
- // The regex is /^transcript\.([^.]+)\.([^.]+)$/. A name like
- // transcript.diarization.json would be indexed as a `diarization`-language
- // subtitle track — which is exactly the trap this constant exists to avoid.
+ // A name like transcript.diarization.json would be indexed as a
+ // `diarization`-language subtitle track — which is exactly the trap this
+ // constant exists to avoid. (sidecar() also refuses such a name at
+ // declaration; see sidecar-server.test.ts.)
assert.equal(DIARIZATION_FILENAME, "diarization.json");
assert.equal(STATUS_FILENAME, DIARIZATION_FILENAME);
- assert.ok(!/^transcript\.([^.]+)\.([^.]+)$/.test(DIARIZATION_FILENAME));
+ assert.ok(!SUB_FILE_RE.test(DIARIZATION_FILENAME));
});
test("write then load round-trips a record", async () => {
diff --git a/common/lib/digest-server.ts b/common/lib/digest-server.ts
@@ -8,8 +8,7 @@
// replaces exactly one section and leaves the other alone, so a metered tags run
// never clobbers local chapters (and vice versa).
-import path from "node:path";
-import { readFile, rename, rm, writeFile } from "node:fs/promises";
+import { sidecar, sidecarField } from "./sidecar-server";
import {
DIGEST_FILENAME,
DIGEST_OVERRIDES_FILENAME,
@@ -33,83 +32,80 @@ import {
// iterations during Stage B tuning cannot grow an unbounded sidecar.
const MAX_HISTORY_ENTRIES = 40;
-export function digestPath(videoDir: string): string {
- return path.join(videoDir, DIGEST_FILENAME);
-}
-
-export function digestOverridesPath(videoDir: string): string {
- return path.join(videoDir, DIGEST_OVERRIDES_FILENAME);
-}
-
// ---------------------------------------------------------------------------
// Reads (tolerant: anything unparseable or structurally wrong reads as absent,
// so one corrupt sidecar can never fail a channel-wide sweep)
// ---------------------------------------------------------------------------
-export async function loadDigest(
- videoDir: string,
-): Promise<DigestRecord | null> {
- try {
- const raw = await readFile(digestPath(videoDir), "utf8");
- const parsed = JSON.parse(raw) as Partial<DigestRecord>;
- if (typeof parsed?.digestSchemaVersion !== "number") return null;
- if (!parsed.sections || typeof parsed.sections !== "object") return null;
- return {
- digestSchemaVersion: parsed.digestSchemaVersion,
- promptVersion:
- typeof parsed.promptVersion === "number" ? parsed.promptVersion : 0,
- contextHash:
- typeof parsed.contextHash === "string" ? parsed.contextHash : "",
- warnings: Array.isArray(parsed.warnings)
- ? (parsed.warnings as DigestWarning[])
- : [],
- sections: parsed.sections,
- // This reader rebuilds the record field by field rather than spreading,
- // so EVERY new field has to be added here or it round-trips to nothing —
- // silently, since the write succeeds and the read just omits it.
- ...(Array.isArray(parsed.failures)
- ? { failures: parsed.failures as DigestSectionFailure[] }
- : {}),
- ...(Array.isArray(parsed.history)
- ? { history: parsed.history as DigestHistoryEntry[] }
- : {}),
- ...(parsed.derivedFrom
- ? { derivedFrom: parsed.derivedFrom as DigestDerivedFrom }
- : {}),
- };
- } catch {
- return null;
- }
+export function coerceDigest(value: unknown): DigestRecord | null {
+ const parsed = value as Partial<DigestRecord> | null;
+ if (typeof parsed?.digestSchemaVersion !== "number") return null;
+ if (!parsed.sections || typeof parsed.sections !== "object") return null;
+ return {
+ digestSchemaVersion: parsed.digestSchemaVersion,
+ promptVersion:
+ typeof parsed.promptVersion === "number" ? parsed.promptVersion : 0,
+ contextHash:
+ typeof parsed.contextHash === "string" ? parsed.contextHash : "",
+ warnings: Array.isArray(parsed.warnings)
+ ? (parsed.warnings as DigestWarning[])
+ : [],
+ sections: parsed.sections,
+ // This reader rebuilds the record field by field rather than spreading,
+ // so EVERY new field has to be added here or it round-trips to nothing —
+ // silently, since the write succeeds and the read just omits it.
+ ...(Array.isArray(parsed.failures)
+ ? { failures: parsed.failures as DigestSectionFailure[] }
+ : {}),
+ ...(Array.isArray(parsed.history)
+ ? { history: parsed.history as DigestHistoryEntry[] }
+ : {}),
+ ...(parsed.derivedFrom
+ ? { derivedFrom: parsed.derivedFrom as DigestDerivedFrom }
+ : {}),
+ };
}
-export async function loadDigestOverrides(
- videoDir: string,
-): Promise<DigestOverrides | null> {
- try {
- const raw = await readFile(digestOverridesPath(videoDir), "utf8");
- const parsed = JSON.parse(raw) as Partial<DigestOverrides>;
- // A hand-authored file may omit `version`; treat it as current rather than
- // discarding human work over a missing scalar.
- const chapters = sanitizeChapters(parsed.chapters);
- const tags = sanitizeTags(parsed.tags);
- if (chapters.length === 0 && tags.length === 0 && !parsed.note) return null;
- return {
- version:
- typeof parsed.version === "number"
- ? parsed.version
- : DIGEST_OVERRIDES_VERSION,
- ...(chapters.length > 0 ? { chapters } : {}),
- ...(tags.length > 0 ? { tags } : {}),
- ...(typeof parsed.note === "string" ? { note: parsed.note } : {}),
- ...(typeof parsed.updatedAt === "string"
- ? { updatedAt: parsed.updatedAt }
- : {}),
- };
- } catch {
- return null;
- }
+export function coerceDigestOverrides(value: unknown): DigestOverrides | null {
+ // `null` threw inside the old try (`parsed.chapters` on null) and so read as
+ // absent; the explicit guard keeps that.
+ if (value === null || value === undefined) return null;
+ const parsed = value as Partial<DigestOverrides>;
+ // A hand-authored file may omit `version`; treat it as current rather than
+ // discarding human work over a missing scalar.
+ const chapters = sanitizeChapters(parsed.chapters);
+ const tags = sanitizeTags(parsed.tags);
+ if (chapters.length === 0 && tags.length === 0 && !parsed.note) return null;
+ return {
+ version:
+ typeof parsed.version === "number"
+ ? parsed.version
+ : DIGEST_OVERRIDES_VERSION,
+ ...(chapters.length > 0 ? { chapters } : {}),
+ ...(tags.length > 0 ? { tags } : {}),
+ ...(typeof parsed.note === "string" ? { note: parsed.note } : {}),
+ ...(typeof parsed.updatedAt === "string"
+ ? { updatedAt: parsed.updatedAt }
+ : {}),
+ };
}
+// Two sidecars, each read field by field above (so the read is also what drops
+// a field it does not name — see the comment in coerceDigest).
+export const digestSidecar = sidecar(DIGEST_FILENAME, sidecarField(coerceDigest));
+export const digestOverridesSidecar = sidecar(
+ DIGEST_OVERRIDES_FILENAME,
+ sidecarField(coerceDigestOverrides),
+);
+
+export const {
+ path: digestPath,
+ load: loadDigest,
+ write: writeDigest,
+} = digestSidecar;
+export const { path: digestOverridesPath, load: loadDigestOverrides } =
+ digestOverridesSidecar;
+
// Hand-authored overrides are coerced, not trusted: an entry missing an id or a
// body is dropped rather than poisoning the merge.
function sanitizeChapters(value: unknown): DigestChapter[] {
@@ -161,19 +157,6 @@ function sanitizeTags(value: unknown): DigestTag[] {
// Writes
// ---------------------------------------------------------------------------
-async function writeJsonAtomic(file: string, value: unknown): Promise<void> {
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(value, null, 2) + "\n");
- await rename(tmp, file);
-}
-
-export async function writeDigest(
- videoDir: string,
- record: DigestRecord,
-): Promise<void> {
- await writeJsonAtomic(digestPath(videoDir), record);
-}
-
export type WriteDigestSectionInput = {
section: DigestSectionKind;
items: DigestItem[];
@@ -320,14 +303,13 @@ export async function writeDigestOverrides(
(overrides.chapters?.length ?? 0) > 0 ||
(overrides.tags?.length ?? 0) > 0 ||
Boolean(overrides.note);
- const file = digestOverridesPath(videoDir);
if (!hasContent) {
// Emptying the override list means "revert to machine output" — remove the
// file rather than leaving an empty shadow behind.
- await rm(file, { force: true });
+ await digestOverridesSidecar.remove(videoDir);
return;
}
- await writeJsonAtomic(file, {
+ await digestOverridesSidecar.write(videoDir, {
version: DIGEST_OVERRIDES_VERSION,
...(overrides.chapters?.length ? { chapters: overrides.chapters } : {}),
...(overrides.tags?.length ? { tags: overrides.tags } : {}),
diff --git a/common/lib/doNotClean-server.ts b/common/lib/doNotClean-server.ts
@@ -1,30 +1,24 @@
-import path from "node:path";
-import { readFile, rename, rm, writeFile } from "node:fs/promises";
import {
DO_NOT_CLEAN_FILENAME,
type DoNotCleanRecord,
} from "./doNotClean";
+import { sidecar, sidecarField } from "./sidecar-server";
-export function doNotCleanPath(videoDir: string): string {
- return path.join(videoDir, DO_NOT_CLEAN_FILENAME);
+// A parseable-but-malformed marker still means "protected" — it reads as an
+// empty record rather than as absent. Only an absent, unreadable or
+// unparseable file reads as null (sidecar-server's rule).
+export function coerceDoNotClean(value: unknown): DoNotCleanRecord {
+ const parsed = value as Partial<DoNotCleanRecord> | null;
+ if (typeof parsed?.setAt === "string") return parsed as DoNotCleanRecord;
+ return { setAt: "" };
}
-export async function loadDoNotClean(
- videoDir: string,
-): Promise<DoNotCleanRecord | null> {
- try {
- const raw = await readFile(doNotCleanPath(videoDir), "utf8");
- const parsed = JSON.parse(raw) as Partial<DoNotCleanRecord>;
- if (typeof parsed?.setAt === "string") {
- return parsed as DoNotCleanRecord;
- }
- // A parseable-but-malformed marker still means "protected" — fall back to
- // an empty record rather than treating it as absent.
- return { setAt: "" };
- } catch {
- return null;
- }
-}
+export const doNotCleanSidecar = sidecar(
+ DO_NOT_CLEAN_FILENAME,
+ sidecarField(coerceDoNotClean),
+);
+
+export const { path: doNotCleanPath, load: loadDoNotClean } = doNotCleanSidecar;
// Presence of a valid sidecar = protected. Cheap existence check for the
// cleanup controllers' per-dir loops.
@@ -37,16 +31,9 @@ export async function setDoNotClean(
enabled: boolean,
note?: string,
): Promise<void> {
- const file = doNotCleanPath(videoDir);
- if (!enabled) {
- await rm(file, { force: true });
- return;
- }
- const record: DoNotCleanRecord = {
+ if (!enabled) return doNotCleanSidecar.remove(videoDir);
+ await doNotCleanSidecar.write(videoDir, {
setAt: new Date().toISOString(),
...(note ? { note } : {}),
- };
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(record, null, 2) + "\n");
- await rename(tmp, file);
+ });
}
diff --git a/common/lib/downloadOutcome-server.ts b/common/lib/downloadOutcome-server.ts
@@ -1,48 +1,39 @@
-import path from "node:path";
-import { readFile, rename, writeFile } from "node:fs/promises";
import {
DOWNLOAD_OUTCOME_FILENAME,
DOWNLOAD_OUTCOME_STATUS_VALUES,
type DownloadOutcomeRecord,
type DownloadOutcomeStatus,
} from "./downloadOutcome";
+import { sidecar, sidecarField } from "./sidecar-server";
-export function downloadOutcomePath(videoDir: string): string {
- return path.join(videoDir, DOWNLOAD_OUTCOME_FILENAME);
-}
-
-export async function loadDownloadOutcome(
- videoDir: string,
-): Promise<DownloadOutcomeRecord | null> {
- try {
- const raw = await readFile(downloadOutcomePath(videoDir), "utf8");
- const parsed = JSON.parse(raw) as Partial<DownloadOutcomeRecord>;
- if (
- typeof parsed?.status === "string" &&
- (DOWNLOAD_OUTCOME_STATUS_VALUES as string[]).includes(parsed.status) &&
- typeof parsed.videoId === "string" &&
- typeof parsed.startedAt === "string" &&
- typeof parsed.finishedAt === "string" &&
- Array.isArray(parsed.attempts)
- ) {
- // Cast: status is already narrowed to a member of the enum tuple.
- return parsed as DownloadOutcomeRecord;
- }
- return null;
- } catch {
- return null;
+export function coerceDownloadOutcome(
+ value: unknown,
+): DownloadOutcomeRecord | null {
+ const parsed = value as Partial<DownloadOutcomeRecord> | null;
+ if (
+ typeof parsed?.status === "string" &&
+ (DOWNLOAD_OUTCOME_STATUS_VALUES as string[]).includes(parsed.status) &&
+ typeof parsed.videoId === "string" &&
+ typeof parsed.startedAt === "string" &&
+ typeof parsed.finishedAt === "string" &&
+ Array.isArray(parsed.attempts)
+ ) {
+ // Cast: status is already narrowed to a member of the enum tuple.
+ return parsed as DownloadOutcomeRecord;
}
+ return null;
}
-export async function writeDownloadOutcome(
- videoDir: string,
- record: DownloadOutcomeRecord,
-): Promise<void> {
- const file = downloadOutcomePath(videoDir);
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(record, null, 2) + "\n");
- await rename(tmp, file);
-}
+export const downloadOutcomeSidecar = sidecar(
+ DOWNLOAD_OUTCOME_FILENAME,
+ sidecarField(coerceDownloadOutcome),
+);
+
+export const {
+ path: downloadOutcomePath,
+ load: loadDownloadOutcome,
+ write: writeDownloadOutcome,
+} = downloadOutcomeSidecar;
// Narrow re-export so callers don't have to import from both modules.
export type { DownloadOutcomeStatus };
diff --git a/common/lib/excludeTruncatedCheck-server.ts b/common/lib/excludeTruncatedCheck-server.ts
@@ -1,30 +1,30 @@
-import path from "node:path";
-import { readFile, rename, rm, writeFile } from "node:fs/promises";
import {
EXCLUDE_TRUNCATED_CHECK_FILENAME,
type ExcludeTruncatedCheckRecord,
} from "./excludeTruncatedCheck";
+import { sidecar, sidecarField } from "./sidecar-server";
-export function excludeTruncatedCheckPath(videoDir: string): string {
- return path.join(videoDir, EXCLUDE_TRUNCATED_CHECK_FILENAME);
-}
-
-export async function loadExcludeTruncatedCheck(
- videoDir: string,
-): Promise<ExcludeTruncatedCheckRecord | null> {
- try {
- const raw = await readFile(excludeTruncatedCheckPath(videoDir), "utf8");
- const parsed = JSON.parse(raw) as Partial<ExcludeTruncatedCheckRecord>;
- if (typeof parsed?.setAt === "string") {
- return parsed as ExcludeTruncatedCheckRecord;
- }
- // Parseable-but-malformed still means "excluded".
- return { setAt: "" };
- } catch {
- return null;
+// Parseable-but-malformed still means "excluded" — the doNotClean rule.
+export function coerceExcludeTruncatedCheck(
+ value: unknown,
+): ExcludeTruncatedCheckRecord {
+ const parsed = value as Partial<ExcludeTruncatedCheckRecord> | null;
+ if (typeof parsed?.setAt === "string") {
+ return parsed as ExcludeTruncatedCheckRecord;
}
+ return { setAt: "" };
}
+export const excludeTruncatedCheckSidecar = sidecar(
+ EXCLUDE_TRUNCATED_CHECK_FILENAME,
+ sidecarField(coerceExcludeTruncatedCheck),
+);
+
+export const {
+ path: excludeTruncatedCheckPath,
+ load: loadExcludeTruncatedCheck,
+} = excludeTruncatedCheckSidecar;
+
// Presence of a valid sidecar = excluded from the truncated check. Cheap
// existence check for the snapshot generator's per-dir loop.
export async function isExcludedFromTruncatedCheck(
@@ -38,16 +38,9 @@ export async function setExcludedFromTruncatedCheck(
enabled: boolean,
note?: string,
): Promise<void> {
- const file = excludeTruncatedCheckPath(videoDir);
- if (!enabled) {
- await rm(file, { force: true });
- return;
- }
- const record: ExcludeTruncatedCheckRecord = {
+ if (!enabled) return excludeTruncatedCheckSidecar.remove(videoDir);
+ await excludeTruncatedCheckSidecar.write(videoDir, {
setAt: new Date().toISOString(),
...(note ? { note } : {}),
- };
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(record, null, 2) + "\n");
- await rename(tmp, file);
+ });
}
diff --git a/common/lib/fileSchemaDocs.test.ts b/common/lib/fileSchemaDocs.test.ts
@@ -0,0 +1,60 @@
+import { readFileSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { renderChannelMarkdown, renderSiteMarkdown } from "./fileSchemaDocs";
+import { CHANNEL_CONFIG_KEYS, parseChannelConfig } from "./channelConfig";
+import { SITE_KEYS, parseSite, siteToDisk } from "./siteSchema";
+
+// SITE.md and CHANNEL.md are GENERATED from the file schemas
+// (common/bin/file-schemas-docs.ts). This is what keeps them generated: a hand
+// edit to either file, or a schema change without a regenerate, fails here.
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+for (const [name, render] of [
+ ["SITE.md", renderSiteMarkdown],
+ ["CHANNEL.md", renderChannelMarkdown],
+] as const) {
+ test(`${name} is what the schema generates`, () => {
+ const committed = readFileSync(path.join(REPO, name), "utf8");
+ assert.equal(
+ committed,
+ render(),
+ `${name} is stale: run pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`,
+ );
+ });
+}
+
+test("every key has a row", () => {
+ const site = renderSiteMarkdown();
+ for (const key of SITE_KEYS) assert.ok(site.includes(`| [\`${key}\`]`), key);
+ const channel = renderChannelMarkdown();
+ for (const key of CHANNEL_CONFIG_KEYS) assert.ok(channel.includes(`| \`${key}\` |`), key);
+});
+
+test("CHANNEL.md's smallest channel is a channel", () => {
+ const m = /`(\{ "handling"[^`]*\})`/.exec(renderChannelMarkdown());
+ assert.ok(m, "the one-line example is present");
+ assert.deepEqual(parseChannelConfig(JSON.parse(m![1])), {
+ handling: "youtube",
+ url: "https://www.youtube.com/@example",
+ });
+});
+
+test("SITE.md names exactly the keys a save always writes", () => {
+ const always = Object.keys(siteToDisk(parseSite("x", {})));
+ const m = /A save ALWAYS writes ([^;]*);/.exec(renderSiteMarkdown());
+ assert.ok(m, "the always-written sentence is present");
+ const named = [...m![1].matchAll(/`([a-zA-Z]+)`/g)].map((x) => x[1]);
+ assert.deepEqual(named, always);
+});
+
+test("CHANNEL.md renders each nested table under one heading", () => {
+ const md = renderChannelMarkdown();
+ for (const key of ["downloadFilter", "audioCheck"]) {
+ const headings = md.split("\n").filter((l) => l.startsWith("#") && l.includes(`\`${key}\``));
+ assert.equal(headings.length, 1, key);
+ }
+});
diff --git a/common/lib/fileSchemaDocs.ts b/common/lib/fileSchemaDocs.ts
@@ -0,0 +1,199 @@
+// THE TWO KEY TABLES GENERATED FROM THE FILE SCHEMAS: SITE.md (a site's
+// `site.json`) and CHANNEL.md (a channel's `config.json`), both at the repo
+// root.
+//
+// one-core phase 3 slice 4b; the sibling of settingsDocs.ts (SETTINGS.md), and
+// rendered with its table renderer. Pure renderers — common/bin/
+// file-schemas-docs.ts writes the files, and fileSchemaDocs.test.ts asserts the
+// committed bytes are what these return, so neither can be edited by hand.
+//
+// Everything comes from the schemas' docs records: the keys and their order
+// from SITE_FIELD_DOCS / CHANNEL_CONFIG_FIELD_DOCS, the defaults (site.json only
+// — a channel's optional keys have no default, absent means inherit) from
+// `parseSite(id, {})`, the nested tables from the records beside each type.
+//
+// No `.example` files: a site needs an id (its directory), and the smallest
+// channel is one line, spelled out in CHANNEL.md.
+
+import { CHANNEL_GROUP_FIELD_DOCS } from "./channelGroups";
+import {
+ AUDIO_CHECK_FIELD_DOCS,
+ CHANNEL_CONFIG_FIELD_DOCS,
+ CHANNEL_SYNC_STATE_KEYS,
+ DOWNLOAD_FILTER_FIELD_DOCS,
+} from "./channelConfig";
+import { SOCIAL_LINK_FIELD_DOCS } from "./settingsSchema";
+import {
+ RELATED_SITE_GROUP_FIELD_DOCS,
+ SITE_CHANNEL_MEMBERSHIP_FIELD_DOCS,
+ SITE_FIELD_DOCS,
+ parseSite,
+ type Site,
+} from "./siteSchema";
+import { cell, defaultCell, isScalar, renderTable, type KeyTable } from "./settingsDocs";
+
+const GENERATED =
+ "<!-- GENERATED by common/bin/file-schemas-docs.ts from the *_FIELD_DOCS records beside each type — do not edit by hand. -->";
+
+const REGENERATE =
+ "Regenerate this file with " +
+ "`pnpm --filter yt-dlp-transcript-common exec tsx bin/file-schemas-docs.ts`.";
+
+// The default as a cell, where `undefined` — a key parseSite emits but a file
+// need not spell — is "absent".
+function siteDefaultCell(v: unknown): string {
+ return v === undefined ? "absent" : defaultCell(v);
+}
+
+const SITE_NESTED: Partial<Record<keyof Site, KeyTable[]>> = {
+ socialLinks: [{ path: "socialLinks[]", docs: SOCIAL_LINK_FIELD_DOCS }],
+ groups: [{ path: "groups[]", docs: CHANNEL_GROUP_FIELD_DOCS }],
+ channels: [{ path: "channels[]", docs: SITE_CHANNEL_MEMBERSHIP_FIELD_DOCS }],
+ relatedSites: [{ path: "relatedSites[]", docs: RELATED_SITE_GROUP_FIELD_DOCS }],
+};
+
+export function renderSiteMarkdown(): string {
+ const d = parseSite("<id>", {}) as Record<string, unknown>;
+ const keys = Object.keys(SITE_FIELD_DOCS) as Array<keyof Site>;
+ const out: string[] = [];
+ out.push("# site.json keys");
+ out.push("");
+ out.push(GENERATED);
+ out.push("");
+ out.push(
+ "One public site: its branding, its channel grouping and which channels " +
+ "it exposes, persisted to `transcripts/sites/<id>/site.json` (the " +
+ "directory under `$SITES_DIR` when that is set). The schema is " +
+ "`common/lib/siteSchema.ts`. Global operational settings are " +
+ "`settings.json` — see [SETTINGS.md](SETTINGS.md). The PUBLIC " +
+ "`/site.json` a built site serves is a different file " +
+ "(`common/lib/siteDescriptor.ts`).",
+ );
+ out.push("");
+ out.push(
+ "Every key is optional on read. A missing key reads as its default, an " +
+ "ill-typed one as its default (or is dropped, for the optional ones), " +
+ "and an unknown one is dropped on the next save. A save ALWAYS writes " +
+ "`siteId`, `siteTitle`, `siteDescription`, `headerTitle`, " +
+ "`homeTagline`, `groups`, `defaultGroupId` and `channels`; every other " +
+ "key is written only when it differs from its default (`socialLinks` " +
+ "whenever it is an array, even an empty one). A save is REFUSED when " +
+ "there is no channel group, when `defaultGroupId` names no group, or " +
+ "when a social link's SVG is not safe to inline.",
+ );
+ out.push("");
+ out.push(REGENERATE);
+ out.push("");
+ out.push("| Key | Default |");
+ out.push("|---|---|");
+ for (const key of keys) {
+ const def = key === "siteId" ? "the directory name" : siteDefaultCell(d[key]);
+ out.push(`| [\`${key}\`](#${key.toLowerCase()}) | ${def} |`);
+ }
+ out.push("");
+ for (const key of keys) {
+ out.push(`## \`${key}\``);
+ out.push("");
+ out.push(SITE_FIELD_DOCS[key]);
+ out.push("");
+ const v = d[key];
+ if (key === "siteId") continue;
+ if (v === undefined || isScalar(v)) {
+ out.push(`Default: ${siteDefaultCell(v)}`);
+ out.push("");
+ for (const table of SITE_NESTED[key] ?? []) renderTable(out, table);
+ } else {
+ for (const table of SITE_NESTED[key] ?? []) renderTable(out, table);
+ out.push("Default:");
+ out.push("");
+ out.push("```json");
+ out.push(JSON.stringify(v, null, 2));
+ out.push("```");
+ out.push("");
+ }
+ }
+ return out.join("\n");
+}
+
+// A nested block's keys are, like the top level's, absent unless spelled —
+// except `audioCheck.enabled`, without which the block is dropped.
+const CHANNEL_NESTED: Partial<Record<string, KeyTable[]>> = {
+ downloadFilter: [
+ { path: "downloadFilter", docs: DOWNLOAD_FILTER_FIELD_DOCS, defaults: () => "absent" },
+ ],
+ audioCheck: [
+ {
+ path: "audioCheck",
+ docs: AUDIO_CHECK_FIELD_DOCS,
+ defaults: (key) => (key === "enabled" ? "required" : "absent"),
+ },
+ ],
+};
+
+function channelKind(key: string): string {
+ if (key === "handling") return "required";
+ if ((CHANNEL_SYNC_STATE_KEYS as readonly string[]).includes(key)) return "sync state";
+ return "config";
+}
+
+export function renderChannelMarkdown(): string {
+ const keys = Object.keys(CHANNEL_CONFIG_FIELD_DOCS);
+ const out: string[] = [];
+ out.push("# Channel config.json keys");
+ out.push("");
+ out.push(GENERATED);
+ out.push("");
+ out.push(
+ "One channel of the corpus, persisted to " +
+ "`transcripts/channels/<slug>/config.json`. The schema is " +
+ "`common/lib/channelConfigSchema.ts` over the coercions in " +
+ "`common/lib/channelConfig.ts`. Which sites expose a channel is " +
+ "`site.json`'s business — see [SITE.md](SITE.md); global settings are " +
+ "[SETTINGS.md](SETTINGS.md).",
+ );
+ out.push("");
+ out.push(
+ "`handling` is the one required key: a file without a valid one is not a " +
+ "channel. The smallest channel is " +
+ '`{ "handling": "youtube", "url": "https://www.youtube.com/@example" }`.',
+ );
+ out.push("");
+ out.push(
+ "Every other key is optional and has NO default of its own: an absent " +
+ "key means whatever its description says — for the per-channel " +
+ "overrides, inherit the global setting of the same name; for `name`, " +
+ "`url`, `dataDir`, `subLangs` and the sync-state stamps, simply unset. " +
+ "So an ill-typed or out-of-range value is not coerced — it is DROPPED, " +
+ "as if the file did not spell it. Unknown keys (including the retired " +
+ "`excludeFromSync`, now a paused `sync` tier in the channel-priority " +
+ "document) are dropped by every read and every write.",
+ );
+ out.push("");
+ out.push(
+ "The three **sync state** keys are not configuration: the sync, sweep and " +
+ "download passes stamp them, the channel form never does, and they live " +
+ "in the same file on purpose. Writers after creation PATCH " +
+ "(`patchChannelConfig`): each re-reads the file at the moment it writes " +
+ "and changes only its own keys, so a stamp and a form save made at once " +
+ "in the editor both land. The two exceptions write a whole config, and " +
+ "only when there is no readable file to patch: a media move and a " +
+ "channel rename record `dataDir` from their own copy of the config.",
+ );
+ out.push("");
+ out.push(REGENERATE);
+ out.push("");
+ out.push("| Key | Kind | Description |");
+ out.push("|---|---|---|");
+ for (const key of keys) {
+ const docs = (CHANNEL_CONFIG_FIELD_DOCS as Record<string, string>)[key];
+ const nested = CHANNEL_NESTED[key] ? ` See [\`${key}\`](#${key.toLowerCase()}).` : "";
+ out.push(`| \`${key}\` | ${channelKind(key)} | ${cell(docs)}${nested} |`);
+ }
+ out.push("");
+ // Each nested table is rendered with its own `#### <path>` heading, which
+ // is the anchor the row links above point at.
+ for (const key of Object.keys(CHANNEL_NESTED)) {
+ for (const table of CHANNEL_NESTED[key] ?? []) renderTable(out, table);
+ }
+ return out.join("\n");
+}
diff --git a/common/lib/jsonFile-server.test.ts b/common/lib/jsonFile-server.test.ts
@@ -0,0 +1,125 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import fs from "node:fs";
+import { mkdtemp, readFile, readdir, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import {
+ jsonFileState,
+ jsonText,
+ pendingJsonWrites,
+ readJsonFile,
+ readJsonFileSync,
+ tmpPathFor,
+ writeJsonAtomic,
+ writeJsonAtomicSync,
+} from "./jsonFile-server";
+
+async function scratch(): Promise<string> {
+ return mkdtemp(path.join(os.tmpdir(), "jsonfile-"));
+}
+
+test("readJsonFile: ok, absent, unparseable, unreadable — and the sync twin agrees", async () => {
+ const dir = await scratch();
+ const good = path.join(dir, "good.json");
+ const bad = path.join(dir, "bad.json");
+ await writeFile(good, '{"a":1}');
+ await writeFile(bad, '{"a":');
+ const cases: Array<[string, unknown]> = [
+ [good, { ok: true, value: { a: 1 } }],
+ [path.join(dir, "missing.json"), { ok: false, reason: "absent" }],
+ [path.join(dir, "no-such-dir", "x.json"), { ok: false, reason: "absent" }],
+ [bad, { ok: false, reason: "unparseable" }],
+ // A directory is readable as a path but not as a file: EISDIR.
+ [dir, { ok: false, reason: "unreadable" }],
+ ];
+ for (const [file, want] of cases) {
+ assert.deepEqual(await readJsonFile(file), want, file);
+ assert.deepEqual(readJsonFileSync(file), want, file);
+ }
+});
+
+test("jsonText: indent 2 vs 0, newline on by default", () => {
+ const v = { a: [1, 2] };
+ assert.equal(jsonText(v), '{\n "a": [\n 1,\n 2\n ]\n}\n');
+ assert.equal(jsonText(v, { indent: 0 }), '{"a":[1,2]}\n');
+ assert.equal(jsonText(v, { indent: 0, newline: false }), '{"a":[1,2]}');
+ assert.equal(jsonText(v, { newline: false }), '{\n "a": [\n 1,\n 2\n ]\n}');
+});
+
+test("writeJsonAtomic: the bytes on disk are jsonText's, per option", async () => {
+ const dir = await scratch();
+ for (const opts of [{}, { indent: 0 as const }, { indent: 0 as const, newline: false }]) {
+ const file = path.join(dir, `f-${JSON.stringify(opts)}.json`);
+ await writeJsonAtomic(file, { x: "y" }, opts);
+ assert.equal(await readFile(file, "utf8"), jsonText({ x: "y" }, opts));
+ }
+});
+
+test("temp names are unique per write, not per process", () => {
+ const a = tmpPathFor("/x/config.json");
+ const b = tmpPathFor("/x/config.json");
+ assert.notEqual(a, b);
+ assert.ok(a.startsWith(`/x/config.json.tmp-${process.pid}-`));
+ assert.match(a, /\.tmp-\d+-\d+-[0-9a-f]{8}$/);
+});
+
+test("a second copy of the module shares the one chain on globalThis", async () => {
+ // Next can load this module twice in one server; simulate the second copy
+ // with a fresh import (a distinct module URL) and check both see one state.
+ const copy = (await import(`./jsonFile-server.ts?copy=${Date.now()}`)) as typeof import("./jsonFile-server");
+ assert.notEqual(copy.writeJsonAtomic, writeJsonAtomic, "a genuinely separate module instance");
+ assert.equal(copy.jsonFileState(), jsonFileState());
+ const dir = await scratch();
+ const file = path.join(dir, "shared.json");
+ const writes = [];
+ for (let i = 0; i < 20; i++) {
+ const w = i % 2 === 0 ? writeJsonAtomic : copy.writeJsonAtomic;
+ writes.push(w(file, { i }));
+ }
+ // While in flight, both copies see the same chain entry.
+ assert.equal(copy.pendingJsonWrites(), pendingJsonWrites());
+ await Promise.all(writes);
+ assert.deepEqual(JSON.parse(await readFile(file, "utf8")), { i: 19 });
+ assert.deepEqual(await readdir(dir), ["shared.json"]);
+});
+
+test("two same-path writers serialise: the last ISSUED lands, no temp is left, no write fails", async () => {
+ const dir = await scratch();
+ const file = path.join(dir, "config.json");
+ const writes = [];
+ for (let i = 0; i < 50; i++) writes.push(writeJsonAtomic(file, { i }));
+ await Promise.all(writes);
+ assert.deepEqual(JSON.parse(await readFile(file, "utf8")), { i: 49 });
+ assert.deepEqual(await readdir(dir), ["config.json"]);
+ assert.equal(pendingJsonWrites(), 0);
+});
+
+test("the value is serialised at the call, not when the chain reaches it", async () => {
+ const dir = await scratch();
+ const file = path.join(dir, "v.json");
+ const v = { n: 1 };
+ const first = writeJsonAtomic(file, { hold: true });
+ const second = writeJsonAtomic(file, v);
+ v.n = 2;
+ await Promise.all([first, second]);
+ assert.deepEqual(JSON.parse(await readFile(file, "utf8")), { n: 1 });
+});
+
+test("a failed write does not block the next one on the same path, and cleans its temp", async () => {
+ const dir = await scratch();
+ const file = path.join(dir, "sub", "x.json");
+ // No parent dir and no mkdir: this one fails.
+ await assert.rejects(writeJsonAtomic(file, { a: 1 }));
+ await writeJsonAtomic(file, { a: 2 }, { mkdir: true });
+ assert.deepEqual(JSON.parse(await readFile(file, "utf8")), { a: 2 });
+ assert.deepEqual(await readdir(path.dirname(file)), ["x.json"]);
+});
+
+test("writeJsonAtomicSync: same bytes, mkdir, no temp left", async () => {
+ const dir = await scratch();
+ const file = path.join(dir, "a", "b.json");
+ writeJsonAtomicSync(file, { k: 1 }, { newline: false, mkdir: true });
+ assert.equal(fs.readFileSync(file, "utf8"), '{\n "k": 1\n}');
+ assert.deepEqual(await readdir(path.dirname(file)), ["b.json"]);
+});
diff --git a/common/lib/jsonFile-server.ts b/common/lib/jsonFile-server.ts
@@ -0,0 +1,221 @@
+// ONE JSON READER AND ONE ATOMIC JSON WRITER for the config files and sidecars.
+// (Not yet every JSON writer: the slice 4b record lists the ones still on the
+// per-pid temp name.)
+//
+// one-core phase 3 slice 4b. Before this module the repo had seven private
+// copies of `writeJsonAtomic` and some twenty inline `tmp + rename` writes, all
+// of them spelling the temp file `${file}.tmp-${process.pid}`. That name is the
+// same for every write this process makes to one file, so two writers in the
+// same process — a sync stamping `lastSyncedAt` while the channel form saves
+// `config.json` — shared one temp file: the second `writeFile` truncated the
+// first's bytes and one `rename` then found no file. This module fixes both
+// halves of that:
+//
+// - every temp name is unique (`${file}.tmp-${pid}-${seq}-${random}`), and
+// - writes to one ABSOLUTE PATH are CHAINED: a write starts only once the
+// previous write to that path has settled, so two same-process writers
+// serialise and the last one to be ISSUED is the one on disk.
+//
+// ONE CHAIN PER PROCESS, NOT PER MODULE COPY. Next can load this module more
+// than once in one server (instrumentation.ts arms the runners; server actions
+// are another bundle layer), so the counter, the write chains and the locks
+// live on `globalThis` — the house pattern (jobs/registry.ts,
+// controller/autoRunner.ts) — and the temp name also carries random bytes, so
+// its uniqueness never rests on a shared counter alone.
+//
+// The chain is per process. A CLI run beside a live editor (e.g.
+// `migrate-channel-priority.ts`) is a different process and is NOT covered —
+// the rename is still atomic, so a reader never sees a torn file, but the
+// last rename wins.
+//
+// The chain serialises WRITES, not read-modify-write cycles. A caller that reads,
+// changes and writes back takes `withJsonFileLock` around the whole cycle, so
+// its read is current — see `patchChannelConfig` in controller/channels.ts.
+//
+// BYTES ARE PRESERVED, per caller. `indent` and `newline` are options, not a
+// house style, because the bytes on disk are what the phase-3 numbers diff: the
+// attribution and diarization sidecars are compact with a trailing newline,
+// the export build's pages are compact without one, the chart/alias/tag stores
+// are indented without one, and everything else is indented with one.
+//
+// SERVER-ONLY (node:fs). Named `-server` so a `"use client"` graph never
+// reaches it.
+
+import { randomBytes } from "node:crypto";
+import fs from "node:fs";
+import { mkdir, readFile, rename, rm, writeFile } from "node:fs/promises";
+import path from "node:path";
+
+export type ReadJsonResult =
+ | { ok: true; value: unknown }
+ | { ok: false; reason: "absent" | "unreadable" | "unparseable" };
+
+function readFailure(err: unknown): ReadJsonResult {
+ const code = (err as NodeJS.ErrnoException | null)?.code;
+ return {
+ ok: false,
+ reason: code === "ENOENT" || code === "ENOTDIR" ? "absent" : "unreadable",
+ };
+}
+
+function parseText(text: string): ReadJsonResult {
+ try {
+ return { ok: true, value: JSON.parse(text) as unknown };
+ } catch {
+ return { ok: false, reason: "unparseable" };
+ }
+}
+
+// The file as JSON. Never throws: `absent` is a missing file (or a missing
+// parent), `unreadable` any other read error (EACCES, EISDIR, EIO, …),
+// `unparseable` a file that is not JSON — which includes a truncated one.
+export async function readJsonFile(file: string): Promise<ReadJsonResult> {
+ let text: string;
+ try {
+ text = await readFile(file, "utf8");
+ } catch (err) {
+ return readFailure(err);
+ }
+ return parseText(text);
+}
+
+// The synchronous twin, for the two readers that are synchronous by contract:
+// `getSettings` (lib/settings.ts) and `getSite` (lib/site.ts).
+export function readJsonFileSync(file: string): ReadJsonResult {
+ let text: string;
+ try {
+ text = fs.readFileSync(file, "utf8");
+ } catch (err) {
+ return readFailure(err);
+ }
+ return parseText(text);
+}
+
+export type WriteJsonOptions = {
+ // 2 (the default) = `JSON.stringify(value, null, 2)`; 0 = compact.
+ indent?: 0 | 2;
+ // Append "\n". Defaults to true — the common case; pass false to keep a
+ // file's historical bytes.
+ newline?: boolean;
+ // Create the parent directory first (`mkdir -p`).
+ mkdir?: boolean;
+};
+
+// Exactly the bytes a write puts on disk.
+export function jsonText(value: unknown, opts: WriteJsonOptions = {}): string {
+ const indent = opts.indent ?? 2;
+ const body =
+ indent === 0 ? JSON.stringify(value) : JSON.stringify(value, null, indent);
+ return opts.newline === false ? body : body + "\n";
+}
+
+type JsonFileState = {
+ tmpSeq: number;
+ chains: Map<string, Promise<void>>;
+ locks: Map<string, Promise<unknown>>;
+};
+
+declare global {
+ // eslint-disable-next-line no-var
+ var __yttJsonFile__: JsonFileState | undefined;
+}
+
+// Exported for the test that proves two module copies share it.
+export function jsonFileState(): JsonFileState {
+ if (!globalThis.__yttJsonFile__) {
+ globalThis.__yttJsonFile__ = { tmpSeq: 0, chains: new Map(), locks: new Map() };
+ }
+ return globalThis.__yttJsonFile__;
+}
+
+// Unique per write, not per process: `${file}.tmp-${pid}-${seq}-${random}`.
+export function tmpPathFor(file: string): string {
+ const seq = ++jsonFileState().tmpSeq;
+ return `${file}.tmp-${process.pid}-${seq}-${randomBytes(4).toString("hex")}`;
+}
+
+async function writeNow(
+ file: string,
+ text: string,
+ makeDir: boolean,
+): Promise<void> {
+ if (makeDir) await mkdir(path.dirname(file), { recursive: true });
+ const tmp = tmpPathFor(file);
+ try {
+ await writeFile(tmp, text);
+ await rename(tmp, file);
+ } catch (err) {
+ await rm(tmp, { force: true }).catch(() => {});
+ throw err;
+ }
+}
+
+// Write `value` as JSON to `file` atomically (tmp + rename), after every write
+// to the same absolute path this process issued earlier has settled. The value
+// is serialised NOW, at the call, so a caller that goes on mutating its object
+// cannot change what lands. A failed earlier write does not block a later one;
+// each caller sees only its own write's error.
+export function writeJsonAtomic(
+ file: string,
+ value: unknown,
+ opts: WriteJsonOptions = {},
+): Promise<void> {
+ const key = path.resolve(file);
+ const text = jsonText(value, opts);
+ const { chains } = jsonFileState();
+ const prev = chains.get(key) ?? Promise.resolve();
+ const next = prev.then(
+ () => writeNow(key, text, opts.mkdir === true),
+ () => writeNow(key, text, opts.mkdir === true),
+ );
+ chains.set(key, next);
+ const release = () => {
+ if (chains.get(key) === next) chains.delete(key);
+ };
+ next.then(release, release);
+ return next;
+}
+
+// The synchronous twin, for the three stores whose API is synchronous
+// (chartsStore, aliasesStore, curatedTagsStore). A synchronous write cannot
+// wait on the async chain, and needs no chain of its own: nothing else in this
+// process runs while it does. It shares the unique temp name.
+export function writeJsonAtomicSync(
+ file: string,
+ value: unknown,
+ opts: WriteJsonOptions = {},
+): void {
+ if (opts.mkdir === true) fs.mkdirSync(path.dirname(file), { recursive: true });
+ const tmp = tmpPathFor(file);
+ try {
+ fs.writeFileSync(tmp, jsonText(value, opts));
+ fs.renameSync(tmp, file);
+ } catch (err) {
+ fs.rmSync(tmp, { force: true });
+ throw err;
+ }
+}
+
+// Run `fn` with exclusive use of `file` among callers of this function in this
+// process: a READ-MODIFY-WRITE cycle that must not interleave with another one
+// on the same path (two `patchChannelConfig`s, say, each of which would
+// otherwise read the file before the other wrote it and so drop its patch).
+// Callers that only write need not take it — writes are chained anyway. Per
+// process, like the write chain.
+export function withJsonFileLock<T>(file: string, fn: () => Promise<T>): Promise<T> {
+ const key = path.resolve(file);
+ const { locks } = jsonFileState();
+ const prev = locks.get(key) ?? Promise.resolve();
+ const next = prev.then(fn, fn);
+ locks.set(key, next);
+ const release = () => {
+ if (locks.get(key) === next) locks.delete(key);
+ };
+ next.then(release, release);
+ return next;
+}
+
+// For tests: how many paths currently have a write in flight.
+export function pendingJsonWrites(): number {
+ return jsonFileState().chains.size;
+}
diff --git a/common/lib/posts-server.ts b/common/lib/posts-server.ts
@@ -11,8 +11,9 @@
// fetchers stop the same way), and uses the same "<extractor> <id>" line format
// so common/lib/archive.ts parses it unchanged.
+import { writeJsonAtomic } from "./jsonFile-server";
import path from "node:path";
-import { readdir, mkdir, readFile, rename, writeFile } from "node:fs/promises";
+import { readdir, mkdir, readFile } from "node:fs/promises";
import { appendFile } from "node:fs/promises";
import { readArchive } from "./archive";
import {
@@ -256,9 +257,7 @@ export async function writePostAvailability(
): Promise<void> {
const file = postsAvailabilityPath(channelRoot);
await mkdir(path.dirname(file), { recursive: true });
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(map, null, 2) + "\n");
- await rename(tmp, file);
+ await writeJsonAtomic(file, map);
}
// Fold new observations in, appending to history ONLY when the availability
@@ -339,7 +338,5 @@ export async function writePostFetchState(
): Promise<void> {
const file = postFetchStatePath(channelRoot);
await mkdir(path.dirname(file), { recursive: true });
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(state, null, 2) + "\n");
- await rename(tmp, file);
+ await writeJsonAtomic(file, state);
}
diff --git a/common/lib/savedVideo-server.ts b/common/lib/savedVideo-server.ts
@@ -1,3 +1,4 @@
+import { writeJsonAtomic } from "./jsonFile-server";
import path from "node:path";
import {
copyFile,
@@ -6,7 +7,6 @@ import {
rename,
rm,
stat,
- writeFile,
} from "node:fs/promises";
import {
SAVED_VIDEO_POINTER_FILENAME,
@@ -45,9 +45,7 @@ async function writePointer(
pointer: SavedVideoPointer,
): Promise<void> {
const file = savedVideoPointerPath(videoDir);
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(pointer, null, 2) + "\n");
- await rename(tmp, file);
+ await writeJsonAtomic(file, pointer);
}
// Repoint a video's saved-video.json at a new absolute store dir. Used when a
diff --git a/common/lib/settings.ts b/common/lib/settings.ts
@@ -22,7 +22,7 @@
// One schema, both directions: the only differences between what a read and a
// write produce are those migrations and those two validators.
-import fs from "node:fs";
+import { readJsonFileSync, writeJsonAtomic } from "./jsonFile-server";
import path from "node:path";
import { getPaths } from "./paths";
import {
@@ -56,11 +56,8 @@ type RawSettings = Record<string, unknown>;
// The file as JSON, or `undefined` when it is missing or not JSON. Never
// throws: a settings read is on every request path.
function readRawSettings(file: string): unknown {
- try {
- return JSON.parse(fs.readFileSync(file, "utf8"));
- } catch {
- return undefined;
- }
+ const read = readJsonFileSync(file);
+ return read.ok ? read.value : undefined;
}
// Only a plain object is a settings file. `null`, `[]`, `3` and a truncated
@@ -244,7 +241,5 @@ export async function writeSettings(next: SiteSettings): Promise<void> {
socialLinks: validatedSocialLinks(next.socialLinks),
});
const file = getPaths().settingsFile;
- const tmp = `${file}.tmp-${process.pid}`;
- await fs.promises.writeFile(tmp, JSON.stringify(merged, null, 2) + "\n");
- await fs.promises.rename(tmp, file);
+ await writeJsonAtomic(file, merged);
}
diff --git a/common/lib/settingsDocs.ts b/common/lib/settingsDocs.ts
@@ -64,13 +64,13 @@ export function renderSettingsExample(): string {
return JSON.stringify(d, null, 2) + "\n";
}
-function isScalar(v: unknown): boolean {
+export function isScalar(v: unknown): boolean {
return v === null || typeof v !== "object";
}
// The default as a table cell: a scalar inline, an empty container inline,
// anything larger by reference to its section.
-function defaultCell(v: unknown): string {
+export function defaultCell(v: unknown): string {
if (isScalar(v)) return "`" + JSON.stringify(v) + "`";
const json = JSON.stringify(v);
if (json === "[]" || json === "{}") return "`" + json + "`";
@@ -78,20 +78,20 @@ function defaultCell(v: unknown): string {
}
// A description inside a table cell: one line, pipes escaped, paragraphs kept.
-function cell(text: string): string {
+export function cell(text: string): string {
return text.replace(/\|/g, "\\|").replace(/\n\n/g, "<br><br>").replace(/\n/g, " ");
}
// One nested key table. `defaults(key)` answers the Default column; a table of
// per-entry fields (list items, map values, tree nodes) has no defaults — each
// entry spells its own — and says so.
-type KeyTable = {
+export type KeyTable = {
path: string;
docs: Readonly<Record<string, string>>;
defaults?: (key: string) => string;
};
-function fromObject(obj: unknown): (key: string) => string {
+export function fromObject(obj: unknown): (key: string) => string {
const r = (obj ?? {}) as Record<string, unknown>;
return (key) => (key in r ? defaultCell(r[key]) : "absent");
}
@@ -208,7 +208,8 @@ export function blockTables(d: SiteSettings): Partial<Record<keyof SiteSettings,
};
}
-function renderTable(out: string[], table: KeyTable): void {
+// Shared with lib/fileSchemaDocs.ts (SITE.md, CHANNEL.md).
+export function renderTable(out: string[], table: KeyTable): void {
out.push(`#### \`${table.path}\``);
out.push("");
if (table.defaults) {
diff --git a/common/lib/sidecar-server.test.ts b/common/lib/sidecar-server.test.ts
@@ -0,0 +1,165 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, readFile, writeFile, readdir } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { SIDECAR_FILENAMES, sidecar, sidecarField } from "./sidecar-server";
+import { SUB_FILE_RE } from "./videoStatus";
+// Every sidecar module, so the enumeration below sees every declaration.
+import "./attribution-server";
+import "./diarization-server";
+import "./digest-server";
+import {
+ availabilitySidecar,
+ loadAvailability,
+ writeAvailability,
+} from "./availability-server";
+import {
+ loadDownloadOutcome,
+ writeDownloadOutcome,
+} from "./downloadOutcome-server";
+import {
+ loadTranscribeOutcome,
+ writeTranscribeOutcome,
+} from "./transcribeOutcome-server";
+import {
+ isDoNotClean,
+ loadDoNotClean,
+ setDoNotClean,
+} from "./doNotClean-server";
+import {
+ isExcludedFromTruncatedCheck,
+ loadExcludeTruncatedCheck,
+ setExcludedFromTruncatedCheck,
+} from "./excludeTruncatedCheck-server";
+import type { AvailabilityRecord } from "./availability";
+import type { DownloadOutcomeRecord } from "./downloadOutcome";
+import type { TranscribeOutcomeRecord } from "./transcribeOutcome";
+
+async function scratch(): Promise<string> {
+ return mkdtemp(path.join(os.tmpdir(), "sidecar-"));
+}
+
+test("every declared sidecar filename escapes SUB_FILE_RE, and all nine are declared", () => {
+ assert.deepEqual([...SIDECAR_FILENAMES].sort(), [
+ "ai-digest.json",
+ "ai-digest.overrides.json",
+ "attribution.json",
+ "availability.json",
+ "diarization.json",
+ "do-not-clean.json",
+ "download-outcome.json",
+ "exclude-truncated-check.json",
+ "transcribe-outcome.json",
+ ]);
+ for (const name of SIDECAR_FILENAMES) {
+ assert.ok(!SUB_FILE_RE.test(name), name);
+ }
+});
+
+test("sidecar() refuses a transcript.<x>.<y> name at declaration", () => {
+ assert.throws(
+ () => sidecar("transcript.speakers.json", sidecarField((v) => v)),
+ /SUB_FILE_RE/,
+ );
+ assert.throws(() => sidecar("a/b.json", sidecarField((v) => v)), /bare filename/);
+});
+
+test("absent, unparseable and coercion-rejected all load as null", async () => {
+ const dir = await scratch();
+ assert.equal(await loadAvailability(dir), null);
+ await writeFile(availabilitySidecar.path(dir), "{ not json");
+ assert.equal(await loadAvailability(dir), null);
+ await writeFile(availabilitySidecar.path(dir), JSON.stringify({ availability: "nope", checkedAt: "x" }));
+ assert.equal(await loadAvailability(dir), null);
+ await writeFile(availabilitySidecar.path(dir), "null");
+ assert.equal(await loadAvailability(dir), null);
+});
+
+test("a coercion that throws reads as null, not as an exception", async () => {
+ const dir = await scratch();
+ const s = sidecar(
+ "throwing-test.json",
+ sidecarField((): null => {
+ throw new Error("boom");
+ }),
+ );
+ await writeFile(s.path(dir), "{}");
+ assert.equal(await s.load(dir), null);
+});
+
+test("doNotClean / excludeTruncatedCheck: a malformed marker still reads as present", async () => {
+ const dir = await scratch();
+ for (const [name, load, is] of [
+ ["do-not-clean.json", loadDoNotClean, isDoNotClean],
+ ["exclude-truncated-check.json", loadExcludeTruncatedCheck, isExcludedFromTruncatedCheck],
+ ] as const) {
+ assert.equal(await load(dir), null);
+ assert.equal(await is(dir), false);
+ await writeFile(path.join(dir, name), JSON.stringify({ nope: 1 }));
+ assert.deepEqual(await load(dir), { setAt: "" });
+ await writeFile(path.join(dir, name), "{ torn");
+ assert.equal(await load(dir), null);
+ }
+});
+
+test("setDoNotClean / setExcludedFromTruncatedCheck: write, load, remove", async () => {
+ const dir = await scratch();
+ await setDoNotClean(dir, true, "kept");
+ const dnc = await loadDoNotClean(dir);
+ assert.equal(dnc?.note, "kept");
+ assert.ok(dnc?.setAt);
+ await setDoNotClean(dir, false);
+ assert.equal(await loadDoNotClean(dir), null);
+
+ await setExcludedFromTruncatedCheck(dir, true);
+ const etc = await loadExcludeTruncatedCheck(dir);
+ assert.ok(etc?.setAt);
+ assert.equal("note" in (etc ?? {}), false);
+ await setExcludedFromTruncatedCheck(dir, false);
+ assert.equal(await loadExcludeTruncatedCheck(dir), null);
+ assert.deepEqual(await readdir(dir), []);
+});
+
+test("availability / download-outcome / transcribe-outcome round-trip, indented with a newline", async () => {
+ const dir = await scratch();
+ const avail: AvailabilityRecord = {
+ checkedAt: "2026-01-01T00:00:00.000Z",
+ availability: "public",
+ history: [{ availability: "public", observedAt: "2026-01-01T00:00:00.000Z", source: "check" }],
+ } as AvailabilityRecord;
+ await writeAvailability(dir, avail);
+ assert.deepEqual(await loadAvailability(dir), avail);
+ assert.equal(
+ await readFile(path.join(dir, "availability.json"), "utf8"),
+ JSON.stringify(avail, null, 2) + "\n",
+ );
+
+ const dlo = {
+ status: "ok",
+ videoId: "v1",
+ startedAt: "2026-01-01T00:00:00.000Z",
+ finishedAt: "2026-01-01T00:01:00.000Z",
+ attempts: [],
+ } as unknown as DownloadOutcomeRecord;
+ await writeDownloadOutcome(dir, dlo);
+ assert.deepEqual(await loadDownloadOutcome(dir), dlo);
+
+ const tro = {
+ videoId: "v1",
+ transcribedAt: "2026-01-01T00:00:00.000Z",
+ } as unknown as TranscribeOutcomeRecord;
+ await writeTranscribeOutcome(dir, tro);
+ assert.deepEqual(await loadTranscribeOutcome(dir), tro);
+});
+
+test("a load keeps keys the shape check does not name (no strip on a sidecar)", async () => {
+ const dir = await scratch();
+ const raw = {
+ videoId: "v1",
+ transcribedAt: "2026-01-01T00:00:00.000Z",
+ extra: { kept: true },
+ };
+ await writeFile(path.join(dir, "transcribe-outcome.json"), JSON.stringify(raw));
+ assert.deepEqual(await loadTranscribeOutcome(dir), raw);
+});
diff --git a/common/lib/sidecar-server.ts b/common/lib/sidecar-server.ts
@@ -0,0 +1,104 @@
+// ONE DECLARATION PER PER-VIDEO SIDECAR: its filename, its read, its write.
+//
+// one-core phase 3 slice 4b. Each `<video dir>/<name>.json` sidecar used to be
+// a hand-written `X-server.ts` with the same four moves — join the path, read +
+// JSON.parse + a shape check inside a try that turns anything wrong into null,
+// tmp + rename on write, `rm -f` to clear — copied per file. `sidecar()` is those
+// four moves once; an `X-server.ts` now declares its sidecar and keeps only what
+// is genuinely its own (a merge, a history append, a derived predicate).
+//
+// const s = sidecar(ATTRIBUTION_FILENAME, sidecarField(coerceAttribution), { indent: 0 });
+//
+// THE NAMING GUARD RUNS AT DECLARATION. `SUB_FILE_RE` (lib/videoStatus.ts)
+// claims any `transcript.<x>.<y>` file in a video dir as a subtitle track; a
+// sidecar so named would be indexed as a language. `sidecar()` throws when its
+// filename matches, and every declaration runs at module load — so a misnamed
+// sidecar fails every test that imports it, and the editor's boot, rather than
+// quietly becoming a `<x>`-language track. `SIDECAR_FILENAMES` lists every
+// declared name for the enumeration test.
+//
+// READS NEVER THROW. Absent, unreadable, unparseable, or rejected by the
+// coercion — all read as null, exactly the rule every sidecar reader had: one
+// corrupt sidecar must never fail a channel-wide sweep. A coercion MAY decide a
+// parseable-but-malformed file is not null (doNotClean: "any marker at all means
+// protected") — that decision is the coercion's, not this module's.
+//
+// NO FRESHNESS CHECK. The plan named an mtime freshness field; no reader
+// consumes one, so none was built (slice 4b record, deviation 1).
+//
+// SERVER-ONLY (zod, node:fs).
+
+import path from "node:path";
+import { rm } from "node:fs/promises";
+import { z } from "zod";
+import { SUB_FILE_RE } from "./videoStatus";
+import {
+ readJsonFile,
+ writeJsonAtomic,
+ type WriteJsonOptions,
+} from "./jsonFile-server";
+
+// A sidecar's schema: any parsed JSON in, the record or null out.
+export type SidecarSchema<T> = z.ZodType<T | null, unknown>;
+
+// A sidecar FIELD: the existing shape check as a zod schema. It is
+// settingsSchema.ts's `settingsField` minus the `.catch`: a settings coercion
+// falls back to a default, a sidecar coercion falls back to null ("absent").
+export function sidecarField<T>(
+ coerce: (value: unknown) => T | null,
+): SidecarSchema<T> {
+ return z.unknown().transform((value): T | null => coerce(value));
+}
+
+export type Sidecar<T> = {
+ filename: string;
+ path: (videoDir: string) => string;
+ load: (videoDir: string) => Promise<T | null>;
+ write: (videoDir: string, value: T) => Promise<void>;
+ remove: (videoDir: string) => Promise<void>;
+};
+
+const declared: string[] = [];
+
+// Every filename declared through `sidecar()` in this process, in declaration
+// order. Populated as each `X-server.ts` loads.
+export const SIDECAR_FILENAMES: readonly string[] = declared;
+
+export function sidecar<T>(
+ filename: string,
+ schema: SidecarSchema<T>,
+ opts: Pick<WriteJsonOptions, "indent" | "newline"> = {},
+): Sidecar<T> {
+ if (SUB_FILE_RE.test(filename)) {
+ throw new Error(
+ `Sidecar "${filename}" matches SUB_FILE_RE (transcript.<x>.<y>) and would be read as a subtitle track — rename it`,
+ );
+ }
+ if (filename.includes("/") || filename.includes(path.sep)) {
+ throw new Error(`Sidecar "${filename}" must be a bare filename`);
+ }
+ if (!declared.includes(filename)) declared.push(filename);
+ const at = (videoDir: string) => path.join(videoDir, filename);
+ return {
+ filename,
+ path: at,
+ async load(videoDir) {
+ const read = await readJsonFile(at(videoDir));
+ if (!read.ok) return null;
+ try {
+ const parsed = schema.safeParse(read.value);
+ return parsed.success ? parsed.data : null;
+ } catch {
+ // A coercion that throws on some input is a bug in the coercion; it
+ // still must not fail a sweep.
+ return null;
+ }
+ },
+ async write(videoDir, value) {
+ await writeJsonAtomic(at(videoDir), value, opts);
+ },
+ async remove(videoDir) {
+ await rm(at(videoDir), { force: true });
+ },
+ };
+}
diff --git a/common/lib/site.ts b/common/lib/site.ts
@@ -1,16 +1,9 @@
import fs from "node:fs";
import path from "node:path";
-import {
- DEFAULT_GROUP_FALLBACK_ID,
- FALLBACK_GROUP,
- parseChannelGroups,
- resolveDefaultGroupId,
- type ChannelGroup,
-} from "./channelGroups";
+import { parseChannelGroups, resolveDefaultGroupId } from "./channelGroups";
import { getPaths, type Paths } from "./paths";
import { TAGS_FILENAME } from "./curatedTags";
import type { SiteChannelIndex } from "./channelPriority";
-import { parseAccent } from "./accent";
import {
getSettings,
normalizeSocialSvg,
@@ -18,6 +11,14 @@ import {
type SiteSettings,
type SocialLink,
} from "./settings";
+import { readJsonFileSync, writeJsonAtomic } from "./jsonFile-server";
+import {
+ isValidSiteId,
+ parseSite,
+ parseSiteUrl,
+ siteToDisk,
+ type Site,
+} from "./siteSchema";
// A Site is a selection + presentation layer over the single global channel
// pool. Branding, social links, the channel grouping layout AND which channels
@@ -25,95 +26,11 @@ import {
// can power several public sites that overlap on channels without duplicating
// any downloads. Operational config (transcribe bin, cookies, rate limits)
// stays global in settings.json — see common/lib/settings.ts.
-
-export type SiteChannelMembership = {
- // Channel slug (directory name under transcripts/channels/).
- slug: string;
- // Group this channel belongs to WITHIN this site. Resolved against
- // site.groups at index-build time; the same channel can sit in different
- // groups on different sites. Falls back to site.defaultGroupId when unknown.
- groupId?: string;
- // Optional explicit ordering hint within the site (lower first).
- order?: number;
-};
-
-export type Site = {
- siteId: string;
- siteTitle: string;
- siteDescription: string;
- headerTitle: string;
- homeTagline: string;
- // Optional per-site brand accent ("#rrggbb"). Overrides the family brass on
- // this site's public build (see siteAccentVars / export layout). Undefined =
- // inherit the family brass.
- accent?: string;
- // Per-site social links. `undefined` (no `socialLinks` key in site.json)
- // means inherit the global default from SiteSettings.socialLinks; an array
- // (even empty) overrides it. Resolve with resolveSocialLinks() at render.
- socialLinks?: SocialLink[];
- // Channel grouping layout for THIS site (cosmetic + default-selection
- // buckets the export UI renders). Mirrors what used to live in
- // SiteSettings.groups but is now per-site.
- groups: ChannelGroup[];
- defaultGroupId: string;
- // The channels this site exposes. A channel absent from this list is not
- // built or deployed for this site even though its data exists in the pool.
- channels: SiteChannelMembership[];
- // Cloudflare Pages project name this site deploys to
- // (`wrangler pages deploy out --project-name <cloudflareProject>`).
- cloudflareProject?: string;
- // Absolute public URL of this site's deployment, e.g. "https://jeralyzer.com".
- // Drives the cross-site footer list (see resolveRelatedSites). A site with no
- // siteUrl is omitted from every other site's list — there's no link target.
- siteUrl?: string;
- // Per-site override that pulls specific siblings to the front of the footer's
- // cross-site list, in named groups. Siblings not named here fall into a
- // trailing auto "Other sites" group. Absent/empty = one flat list of every
- // sibling. siteIds are resolved against the live pool at render time, so an
- // id for a site that doesn't exist (yet) is harmless — it's just skipped.
- relatedSites?: RelatedSiteGroup[];
- // Whether this site ships an installable PWA (service worker + web manifest).
- // Default (undefined/false) = a "dumb instance": it serves the CORS-enabled
- // JSON federation contract but is not independently installable, so a visitor
- // trusts only the hub PWA. Set true to make this site its own installable PWA.
- // See export/app/lib/mode.ts (shipsPwa) and common/lib/siteDescriptor.ts.
- pwa?: boolean;
- // Whether the site build generates downloadable transcript/live-chat archive
- // zips into public/archives (and links them on the Downloads page). This is
- // an opt-OUT: undefined/true = on, only explicit `false` disables. Also gated
- // by the global setting and a per-build flag (see compose-site.ts). Default-on
- // because bulk download is the point of publishing a corpus.
- archives?: boolean;
- // Per-site override for the served-file size cap (bytes). Any archive larger
- // than this is dropped from what's served and flagged in the manifest so a
- // capped host (Cloudflare Pages: 25 MB) won't reject the deploy. 0 = no cap.
- // Absent = the global DEFAULT_ARCHIVE_MAX_BYTES / MAX_ARCHIVE_BYTES env.
- archiveMaxBytes?: number;
- // Whether this site publishes the Duplicates page (and its Header nav link).
- // Opt-OUT: undefined/true = on, only explicit `false` hides it. Even when on,
- // the link/page auto-hide when the site has no in-scope duplicate clusters (the
- // build simply writes no duplicates.json — see compose-site.ts / hasDuplicates).
- duplicates?: boolean;
- // Per-site override for the hub this site belongs under (the PWA it points
- // visitors toward). Absent = inherit the family default SiteSettings.homepageUrl.
- // Resolve with resolveHubUrl(). Surfaced on /site.json so a hub can tell member
- // sites (that name it) from arbitrary added origins.
- hubUrl?: string;
-};
-
-export type RelatedSiteGroup = {
- // Optional muted heading shown above the group; omit for an unlabeled group.
- label?: string;
- // Sibling site ids, in display order.
- siteIds: string[];
-};
-
-// siteId shares the group-id grammar: lowercase slug, used as a directory name.
-export const SITE_ID_RE = /^[a-z0-9][a-z0-9-]*$/;
-
-export function isValidSiteId(id: unknown): id is string {
- return typeof id === "string" && SITE_ID_RE.test(id);
-}
+//
+// The file's SHAPE — every key, its default, its coercion, its documentation —
+// is lib/siteSchema.ts (one-core phase 3 slice 4b), re-exported here in full so
+// no importer moved. What is left in this file is I/O and the resolvers.
+export * from "./siteSchema";
export function siteDir(paths: Paths, siteId: string): string {
return path.join(paths.sitesDir, siteId);
@@ -168,138 +85,6 @@ export function siteStatsDir(paths: Paths, siteId: string): string {
return path.join(siteIndexDir(paths, siteId), "stats");
}
-function defaults(siteId: string): Site {
- return {
- siteId,
- siteTitle: "Transcript Browser",
- siteDescription: "Browse and search video transcripts",
- headerTitle: "Transcript Browser",
- homeTagline: "",
- // socialLinks left undefined = inherit the global default.
- // A single default group selected by default keeps a site with no explicit
- // group config behaving like the pre-grouping UI (one bucket, all checked).
- groups: [{ ...FALLBACK_GROUP }],
- defaultGroupId: DEFAULT_GROUP_FALLBACK_ID,
- channels: [],
- };
-}
-
-export function parseSiteChannels(input: unknown): SiteChannelMembership[] {
- if (!Array.isArray(input)) return [];
- const out: SiteChannelMembership[] = [];
- const seen = new Set<string>();
- for (const raw of input) {
- if (!raw || typeof raw !== "object") continue;
- const r = raw as Record<string, unknown>;
- const slug = typeof r.slug === "string" ? r.slug.trim() : "";
- if (!slug || seen.has(slug)) continue;
- seen.add(slug);
- const entry: SiteChannelMembership = { slug };
- if (typeof r.groupId === "string" && r.groupId.trim()) {
- entry.groupId = r.groupId.trim();
- }
- if (typeof r.order === "number" && Number.isFinite(r.order)) {
- entry.order = Math.floor(r.order);
- }
- out.push(entry);
- }
- return out;
-}
-
-// Normalize a raw siteUrl into a trimmed absolute http(s) URL with no trailing
-// slash, or undefined when missing/not a usable absolute URL. Relative or
-// scheme-less values are rejected — a cross-site link must be absolute.
-export function parseSiteUrl(input: unknown): string | undefined {
- if (typeof input !== "string") return undefined;
- const trimmed = input.trim().replace(/\/+$/, "");
- if (!/^https?:\/\/\S+/i.test(trimmed)) return undefined;
- return trimmed;
-}
-
-// Parse the relatedSites override: an ordered list of { label?, siteIds[] }
-// groups. Keeps only valid site ids, dedupes within a group, drops groups with
-// no valid ids, and preserves order. Existence against the live pool is NOT
-// checked here — resolveRelatedSites does that at render time.
-export function parseRelatedSites(input: unknown): RelatedSiteGroup[] {
- if (!Array.isArray(input)) return [];
- const out: RelatedSiteGroup[] = [];
- for (const raw of input) {
- if (!raw || typeof raw !== "object") continue;
- const r = raw as Record<string, unknown>;
- const ids = Array.isArray(r.siteIds) ? r.siteIds : [];
- const siteIds: string[] = [];
- const seen = new Set<string>();
- for (const id of ids) {
- if (!isValidSiteId(id) || seen.has(id)) continue;
- seen.add(id);
- siteIds.push(id);
- }
- if (siteIds.length === 0) continue;
- const label =
- typeof r.label === "string" && r.label.trim() ? r.label.trim() : undefined;
- out.push(label ? { label, siteIds } : { siteIds });
- }
- return out;
-}
-
-// Parse a raw site.json object into a fully-resolved Site, applying defaults
-// and validating groups/defaultGroupId/channel membership. Mirrors getSettings.
-export function parseSite(siteId: string, raw: unknown): Site {
- const base = defaults(siteId);
- const r =
- raw && typeof raw === "object" ? (raw as Record<string, unknown>) : {};
- const str = (key: keyof Site, fallback: string): string =>
- typeof r[key] === "string" ? (r[key] as string) : fallback;
-
- const groups = parseChannelGroups(r.groups);
- const resolvedGroups = groups.length > 0 ? groups : [{ ...FALLBACK_GROUP }];
- const defaultGroupId = resolveDefaultGroupId(r.defaultGroupId, resolvedGroups);
-
- // Drop memberships whose groupId is unknown to a concrete group (fold to
- // default at render time via resolveChannelGroupId, but keep the explicit
- // value when valid so the editor round-trips it).
- const channels = parseSiteChannels(r.channels).map((c) => {
- if (c.groupId && !resolvedGroups.some((g) => g.id === c.groupId)) {
- const { groupId: _drop, ...rest } = c;
- return rest;
- }
- return c;
- });
-
- return {
- siteId,
- siteTitle: str("siteTitle", base.siteTitle),
- siteDescription: str("siteDescription", base.siteDescription),
- headerTitle: str("headerTitle", base.headerTitle),
- homeTagline: str("homeTagline", base.homeTagline),
- // Key present (array) = override; absent = inherit the global default.
- socialLinks: Array.isArray(r.socialLinks)
- ? parseSocialLinks(r.socialLinks)
- : undefined,
- groups: resolvedGroups,
- defaultGroupId,
- channels,
- cloudflareProject:
- typeof r.cloudflareProject === "string" && r.cloudflareProject.trim()
- ? r.cloudflareProject.trim()
- : undefined,
- accent: parseAccent(r.accent),
- siteUrl: parseSiteUrl(r.siteUrl),
- relatedSites: parseRelatedSites(r.relatedSites),
- pwa: r.pwa === true,
- // Opt-out: only an explicit false disables. Absent/true stays on.
- archives: r.archives !== false,
- duplicates: r.duplicates !== false,
- archiveMaxBytes:
- typeof r.archiveMaxBytes === "number" &&
- Number.isFinite(r.archiveMaxBytes) &&
- r.archiveMaxBytes >= 0
- ? Math.floor(r.archiveMaxBytes)
- : undefined,
- hubUrl: parseSiteUrl(r.hubUrl),
- };
-}
-
// The hub URL this site points visitors toward: its own override, else the
// family default (SiteSettings.homepageUrl). Undefined when neither is set.
export function resolveHubUrl(
@@ -390,13 +175,10 @@ export function getSite(siteId: string, paths: Paths = getPaths()): Site {
if (!isValidSiteId(siteId)) {
throw new Error(`Invalid site id: ${String(siteId)}`);
}
- let parsed: unknown = {};
- try {
- parsed = JSON.parse(fs.readFileSync(siteConfigFile(paths, siteId), "utf8"));
- } catch {
- parsed = {};
- }
- return parseSite(siteId, parsed);
+ // Missing, unreadable or not JSON reads as the empty file: every field its
+ // default. A read never throws past the id check.
+ const read = readJsonFileSync(siteConfigFile(paths, siteId));
+ return parseSite(siteId, read.ok ? read.value : {});
}
// List the ids of all configured sites (directories under sitesDir that hold a
@@ -459,44 +241,11 @@ export async function writeSite(
socialLinks.push({ ...link, svg });
}
}
- const channels = parseSiteChannels(site.channels).filter(
- (c) => !c.groupId || groups.some((g) => g.id === c.groupId),
+ await writeJsonAtomic(
+ siteConfigFile(paths, site.siteId),
+ siteToDisk({ ...site, groups, defaultGroupId, socialLinks }),
+ { mkdir: true },
);
- const merged: Site = {
- siteId: site.siteId,
- siteTitle: site.siteTitle,
- siteDescription: site.siteDescription,
- headerTitle: site.headerTitle,
- homeTagline: site.homeTagline,
- ...(socialLinks !== undefined ? { socialLinks } : {}),
- groups,
- defaultGroupId,
- channels,
- ...(site.cloudflareProject && site.cloudflareProject.trim()
- ? { cloudflareProject: site.cloudflareProject.trim() }
- : {}),
- ...(parseAccent(site.accent) ? { accent: parseAccent(site.accent) } : {}),
- ...(parseSiteUrl(site.siteUrl) ? { siteUrl: parseSiteUrl(site.siteUrl) } : {}),
- ...(parseRelatedSites(site.relatedSites).length > 0
- ? { relatedSites: parseRelatedSites(site.relatedSites) }
- : {}),
- ...(site.pwa ? { pwa: true } : {}),
- // Persist only the non-default: archives is on unless explicitly disabled.
- ...(site.archives === false ? { archives: false } : {}),
- ...(site.duplicates === false ? { duplicates: false } : {}),
- ...(typeof site.archiveMaxBytes === "number" &&
- Number.isFinite(site.archiveMaxBytes) &&
- site.archiveMaxBytes >= 0
- ? { archiveMaxBytes: Math.floor(site.archiveMaxBytes) }
- : {}),
- ...(parseSiteUrl(site.hubUrl) ? { hubUrl: parseSiteUrl(site.hubUrl) } : {}),
- };
- const dir = siteDir(paths, site.siteId);
- await fs.promises.mkdir(dir, { recursive: true });
- const file = siteConfigFile(paths, site.siteId);
- const tmp = `${file}.tmp-${process.pid}`;
- await fs.promises.writeFile(tmp, JSON.stringify(merged, null, 2) + "\n");
- await fs.promises.rename(tmp, file);
}
export async function deleteSite(
diff --git a/common/lib/siteSchema.test.ts b/common/lib/siteSchema.test.ts
@@ -0,0 +1,229 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, readFile } from "node:fs/promises";
+import fs from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import type { z } from "zod";
+import {
+ SITE_FIELD_DOCS,
+ SITE_KEYS,
+ parseSite,
+ siteFieldsSchema,
+ siteToDisk,
+ type Site,
+} from "./siteSchema";
+import { getSite, siteConfigFile, writeSite } from "./site";
+import type { Paths } from "./paths";
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+
+// The schema's per-key output and the hand-written `Site` name the same keys,
+// and the output is assignable to `Site` (defaultGroupId aside: it is resolved
+// by the object step). The reverse direction does not hold for the optional
+// keys — zod emits them as required `T | undefined` — which is slice 4a's
+// deviation 2 again.
+type Fields = z.output<typeof siteFieldsSchema>;
+type SameKeys<A, B> = [keyof A] extends [keyof B]
+ ? [keyof B] extends [keyof A]
+ ? true
+ : false
+ : false;
+const keysMatch: SameKeys<Fields, Site> = true;
+const outputFits: Omit<Fields, "defaultGroupId"> extends Omit<Site, "defaultGroupId">
+ ? true
+ : false = true;
+
+const SVG =
+ '<svg viewBox="0 0 24 24"><path d="M12 2a10 10 0 0 0-3 19.5"/></svg>';
+
+test("the shape pins: schema keys = Site keys = SITE_FIELD_DOCS keys", () => {
+ assert.equal(keysMatch, true);
+ assert.equal(outputFits, true);
+ assert.deepEqual(
+ Object.keys(siteFieldsSchema.shape),
+ Object.keys(SITE_FIELD_DOCS),
+ );
+ assert.deepEqual([...SITE_KEYS], Object.keys(SITE_FIELD_DOCS));
+});
+
+test("empty, null, [] and a number all read as the defaults, every key emitted", () => {
+ const want = parseSite("s", {});
+ assert.deepEqual(Object.keys(want), Object.keys(SITE_FIELD_DOCS));
+ assert.equal(want.siteId, "s");
+ assert.equal(want.siteTitle, "Transcript Browser");
+ assert.equal(want.groups.length, 1);
+ assert.equal(want.groups[0].id, "default");
+ assert.equal(want.defaultGroupId, "default");
+ assert.equal(want.archives, true);
+ assert.equal(want.duplicates, true);
+ assert.equal(want.pwa, false);
+ assert.deepEqual(want.relatedSites, []);
+ assert.equal(want.socialLinks, undefined);
+ assert.ok("socialLinks" in want);
+ for (const raw of [null, [], 3, "x", undefined]) {
+ assert.deepEqual(parseSite("s", raw), want, JSON.stringify(raw));
+ }
+});
+
+test("the siteId comes from the caller, never the file", () => {
+ assert.equal(parseSite("real", { siteId: "other" }).siteId, "real");
+});
+
+test("unknown keys are dropped", () => {
+ const site = parseSite("s", { bogus: 1, siteTitle: "T" }) as Record<string, unknown>;
+ assert.equal("bogus" in site, false);
+ assert.equal(site.siteTitle, "T");
+});
+
+test("a membership naming an unknown group loses its groupId; a known one keeps it", () => {
+ const site = parseSite("s", {
+ groups: [
+ { id: "a", name: "A", selectedByDefault: true },
+ { id: "b", name: "B" },
+ ],
+ defaultGroupId: "zzz",
+ channels: [
+ { slug: "one", groupId: "b" },
+ { slug: "two", groupId: "nope", order: 2.7 },
+ { slug: "one" },
+ { slug: " " },
+ ],
+ });
+ assert.equal(site.defaultGroupId, "a");
+ assert.deepEqual(site.channels, [
+ { slug: "one", groupId: "b" },
+ { slug: "two", order: 2 },
+ ]);
+});
+
+test("socialLinks: absent inherits (undefined), [] overrides", () => {
+ assert.equal(parseSite("s", {}).socialLinks, undefined);
+ assert.deepEqual(parseSite("s", { socialLinks: [] }).socialLinks, []);
+ assert.equal(parseSite("s", { socialLinks: "x" }).socialLinks, undefined);
+});
+
+test("archives / duplicates are off only when explicitly false", () => {
+ for (const v of [undefined, true, 0, "false", null]) {
+ assert.equal(parseSite("s", { archives: v }).archives, true, String(v));
+ assert.equal(parseSite("s", { duplicates: v }).duplicates, true, String(v));
+ }
+ assert.equal(parseSite("s", { archives: false }).archives, false);
+ assert.equal(parseSite("s", { duplicates: false }).duplicates, false);
+});
+
+test("siteToDisk persists only the non-defaults", () => {
+ const disk = siteToDisk(parseSite("s", {})) as Record<string, unknown>;
+ assert.deepEqual(Object.keys(disk), [
+ "siteId",
+ "siteTitle",
+ "siteDescription",
+ "headerTitle",
+ "homeTagline",
+ "groups",
+ "defaultGroupId",
+ "channels",
+ ]);
+ const full = siteToDisk(
+ parseSite("s", {
+ archives: false,
+ duplicates: false,
+ pwa: true,
+ archiveMaxBytes: 1024.9,
+ siteUrl: "https://a.example/",
+ hubUrl: "http://hub.example",
+ accent: "#ABCDEF",
+ cloudflareProject: " proj ",
+ relatedSites: [{ label: "x", siteIds: ["b"] }],
+ socialLinks: [],
+ }),
+ );
+ assert.equal(full.archives, false);
+ assert.equal(full.duplicates, false);
+ assert.equal(full.pwa, true);
+ assert.equal(full.archiveMaxBytes, 1024);
+ assert.equal(full.siteUrl, "https://a.example");
+ assert.equal(full.accent, "#abcdef");
+ assert.equal(full.cloudflareProject, "proj");
+ assert.deepEqual(full.socialLinks, []);
+});
+
+function fixtures(): Array<[string, unknown]> {
+ const out: Array<[string, unknown]> = [
+ ["empty", {}],
+ [
+ "everything",
+ {
+ siteId: "s",
+ siteTitle: "T",
+ siteDescription: "D",
+ headerTitle: "H",
+ homeTagline: "tag",
+ accent: "#123456",
+ socialLinks: [{ label: "L", url: "https://x.example", svg: SVG }],
+ groups: [
+ { id: "a", name: " A ", selectedByDefault: true, order: 1.5, inline: true },
+ { id: "b", name: "B", description: " d " },
+ { id: "a", name: "dup" },
+ { id: "BAD", name: "x" },
+ ],
+ defaultGroupId: "b",
+ channels: [{ slug: "c1", groupId: "a", order: 3 }, { slug: "c2", groupId: "q" }],
+ cloudflareProject: "p",
+ siteUrl: "https://s.example//",
+ relatedSites: [{ siteIds: ["x", "x", "BAD"] }, { label: " ", siteIds: [] }],
+ pwa: true,
+ archives: false,
+ archiveMaxBytes: -1,
+ duplicates: false,
+ hubUrl: "ftp://nope",
+ bogus: true,
+ },
+ ],
+ ];
+ const e2e = path.join(HERE, "..", "..", "editor", "e2e", "fixtures", "sites", "testsite", "site.json");
+ out.push(["e2e testsite", JSON.parse(fs.readFileSync(e2e, "utf8"))]);
+ return out;
+}
+
+test("round-trip: parseSite(siteToDisk(parseSite(x))) = parseSite(x)", () => {
+ for (const [name, raw] of fixtures()) {
+ const once = parseSite("s", raw);
+ const twice = parseSite("s", JSON.parse(JSON.stringify(siteToDisk(once))));
+ assert.deepEqual(twice, once, name);
+ }
+});
+
+function scratchPaths(dir: string): Paths {
+ return { sitesDir: dir } as Paths;
+}
+
+test("writeSite throws on no groups, a default outside the groups, and an unsafe SVG", async () => {
+ const paths = scratchPaths(await mkdtemp(path.join(os.tmpdir(), "site-")));
+ const base = parseSite("s", {});
+ await assert.rejects(writeSite({ ...base, groups: [] }, paths), /At least one channel group/);
+ await assert.rejects(
+ writeSite({ ...base, defaultGroupId: "elsewhere" }, paths),
+ /not in the configured groups/,
+ );
+ await assert.rejects(
+ writeSite(
+ { ...base, socialLinks: [{ label: "L", url: "https://x.example", svg: "<svg><script/></svg>" }] },
+ paths,
+ ),
+ /invalid SVG/,
+ );
+ assert.equal(fs.existsSync(siteConfigFile(paths, "s")), false);
+});
+
+test("writeSite → getSite round-trips, and the file holds only non-defaults", async () => {
+ const paths = scratchPaths(await mkdtemp(path.join(os.tmpdir(), "site-")));
+ const site = parseSite("s", { siteTitle: "Mine", archives: false });
+ await writeSite(site, paths);
+ assert.deepEqual(getSite("s", paths), site);
+ const disk = JSON.parse(await readFile(siteConfigFile(paths, "s"), "utf8"));
+ assert.equal(disk.archives, false);
+ assert.equal("duplicates" in disk, false);
+ assert.equal("pwa" in disk, false);
+});
diff --git a/common/lib/siteSchema.ts b/common/lib/siteSchema.ts
@@ -0,0 +1,326 @@
+// THE site.json SCHEMA — one definition of `sites/<id>/site.json`, used by the
+// reader, the writer and SITE.md.
+//
+// one-core phase 3 slice 4b, on slice 4a's pattern (lib/settingsSchema.ts):
+// every field is `settingsField(coerce)` over the parser that already existed —
+// parseChannelGroups, resolveDefaultGroupId, parseSiteChannels,
+// parseSocialLinks, parseAccent, parseSiteUrl, parseRelatedSites — so no
+// boundary moved; zod supplies the key plumbing and strips unknown keys. No
+// `.default()`, no `.passthrough()`.
+//
+// WHAT A READ PROMISES (parseSite): it never throws, and it emits EVERY key —
+// an absent optional key is present with the value `undefined` — exactly as the
+// hand-written parseSite did. Two fields depend on a sibling (the default group
+// must be one of the groups; a membership's group must exist), so the schema is
+// the per-key object followed by ONE object-level step that resolves them.
+//
+// WHAT A WRITE PROMISES (writeSite in lib/site.ts): the throwing validations,
+// then `siteToDisk` below — which persists only what is not a default — through
+// the atomic writer. `siteToDisk(parseSite(x))` parses back to `parseSite(x)`
+// (siteSchema.test.ts pins that over fixtures).
+//
+// The types moved here from lib/site.ts with their parsers (lib/site.ts
+// re-exports all of it, so no importer changed); each field's documentation is
+// its `*_FIELD_DOCS` entry, type-checked complete (lib/fieldDocs.ts) and
+// rendered into SITE.md by common/bin/file-schemas-docs.ts.
+//
+// SERVER-ONLY: zod. The one client importer of these types, SiteForm.tsx,
+// imports them with `import type`. The PUBLIC `export/public/site.json`
+// (lib/siteDescriptor.ts) is a different file with a different schema.
+
+import { z } from "zod";
+import {
+ FALLBACK_GROUP,
+ parseChannelGroups,
+ resolveDefaultGroupId,
+ type ChannelGroup,
+} from "./channelGroups";
+import { parseAccent } from "./accent";
+import { parseSocialLinks, type SocialLink } from "./settingsSchema";
+import { settingsField } from "./settingsFieldSchemas";
+import type { FieldDocs } from "./fieldDocs";
+
+// Each field is documented in SITE_CHANNEL_MEMBERSHIP_FIELD_DOCS below.
+export type SiteChannelMembership = {
+ slug: string;
+ groupId?: string;
+ order?: number;
+};
+
+export const SITE_CHANNEL_MEMBERSHIP_FIELD_DOCS: FieldDocs<SiteChannelMembership> = {
+ slug: "Channel slug (its directory name under `transcripts/channels/`). Blank and duplicate slugs are dropped.",
+ groupId:
+ "Group this channel belongs to WITHIN this site. The same channel can sit in different groups on different sites. A value naming no configured group is dropped on read and falls back to `defaultGroupId` at render time.",
+ order: "Optional explicit ordering hint within the site (lower first); floored to an integer.",
+};
+
+// Each field is documented in RELATED_SITE_GROUP_FIELD_DOCS below.
+export type RelatedSiteGroup = {
+ label?: string;
+ siteIds: string[];
+};
+
+export const RELATED_SITE_GROUP_FIELD_DOCS: FieldDocs<RelatedSiteGroup> = {
+ label: "Optional muted heading shown above the group; omit for an unlabeled group.",
+ siteIds:
+ "Sibling site ids, in display order. Invalid and repeated ids are dropped, and a group left with none is dropped. Ids are resolved against the live pool at render time, so an id for a site that does not exist (yet) is harmless — it is skipped.",
+};
+
+// A Site is a selection + presentation layer over the single global channel
+// pool. Each field is documented in SITE_FIELD_DOCS below.
+export type Site = {
+ siteId: string;
+ siteTitle: string;
+ siteDescription: string;
+ headerTitle: string;
+ homeTagline: string;
+ socialLinks?: SocialLink[];
+ groups: ChannelGroup[];
+ defaultGroupId: string;
+ channels: SiteChannelMembership[];
+ cloudflareProject?: string;
+ accent?: string;
+ siteUrl?: string;
+ relatedSites?: RelatedSiteGroup[];
+ pwa?: boolean;
+ archives?: boolean;
+ duplicates?: boolean;
+ archiveMaxBytes?: number;
+ hubUrl?: string;
+};
+
+export const SITE_FIELD_DOCS: FieldDocs<Site> = {
+ siteId:
+ "The site's id: a lowercase slug (`[a-z0-9][a-z0-9-]*`), and its directory name under `sites/`. The directory is authoritative — a read takes the id from the path, never from the file.",
+ siteTitle: "The site's title (browser tab, manifest, headings).",
+ siteDescription: "One-line description (meta description, manifest).",
+ headerTitle: "The title shown in the site header.",
+ homeTagline: "Tagline under the home page title. Empty = none.",
+ socialLinks:
+ "Per-site social links. ABSENT means inherit the global default (`settings.json` `socialLinks`); an array — even an empty one — overrides it. Each link's SVG must be safe to inline or the save is refused.",
+ groups:
+ "Channel grouping layout for THIS site: the buckets the export UI renders channel checkboxes in, and which are selected by default. At least one is required on save; a file with none reads as one inline fallback group.",
+ defaultGroupId:
+ "The group a channel falls into when its membership names none (or an unknown one). Must name a configured group on save; on read an unknown value resolves to the first group.",
+ channels:
+ "The channels this site exposes. A channel absent from this list is not built or deployed for this site even though its data exists in the pool.",
+ cloudflareProject:
+ "Cloudflare Pages project name this site deploys to (`wrangler pages deploy out --project-name <cloudflareProject>`). Trimmed; blank = none.",
+ accent:
+ 'Per-site brand accent, `"#rrggbb"`. Overrides the family brass on this site\'s public build. Absent = inherit the family brass. Any other spelling is dropped.',
+ siteUrl:
+ "Absolute public URL of this site's deployment, e.g. `https://jeralyzer.pages.dev` (trimmed, trailing slashes removed; anything not absolute http(s) is dropped). Drives the cross-site footer: a site with no siteUrl is omitted from every other site's list.",
+ relatedSites:
+ "Pulls specific siblings to the front of the footer's cross-site list, in named groups. Siblings not named here fall into a trailing \"Other sites\" group. Absent/empty = one flat list of every sibling.",
+ pwa:
+ "Whether this site ships an installable PWA (service worker + web manifest). Default false: a \"dumb instance\" that serves the CORS-enabled JSON federation contract but is not independently installable, so a visitor trusts only the hub PWA. Stored only when true.",
+ archives:
+ "Whether the site build generates downloadable transcript/live-chat archive zips (and links them on the Downloads page). Opt-OUT: absent/true = on, only an explicit `false` disables. Also gated by the global setting and a per-build flag.",
+ duplicates:
+ "Whether this site publishes the Duplicates page (and its header link). Opt-OUT: absent/true = on, only an explicit `false` hides it. Even when on, the page auto-hides when the site has no in-scope duplicate clusters.",
+ archiveMaxBytes:
+ "Per-site served-file size cap in bytes: any archive larger is dropped from what is served and flagged in the manifest, so a capped host (Cloudflare Pages: 25 MB) will not reject the deploy. 0 = no cap. Absent = the global default. Negative or non-numeric values are dropped.",
+ hubUrl:
+ "Per-site override for the hub this site belongs under (the PWA it points visitors toward). Absent = the family default, `settings.json` `homepageUrl`. Surfaced on the public /site.json so a hub can tell member sites from arbitrary added origins.",
+};
+
+// siteId shares the group-id grammar: lowercase slug, used as a directory name.
+export const SITE_ID_RE = /^[a-z0-9][a-z0-9-]*$/;
+
+export function isValidSiteId(id: unknown): id is string {
+ return typeof id === "string" && SITE_ID_RE.test(id);
+}
+
+export const SITE_DEFAULT_TITLE = "Transcript Browser";
+export const SITE_DEFAULT_DESCRIPTION = "Browse and search video transcripts";
+
+export function parseSiteChannels(input: unknown): SiteChannelMembership[] {
+ if (!Array.isArray(input)) return [];
+ const out: SiteChannelMembership[] = [];
+ const seen = new Set<string>();
+ for (const raw of input) {
+ if (!raw || typeof raw !== "object") continue;
+ const r = raw as Record<string, unknown>;
+ const slug = typeof r.slug === "string" ? r.slug.trim() : "";
+ if (!slug || seen.has(slug)) continue;
+ seen.add(slug);
+ const entry: SiteChannelMembership = { slug };
+ if (typeof r.groupId === "string" && r.groupId.trim()) {
+ entry.groupId = r.groupId.trim();
+ }
+ if (typeof r.order === "number" && Number.isFinite(r.order)) {
+ entry.order = Math.floor(r.order);
+ }
+ out.push(entry);
+ }
+ return out;
+}
+
+// Normalize a raw siteUrl into a trimmed absolute http(s) URL with no trailing
+// slash, or undefined when missing/not a usable absolute URL. Relative or
+// scheme-less values are rejected — a cross-site link must be absolute.
+export function parseSiteUrl(input: unknown): string | undefined {
+ if (typeof input !== "string") return undefined;
+ const trimmed = input.trim().replace(/\/+$/, "");
+ if (!/^https?:\/\/\S+/i.test(trimmed)) return undefined;
+ return trimmed;
+}
+
+// Parse the relatedSites override: an ordered list of { label?, siteIds[] }
+// groups. Keeps only valid site ids, dedupes within a group, drops groups with
+// no valid ids, and preserves order. Existence against the live pool is NOT
+// checked here — resolveRelatedSites does that at render time.
+export function parseRelatedSites(input: unknown): RelatedSiteGroup[] {
+ if (!Array.isArray(input)) return [];
+ const out: RelatedSiteGroup[] = [];
+ for (const raw of input) {
+ if (!raw || typeof raw !== "object") continue;
+ const r = raw as Record<string, unknown>;
+ const ids = Array.isArray(r.siteIds) ? r.siteIds : [];
+ const siteIds: string[] = [];
+ const seen = new Set<string>();
+ for (const id of ids) {
+ if (!isValidSiteId(id) || seen.has(id)) continue;
+ seen.add(id);
+ siteIds.push(id);
+ }
+ if (siteIds.length === 0) continue;
+ const label =
+ typeof r.label === "string" && r.label.trim() ? r.label.trim() : undefined;
+ out.push(label ? { label, siteIds } : { siteIds });
+ }
+ return out;
+}
+
+// The per-key coercions. Each is total over `unknown`.
+const stringOr = (fallback: string) => (v: unknown): string =>
+ typeof v === "string" ? v : fallback;
+
+function groupsOrFallback(v: unknown): ChannelGroup[] {
+ const groups = parseChannelGroups(v);
+ // A single default group selected by default keeps a site with no explicit
+ // group config behaving like the pre-grouping UI (one bucket, all checked).
+ return groups.length > 0 ? groups : [{ ...FALLBACK_GROUP }];
+}
+
+function archiveMaxBytesOf(v: unknown): number | undefined {
+ return typeof v === "number" && Number.isFinite(v) && v >= 0
+ ? Math.floor(v)
+ : undefined;
+}
+
+// The per-key object. Its key order is the order parseSite has always emitted
+// (and SITE_FIELD_DOCS's). Built ONCE: `siteId` is never read from the file —
+// parseSite fills it from the caller — so nothing in the schema depends on it.
+const d = SITE_FIELD_DOCS;
+export const siteFieldsSchema = z.object({
+ siteId: settingsField((): string => "").describe(d.siteId),
+ siteTitle: settingsField(stringOr(SITE_DEFAULT_TITLE)).describe(d.siteTitle),
+ siteDescription: settingsField(stringOr(SITE_DEFAULT_DESCRIPTION)).describe(
+ d.siteDescription,
+ ),
+ headerTitle: settingsField(stringOr(SITE_DEFAULT_TITLE)).describe(d.headerTitle),
+ homeTagline: settingsField(stringOr("")).describe(d.homeTagline),
+ // Key present (array) = override; absent = inherit the global default.
+ socialLinks: settingsField((v): SocialLink[] | undefined =>
+ Array.isArray(v) ? parseSocialLinks(v) : undefined,
+ ).describe(d.socialLinks),
+ groups: settingsField(groupsOrFallback).describe(d.groups),
+ // Resolved against `groups` in the object step below.
+ defaultGroupId: settingsField((v): unknown => v).describe(d.defaultGroupId),
+ channels: settingsField(parseSiteChannels).describe(d.channels),
+ cloudflareProject: settingsField((v): string | undefined =>
+ typeof v === "string" && v.trim() ? v.trim() : undefined,
+ ).describe(d.cloudflareProject),
+ accent: settingsField(parseAccent).describe(d.accent),
+ siteUrl: settingsField(parseSiteUrl).describe(d.siteUrl),
+ relatedSites: settingsField(parseRelatedSites).describe(d.relatedSites),
+ pwa: settingsField((v): boolean => v === true).describe(d.pwa),
+ // Opt-out: only an explicit false disables. Absent/true stays on.
+ archives: settingsField((v): boolean => v !== false).describe(d.archives),
+ duplicates: settingsField((v): boolean => v !== false).describe(d.duplicates),
+ archiveMaxBytes: settingsField(archiveMaxBytesOf).describe(d.archiveMaxBytes),
+ hubUrl: settingsField(parseSiteUrl).describe(d.hubUrl),
+});
+
+// The whole schema: the per-key object, then the two sibling-dependent fields.
+//
+// THE OBJECT STEP ALSO RE-EMITS EVERY KEY, in order. zod 4 omits a key that was
+// absent from the input when its transform returns `undefined`, so the per-key
+// output lacks `socialLinks`, `accent`, … on a file that does not spell them —
+// where parseSite has always emitted them, present and `undefined`. Rebuilding
+// from SITE_KEYS keeps that shape (and the key order) exactly.
+export const siteSchema = siteFieldsSchema.transform((s): Site => {
+ const groups = s.groups;
+ const resolved: Partial<Record<keyof Site, unknown>> = {
+ ...s,
+ defaultGroupId: resolveDefaultGroupId(s.defaultGroupId, groups),
+ // Drop a membership's groupId that names no configured group (it folds to
+ // the default at render time via resolveChannelGroupId); keep a valid one
+ // so the editor round-trips it.
+ channels: s.channels.map((c) => {
+ if (c.groupId && !groups.some((g) => g.id === c.groupId)) {
+ const { groupId: _drop, ...rest } = c;
+ return rest;
+ }
+ return c;
+ }),
+ };
+ const out: Partial<Record<keyof Site, unknown>> = {};
+ for (const key of SITE_KEYS) out[key] = resolved[key];
+ return out as Site;
+});
+
+// The keys of site.json, in the schema's order — SITE.md's order, and the
+// unknown-key oracle.
+export const SITE_KEYS = Object.keys(SITE_FIELD_DOCS) as ReadonlyArray<keyof Site>;
+
+// Parse a raw site.json value into a fully-resolved Site. Never throws: a
+// missing, non-object or ill-typed file reads as the defaults field by field.
+export function parseSite(siteId: string, raw: unknown): Site {
+ const obj = raw && typeof raw === "object" && !Array.isArray(raw) ? raw : {};
+ const site = siteSchema.parse(obj);
+ site.siteId = siteId;
+ return site;
+}
+
+// What a save puts on disk for an ALREADY-VALIDATED site: only the keys that
+// are not defaults. Pure — writeSite (lib/site.ts) runs the throwing
+// validations first and passes their normalized results in.
+export function siteToDisk(site: Site): Site {
+ const groups = parseChannelGroups(site.groups);
+ const channels = parseSiteChannels(site.channels).filter(
+ (c) => !c.groupId || groups.some((g) => g.id === c.groupId),
+ );
+ const relatedSites = parseRelatedSites(site.relatedSites);
+ const accent = parseAccent(site.accent);
+ const siteUrl = parseSiteUrl(site.siteUrl);
+ const hubUrl = parseSiteUrl(site.hubUrl);
+ const archiveMaxBytes = archiveMaxBytesOf(site.archiveMaxBytes);
+ return {
+ siteId: site.siteId,
+ siteTitle: site.siteTitle,
+ siteDescription: site.siteDescription,
+ headerTitle: site.headerTitle,
+ homeTagline: site.homeTagline,
+ // undefined socialLinks = inherit the global default; only an explicit
+ // override (an array, even empty) is persisted.
+ ...(site.socialLinks !== undefined ? { socialLinks: site.socialLinks } : {}),
+ groups,
+ defaultGroupId: site.defaultGroupId,
+ channels,
+ ...(site.cloudflareProject && site.cloudflareProject.trim()
+ ? { cloudflareProject: site.cloudflareProject.trim() }
+ : {}),
+ ...(accent ? { accent } : {}),
+ ...(siteUrl ? { siteUrl } : {}),
+ ...(relatedSites.length > 0 ? { relatedSites } : {}),
+ ...(site.pwa ? { pwa: true } : {}),
+ // Persist only the non-default: archives is on unless explicitly disabled.
+ ...(site.archives === false ? { archives: false } : {}),
+ ...(site.duplicates === false ? { duplicates: false } : {}),
+ ...(archiveMaxBytes !== undefined ? { archiveMaxBytes } : {}),
+ ...(hubUrl ? { hubUrl } : {}),
+ };
+}
+
diff --git a/common/lib/transcribeOutcome-server.ts b/common/lib/transcribeOutcome-server.ts
@@ -1,38 +1,29 @@
-import path from "node:path";
-import { readFile, rename, writeFile } from "node:fs/promises";
import {
TRANSCRIBE_OUTCOME_FILENAME,
type TranscribeOutcomeRecord,
} from "./transcribeOutcome";
+import { sidecar, sidecarField } from "./sidecar-server";
-export function transcribeOutcomePath(videoDir: string): string {
- return path.join(videoDir, TRANSCRIBE_OUTCOME_FILENAME);
-}
-
-export async function loadTranscribeOutcome(
- videoDir: string,
-): Promise<TranscribeOutcomeRecord | null> {
- try {
- const raw = await readFile(transcribeOutcomePath(videoDir), "utf8");
- const parsed = JSON.parse(raw) as Partial<TranscribeOutcomeRecord>;
- if (
- typeof parsed?.videoId === "string" &&
- typeof parsed.transcribedAt === "string"
- ) {
- return parsed as TranscribeOutcomeRecord;
- }
- return null;
- } catch {
- return null;
+export function coerceTranscribeOutcome(
+ value: unknown,
+): TranscribeOutcomeRecord | null {
+ const parsed = value as Partial<TranscribeOutcomeRecord> | null;
+ if (
+ typeof parsed?.videoId === "string" &&
+ typeof parsed.transcribedAt === "string"
+ ) {
+ return parsed as TranscribeOutcomeRecord;
}
+ return null;
}
-export async function writeTranscribeOutcome(
- videoDir: string,
- record: TranscribeOutcomeRecord,
-): Promise<void> {
- const file = transcribeOutcomePath(videoDir);
- const tmp = `${file}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(record, null, 2) + "\n");
- await rename(tmp, file);
-}
+export const transcribeOutcomeSidecar = sidecar(
+ TRANSCRIBE_OUTCOME_FILENAME,
+ sidecarField(coerceTranscribeOutcome),
+);
+
+export const {
+ path: transcribeOutcomePath,
+ load: loadTranscribeOutcome,
+ write: writeTranscribeOutcome,
+} = transcribeOutcomeSidecar;
diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts
@@ -83,7 +83,9 @@ export type SubTrack = {
// primary transcript outputs (transcript.en.vtt, transcript.json) or a
// derived/auxiliary file (transcript.cues.json). Live chat lands as
// transcript.live_chat.json; non-en languages as transcript.<lang>.vtt.
-const SUB_FILE_RE = /^transcript\.([^.]+)\.([^.]+)$/;
+// Exported for lib/sidecar-server.ts, which refuses to declare a sidecar this
+// matches (a sidecar so named would be read as a subtitle track).
+export const SUB_FILE_RE = /^transcript\.([^.]+)\.([^.]+)$/;
const SUB_EXT_VALUES = ["vtt", "json", "json3", "srv1", "srv2", "srv3"] as const;
type SubExt = (typeof SUB_EXT_VALUES)[number];
function isSubExt(value: string): value is SubExt {
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -4,12 +4,12 @@ import { execa } from "execa";
import pLimit from "p-limit";
import { readArchive } from "../lib/archive";
import {
- parseChannelConfig,
type AudioFormat,
type ChannelConfig,
type ChannelHandling,
} from "../lib/channelConfig";
import { getSettings } from "../lib/settings";
+import { patchChannelConfig } from "../controller/channels";
import { diskGate } from "../lib/diskSpace";
import { detectPlatform } from "../lib/platform";
import { isRealAudioFile } from "../lib/videoStatus";
@@ -1807,40 +1807,27 @@ async function runChildAndStream(
}
}
+// The three sync-state stamps (CHANNEL_SYNC_STATE_KEYS). Each re-reads the
+// config at the moment it writes, so an edit made while the sync ran survives;
+// a channel whose config.json is gone or unreadable is not recreated.
async function touchLastSync(opts: RunYtdlpOpts): Promise<void> {
- const file = path.join(channelRoot(opts), "config.json");
- await updateConfigField(file, "lastSyncedAt", new Date().toISOString());
+ await patchChannelConfig(opts.paths, opts.channelSlug, {
+ lastSyncedAt: new Date().toISOString(),
+ });
}
async function touchLastFullDownload(opts: RunYtdlpOpts): Promise<void> {
- const file = path.join(channelRoot(opts), "config.json");
- await updateConfigField(file, "lastFullDownloadAt", new Date().toISOString());
+ await patchChannelConfig(opts.paths, opts.channelSlug, {
+ lastFullDownloadAt: new Date().toISOString(),
+ });
}
// Stamps when this channel last paid for a full enumeration, which is what the
// cadence gate reads to keep the next N syncs on the cheap paged walk.
async function touchLastFullSweep(opts: RunYtdlpOpts): Promise<void> {
- const file = path.join(channelRoot(opts), "config.json");
- await updateConfigField(file, "lastFullSweepAt", new Date().toISOString());
-}
-
-async function updateConfigField(
- configPath: string,
- field: "lastSyncedAt" | "lastFullDownloadAt" | "lastFullSweepAt",
- value: string,
-): Promise<void> {
- let raw: string;
- try {
- raw = await readFile(configPath, "utf8");
- } catch {
- return;
- }
- const parsed = parseChannelConfig(JSON.parse(raw));
- if (!parsed) return;
- parsed[field] = value;
- const tmp = `${configPath}.tmp-${process.pid}`;
- await writeFile(tmp, JSON.stringify(parsed, null, 2) + "\n");
- await rename(tmp, configPath);
+ await patchChannelConfig(opts.paths, opts.channelSlug, {
+ lastFullSweepAt: new Date().toISOString(),
+ });
}
// The channel's archive file stores yt-dlp's native extractor ids (e.g.
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -2,6 +2,7 @@
## [Unreleased]
- **Every channel table and every job-in-flight line is now drawn one way.** The /channels rack, the dashboard's Channels table and the work tables on the operation pages and /cleanup are one table with a column set per page, over one channel row built on the server (which no longer ships a channel's config to the browser); the dashboard's "Needs work" seed is computed by the same code the widget endpoint serves. On the jobs side, /jobs rows, the "Active jobs" cards on channel/video/operation pages, the monitor widget's Active jobs strip and the operations board's "In flight" list are one job row in three sizes, with one rule for which buttons (Retry / Reorder / Drain / Cancel / Force-release) a job gets. **What you might notice:** a work table's report column reads "stale"/"missing" like the rack's instead of a date; the dashboard's Sync button is the rack's; a lane line on /jobs offers Force-release while its runner is running; widget job lines show who asked for the job; an in-flight download on the operations board links to its job page. Nothing a count says moved.
+- **`site.json`, each channel's `config.json` and the per-video sidecars now have one schema each, and the two config files have generated key tables.** **`SITE.md`** and **`CHANNEL.md`** (new, repo root) list every key with its default and meaning, generated by `common/bin/file-schemas-docs.ts` and checked by a test. Nothing an operator has configured reads or saves differently: every live `site.json` and `config.json`, and a 1,763-file sample of sidecars, read and write back byte-for-byte as before. **Fixed:** a social-channel fetch no longer undoes Configure-form edits made while it was running (it used to write back the whole config it read when it started). Every change to a channel's config now re-reads the file at the moment it saves and changes only its own fields, so a sync stamping its time and a form save made at the same moment both land. Two writes to the same file from the editor no longer share one temporary file.
- **`settings.json` has one schema and one writer, and its key table is generated.** Every key, its default, its clamp and its documentation is now one zod schema (`common/lib/settingsSchema.ts`); `getSettings`/`writeSettings` both parse through it, and every settings form saves through one helper (`editor/app/settings/saveSettings.ts`) that merges only what the form changed. **`SETTINGS.md`** (new, repo root) lists every key with its default and what it does, and `settings.json.example` is now the full default object — both generated by `common/bin/settings-example.ts` and checked by a test, so neither can drift. Nothing an operator has configured reads differently. **Fixed:** adding or editing a storage location on `/storage` no longer erases the record of which location the saved-video store is on (`storage.savedVideosLocationId`).
- **Every live panel now polls one endpoint, `/api/view/<name>`, and the eight old addresses still answer.** The change token, the job head, workers, the operations board, the sync console and the widget's three strips were eight separate API routes that each did the same thing; they are one route serving eight named views (`pulse`, `activeJobs`, `workers`, `autoQueueStatus`, `schedulerStatus`, `widgetSync`, `widgetActionable`, `cleanable`), and the editor's own pages poll it there. `/api/pulse`, `/api/jobs/active`, `/api/workers`, `/api/auto-queue/status`, `/api/scheduler/status` and `/api/widget/{sync,actionable,cleanable}` are kept as **rewrites**, not redirects — same method, status, body and query string (`?rev=` included) — so a monitor widget pinned in a browser, or any script polling the old path, keeps working untouched. `/api/widget/presets` is unchanged. An unknown view name is a 404. **One behaviour change you might notice:** the operations board (every 3 s) and the sync console (every 5 s) now send their next poll only after the previous one answers, and abandon a poll that takes longer than 10–15 s — so a slow editor no longer piles requests up behind itself, and a hung request no longer stops the page updating.
- **A lane's pause is one key on the lane, and the four old pause fields are gone from `settings.json`.** Holding a lane has been `autoQueue.<lane>.held` since the runner work landed; until now the file also still carried the four flags that used to mean it — `transcriptionsPaused`, `downloadsPaused`, `digest.digestsPaused` and the backwards `backfill.enabled` (where *enabled* meant *not held*) — which were read only when a lane had no `held` yet, to carry an older file's pause across. Every lane now carries its own key, so those four are **deleted**: nothing reads them, no form writes them, and the next settings save drops them from the file. A settings.json that still spells one of them holds nothing with it, so a hand-edited file (or a very old backup restored over a newer one) can no longer resurrect a pause you had lifted, or lift one you had set. "Run the backfill lane" on the diarization page and the Hold/Pause buttons write the one key, as they already did. **UPGRADING: boot once on the release that writes `held` before taking this one.** That release is the one that moved the gate onto the lane and carried the old fields across on read; a single boot of it (any settings save, or just starting the editor and pausing/resuming anything) puts `autoQueue.<lane>.held` in your settings.json, after which **nothing you can see changes here** — the same buttons, the same labels, the same pauses. An install that jumps straight from an older release to this one has no `held` keys at all and **loses its pauses**: transcription, downloads and digests come up running, and the backfill lane comes up held. Re-set them from the dashboard, or add the keys by hand before starting.
@@ -55,12 +56,10 @@
- **A scope naming an operation that no longer exists no longer arms a sweep that runs forever doing nothing.** Settings sanitation keeps an unknown operation id, and the resolver then matches nothing with it — so the sweep starts, reports itself armed, and holds at zero. Unknown ids are now dropped at the moment of arming, with the console saying which; a scope that names *only* unknown operations is refused outright rather than started.
- **The feed and the gate stopped sounding alike.** *Start sweep* and *Pause Backfill* were the same shape of phrase for two acts whose costs to undo differ by a week of GPU time. The feed **sweeps** — *Sweep every channel* / *Stop sweeping* — and the gate **holds** — *Hold the lane* / *Resume the lane*. The lane heading says what it holds and what this panel does with it: *Speakers · sweep*. The settings fieldset no longer implies scope lives there, and points at the console instead.
- **The plan costs nothing to draw.** The page was already reading every channel's snapshot for the comparison rail and throwing the rest away — third cycle running that this has been true. The run's own planner could not be used: it *walks the corpus* when a channel has never been reported (~474,559 file touches, ~4 seconds) and this refreshes every three seconds. So the preview counts off the snapshots with the run's own arithmetic — verified against the live archive at 68 channels, 78,757 outstanding: **identical totals, identical per-channel counts, identical ordering** — and a channel nothing has reported on is drawn as *not reported yet* rather than as a confident zero. The guard test that keeps the corpus walk out of render paths grepped for one identifier and would have let the sweep planner straight through; it now names every function that reaches it.
-
- **One word was standing in for three different things, and hiding a fourth.** "Backfill" named a *lane* (three operations sharing a CPU queue), a *per-channel action*, and — on the download stage — a completely unrelated **download job** for subtitle tracks. Nobody arms, pauses or runs "a backfill"; it is a queue key. Every screen where an operator reads a **figure** now reads an **operation name**: the channel transit line's station is *Speakers*, its stage panel is *Speaker work · 3 operations* with a *Run speaker work* button, the download stage says *Fetch missing subtitle tracks*, and the lane card and `/auto-queue` name their members. The lane name survives in exactly one place — the settings fieldset and the lane card, where the one shared pause and the one shared sweep live — and there it now **lists what it holds**, because "these operations share a queue and a pause" is a true and load-bearing fact rather than a leaked implementation detail.
- **The transit line's Speakers station stopped summing three operations into one number.** It set its coverage by adding every operation on the lane together — the one thing this codebase forbids everywhere else — which on the live corpus meant adding diarization (one audio pass per video, 4 done of 11,338) to speaker-names-from-the-transcript (**~1 model call per transcript chunk**, 1 done of 11,338) and printing the total under a label that named the queue. The numeral now belongs to exactly one operation, and each member states itself, with its own band and its own denominator, in the station foot.
- **An armed operation now says what one unit of it costs.** The fact that hid: speaker-names-from-the-transcript is switched **on**, can reach 11,337 videos of one channel — on the order of **194,000 model calls corpus-wide** — has completed one video, and read as a quiet row on every screen, because "11,337 reachable" is the same shape of number whether the unit is an audio pass or a per-chunk model call. Each operation declares its cost basis, printed beside its backlog wherever it is armed. Deliberately **no threshold and no editorialising**: what is affordable is the operator's call, and a "this is a lot" cutoff would be a magic number the next operation gets wrong. Nothing here changes a setting.
- **Every pipeline is now visible on /channels, not just two of them.** The table printed two bare integers — `Downloads` and `Transcripts` — and said nothing whatsoever about the four derived pipelines, which were collapsed everywhere else behind the word *backfill*. All six now draw the same **state band** the comparison rail on /auto-queue uses, at table scale: one fill per population, `can run now` the only saturated colour on the page, and pattern (solid / hatched / dotted / hollow) carrying the meaning ahead of hue, because no four-colour palette clears all-pairs colour-blindness. **Not a percent bar, deliberately** — digest sits at 0 done on every large channel and diarization is 99.96% media-gone, so "% complete" renders `0%` on all 68 rows and says nothing; what varies, and what an operator needs, is the *shape* of the remainder. Every pipeline column **sorts by what can run now**, which answers a question the page has never been able to answer: *which channel has the most diarizable audio left right now* previously meant opening 68 channel pages one at a time. The costs nothing extra to draw — `/channels` was already reading every channel's full snapshot for its counts and throwing the rest away.
-
- **`docker compose up -d` now stands up a working archive.** The repo had two Dockerfiles and neither ran the app: one fans per-site export builds out across containers, the other runs sharded e2e. So the only way to host this was to install the whole Unix toolchain by hand, which is why the Windows instructions said "use WSL2 and follow the Linux steps". There is now a runtime image and a compose stack — the editor plus Caddy by default, with the published site, the project homepage and umtool behind compose **profiles**, so somebody who only wants an archive runs two containers rather than five. First boot creates the volumes, downloads a speech model, and seeds a `settings.json` **carrying one enabled worker**: the defaults ship `workers: []`, zero workers means zero transcription slots, and a fresh container that looks healthy and silently transcribes nothing is the worst possible first run. Two things are deliberately not baked into the image and cannot be: the corpus, and the export site — that site is a static render *of* a corpus, and there is no corpus at image-build time, so `docker/publish-site.sh` builds it at run time into the volume the `site` service serves.
- **Nothing the container runs is reachable from outside the machine until you say so.** The editor has no authentication of any kind, shells out to yt-dlp, and deletes media — so **no application container publishes a port at all**. Caddy is the single front door, and every one of the four ports it publishes binds `127.0.0.1` by default, the public sites included; opening one is a deliberate edit of a single line in `.env`. Because a bind address is exactly the sort of thing that gets changed in a hurry, there is also a rail: if a private app is bound off-loopback with nothing checking credentials, **the containers refuse to start** — both the app and Caddy, which is the process that actually opens the ports — and print the four ways to fix it. `basic_auth` is built into Caddy so a password needs nothing installed; Tinyauth and Authelia attach through `forward_auth` as documented drop-in overlays; `ARCHILYZER_AUTH_MODE=none` is the one explicit escape hatch for people who already have their own front door.
- **The GPU is usable from the container, including on AMD.** Alongside the default CPU whisper.cpp image there is a **Vulkan** target running parakeet.cpp — one build that covers AMD (RADV), Intel and NVIDIA, needing nothing on the host but a render node at `/dev/dri` and no vendor container toolkit — and a CUDA target for NVIDIA whisper. Measured on an RX 6600 XT, through the app's own overlapping-window wrapper: **3.4 s against 36.3 s** for the same 33-second clip pinned to the CPU. That gap is also the thing to watch for, because the failure here is silent — a Vulkan container with no `/dev/dri` does not error, it transcribes correctly on the CPU about ten times slower. The entrypoint prints which one it got on every boot, and the image ships `vulkaninfo` so you can ask directly.
diff --git a/editor/app/channels/[slug]/socialActions.ts b/editor/app/channels/[slug]/socialActions.ts
@@ -14,8 +14,8 @@ import {
resolveQueueKey,
} from "yt-dlp-transcript-common/lib/queueKeys";
import {
+ patchChannelConfig,
readChannelConfig,
- writeChannelConfig,
} from "yt-dlp-transcript-common/controller/channels";
import { isSocialChannel } from "yt-dlp-transcript-common/lib/channelConfig";
import { fetchPosts } from "yt-dlp-transcript-common/controller/fetchPosts";
@@ -56,7 +56,7 @@ export async function setPostFetcherAction(
error: `${fetcher.label} handles ${fetcher.platform}, not ${config.platform}.`,
};
}
- await writeChannelConfig(paths, slug, { ...config, postFetcher: wanted });
+ await patchChannelConfig(paths, slug, { postFetcher: wanted });
revalidatePath(`/channels/${slug}`);
return { ok: true };
}
diff --git a/editor/app/channels/actions.ts b/editor/app/channels/actions.ts
@@ -17,8 +17,8 @@ import {
deleteChannel,
isValidChannelSlug,
listChannelConfigs,
+ patchChannelConfig,
readChannelConfig,
- writeChannelConfig,
} from "yt-dlp-transcript-common/controller/channels";
import { renameChannel } from "yt-dlp-transcript-common/controller/renameChannel";
import { inspectChannelMedia } from "yt-dlp-transcript-common/lib/channelMedia";
@@ -309,14 +309,15 @@ export async function updateChannelAction(
}
// The form parser only emits keys whose form value is meaningful, so a
// cleared input is absent from `parsed.config`. A plain spread would keep
- // the stale value from `existing`. Clear every form-managed key from the
- // baseline first, then layer the parsed config — non-form fields like
- // subLangs / lastSyncedAt / lastFullDownloadAt / excludeFromBuild are
- // preserved automatically.
- const merged: typeof existing = { ...existing };
- for (const key of CHANNEL_FORM_FIELDS) delete merged[key];
- Object.assign(merged, parsed.config);
- await writeChannelConfig(paths, slug, merged);
+ // the stale value. So the patch UNSETS every form-managed key first, then
+ // layers the parsed config — non-form fields like subLangs, the sync-state
+ // stamps (CHANNEL_SYNC_STATE_KEYS) and excludeFromBuild are preserved, and
+ // are re-read at write time, so a sync that stamped lastSyncedAt while the
+ // form was open is not reverted.
+ const written = await patchChannelConfig(paths, slug, parsed.config, {
+ unset: CHANNEL_FORM_FIELDS,
+ });
+ if (!written) return { error: `Channel "${slug}" not found` };
try {
await applySiteWrites(siteWrites, paths);
} catch (e) {
@@ -464,13 +465,9 @@ export async function toggleChannelBuildInclusionAction(
const paths = getPaths();
const existing = await readChannelConfig(paths, slug);
if (!existing) return { error: `Channel "${slug}" not found` };
- const next = { ...existing };
- if (existing.excludeFromBuild) {
- delete next.excludeFromBuild;
- } else {
- next.excludeFromBuild = true;
- }
- await writeChannelConfig(paths, slug, next);
+ await (existing.excludeFromBuild
+ ? patchChannelConfig(paths, slug, {}, { unset: ["excludeFromBuild"] })
+ : patchChannelConfig(paths, slug, { excludeFromBuild: true }));
revalidatePath("/channels");
return undefined;
}
@@ -485,13 +482,9 @@ export async function toggleChannelCleanupInclusionAction(
const paths = getPaths();
const existing = await readChannelConfig(paths, slug);
if (!existing) return { error: `Channel "${slug}" not found` };
- const next = { ...existing };
- if (existing.excludeFromCleanup) {
- delete next.excludeFromCleanup;
- } else {
- next.excludeFromCleanup = true;
- }
- await writeChannelConfig(paths, slug, next);
+ await (existing.excludeFromCleanup
+ ? patchChannelConfig(paths, slug, {}, { unset: ["excludeFromCleanup"] })
+ : patchChannelConfig(paths, slug, { excludeFromCleanup: true }));
revalidatePath("/cleanup");
revalidatePath("/channels");
return undefined;
diff --git a/editor/app/scheduler/actions.ts b/editor/app/scheduler/actions.ts
@@ -2,14 +2,12 @@
import { revalidatePath } from "next/cache";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import {
- readChannelConfig,
- writeChannelConfig,
-} from "yt-dlp-transcript-common/controller/channels";
+import { patchChannelConfig } from "yt-dlp-transcript-common/controller/channels";
import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler";
import {
SYNC_INTERVAL_MAX_MINUTES,
SYNC_INTERVAL_MIN_MINUTES,
+ type ChannelConfig,
} from "yt-dlp-transcript-common/lib/channelConfig";
import {
getSettings,
@@ -83,18 +81,19 @@ export async function setChannelCadencesAction(
const paths = getPaths();
for (const slug of slugs) {
- const existing = await readChannelConfig(paths, slug);
- if (!existing) return { ok: false, error: `Channel "${slug}" not found` };
- const next = { ...existing };
+ // `minutes: undefined` = inherit the global default = unset the override.
+ const patch: Partial<ChannelConfig> = {};
+ const unset: Array<keyof ChannelConfig> = [];
if (sync !== "keep") {
- if (sync.minutes === undefined) delete next.syncIntervalMinutes;
- else next.syncIntervalMinutes = sync.minutes;
+ if (sync.minutes === undefined) unset.push("syncIntervalMinutes");
+ else patch.syncIntervalMinutes = sync.minutes;
}
if (sweep !== "keep") {
- if (sweep.minutes === undefined) delete next.fullSweepIntervalMinutes;
- else next.fullSweepIntervalMinutes = sweep.minutes;
+ if (sweep.minutes === undefined) unset.push("fullSweepIntervalMinutes");
+ else patch.fullSweepIntervalMinutes = sweep.minutes;
}
- await writeChannelConfig(paths, slug, next);
+ const written = await patchChannelConfig(paths, slug, patch, { unset });
+ if (!written) return { ok: false, error: `Channel "${slug}" not found` };
// Keep the channel report/badges in sync with the config edit.
requestChannelSnapshot(paths, slug);
revalidatePath(`/channels/${slug}`);
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -131,10 +131,13 @@ source tree).
(`DisplaySummary` records), documented in `common/lib/corpus.ts:35-36` and served by
`common/components/summariesCache.ts`. Nothing to do with AI summaries. Use **`digest`**.
-**Never name a sidecar `transcript.<x>.<y>`.** `common/lib/videoStatus.ts:137` —
-`const SUB_FILE_RE = /^transcript\.([^.]+)\.([^.]+)$/;` treats any such file as a subtitle
-track. The exclusion list in `readSubTracks` (`:238-256`) is hardcoded. Use
-`ai-digest.json`, `diarization.json`, `attribution.json`.
+**Never name a sidecar `transcript.<x>.<y>`.** `common/lib/videoStatus.ts:88` —
+`export const SUB_FILE_RE = /^transcript\.([^.]+)\.([^.]+)$/;` treats any such file as a
+subtitle track. The exclusion list in `readSubTracks` is hardcoded. Use `ai-digest.json`,
+`diarization.json`, `attribution.json`. Since one-core phase 3 slice 4b (2026-09-24) a
+sidecar is declared with `sidecar(filename, schema)` (`common/lib/sidecar-server.ts`), which
+THROWS at declaration — module load — on a name `SUB_FILE_RE` matches;
+`sidecar-server.test.ts` enumerates `SIDECAR_FILENAMES`.
**`report` already means three things**: `reportDebouncePreset` in settings, the
`refresh-report` job kind, and `/ask`'s `ReportPanel`. Use `feedback` / `review` instead.
diff --git a/plans/one-core-phase-3.md b/plans/one-core-phase-3.md
@@ -917,3 +917,181 @@ that range when bisecting.
7. Lane line: Force-release only when the runner job is stuck (deviation 3 narrowed).
8. Dead exports removed (loadActionable's `reportStateOf` re-export, function-form column
labels + `ChannelHeadCtx`, JobRow's internal helpers).
+
+## Slice 4b, as shipped — the rest of the file schemas (2026-09-24)
+
+Branch `one-core/phase-3-s4b` off `main` `1e27f7c3`, eight commits, then `main` (`3241fed2`,
+slice 3a) merged in — merge sha and the post-merge gates in the follow-up below.
+
+| sha | what |
+|---|---|
+| `03164485` | `lib/jsonFile-server.ts` (+test): `readJsonFile` (+sync twin) → `{ok, value} \| {ok:false, reason: absent\|unreadable\|unparseable}`; `writeJsonAtomic` — unique temp name, writes chained per absolute path; `writeJsonAtomicSync` for the three synchronous stores. The 7 private copies (digest-server, chartsStore, aliasesStore, curatedTagsStore, buildIndex, buildStats, compose-homepage) and the inline tmp JSON writes in site.ts, settings.ts, controller/channels.ts, runYtdlp.ts, channelSnapshot.ts, the 7 single-file sidecar servers and the write halves of savedVideo-/posts-/clipWindow-server folded. `getSettings` reads through `readJsonFileSync`. Every file keeps its bytes |
+| `bb7fe82c` | `lib/sidecar-server.ts` (+test): `sidecar(filename, schema, {indent})` → `{filename, path, load, write, remove}`; throws at declaration on a `SUB_FILE_RE` match; `SIDECAR_FILENAMES`; `sidecarField(coerce)`. 8 pairs / 9 files declared, each shape check an exported `coerceX`, the old `loadX`/`writeX`/`xPath` names kept. `SUB_FILE_RE` exported (`videoStatus.ts:88`); `diarization.test.ts` uses it |
+| `2bf65626` | `lib/siteSchema.ts` (+test): per-key `settingsField` object over the existing parsers + one object step (defaultGroupId, membership groupIds, re-emit every key); `parseSite`, `siteToDisk`; the Site types, `SITE_ID_RE` and the three parsers moved in; `site.ts` = I/O + resolvers, `export *`. `SITE_FIELD_DOCS`, `SITE_CHANNEL_MEMBERSHIP_FIELD_DOCS`, `RELATED_SITE_GROUP_FIELD_DOCS`; `CHANNEL_GROUP_FIELD_DOCS` in `channelGroups.ts` |
+| `e75047f8` | `CHANNEL_CONFIG_COERCIONS` in `channelConfig.ts` (pure) — the one coercion set, extracted verbatim; `parseChannelConfig` composes it zod-free. `lib/channelConfigSchema.ts` (server-only zod, +test) composes the same functions + `stripUndefined`. `CHANNEL_CONFIG_FIELD_DOCS` (the type's comments moved in), `AUDIO_CHECK_FIELD_DOCS`, `DOWNLOAD_FILTER_FIELD_DOCS`, `CHANNEL_CONFIG_KEYS` (from the docs record), `CHANNEL_SYNC_STATE_KEYS` |
+| `f6a08bd1` | `controller/channels.ts`: `readChannelConfigFile`, strict throwing `writeChannelConfig`, `patchChannelConfig(paths, slug, patch, {unset})` under `withJsonFileLock`. Repointed: runYtdlp's three stamps (`updateConfigField` deleted), fetchPosts, relocateChannelMedia ×2, storageLocations ×2, renameChannel, socialActions, the channel form (`{unset: CHANNEL_FORM_FIELDS}`), both exclude toggles, the scheduler's bulk cadence save. buildIndex/buildStats readers folded (+tests) |
+| `7c03d7c9` | `bin/file-schemas-docs.ts` (+`--check`), `lib/fileSchemaDocs.ts` (+test) → root `SITE.md`, `CHANNEL.md`; settingsDocs' table helpers exported; SETUP.md, AGENTS.md links; FACTS `SUB_FILE_RE` anchor corrected |
+| `973e59ee` | review fixes — below |
+| (this) | `plans/tools/phase3-files-numbers.ts`, this record, changelog |
+
+**Plan correction — `build:index` is not read-only.** The plan said to time `pnpm --filter
+export build:index` "on the real corpus … read-only over `transcripts/`". It is not: the build
+writes its LMDB to `paths.lmdbPath = $TRANSCRIPTS_DIR/index.mdb` (14 GB on the live corpus),
+which is not env-overridable. Pointed at the primary's `transcripts/` it would have rewritten
+production's index under the live editor. Both timed runs instead used a scratch corpus
+(`$CLAUDE_JOB_DIR/tmp/s4b-buildindex.sh`): a directory of SYMLINKS to the live `channels/`,
+`sites/`, `saved-videos/` and the corpus-wide JSON files, with no `index.mdb`, and
+`EXPORT_PUBLIC_DIR` in scratch — so every read is the real corpus, every write is scratch, and
+each run is a cold full build (the incremental path would skip the very per-video coercions
+being timed).
+
+**Deviations.**
+1. *No mtime freshness in `sidecar()`* — no reader consumes one.
+2. *The channel coercions live in `channelConfig.ts`, and the zod schema wraps them.*
+ `channelConfig.ts` is value-imported by six `"use client"` modules, so `parseChannelConfig`
+ cannot delegate to zod. `CHANNEL_CONFIG_COERCIONS` there is the one set; both
+ `parseChannelConfig` and `channelConfigSchema` compose it; a test pins that the two agree
+ over generated inputs.
+3. *`writeJsonAtomic` takes `newline` as well as `indent`.* The plan's rule ("`\n` only at
+ indent 2") would have changed bytes: attribution/diarization are compact WITH a newline, the
+ chart/alias/tag stores indented WITHOUT one, the export pages compact without one.
+4. *A synchronous `writeJsonAtomicSync`* for chartsStore/aliasesStore/curatedTagsStore (a
+ synchronous API cannot await the chain; nothing else runs while it does).
+5. *The site types and parsers moved into `siteSchema.ts`* (4a's settings pattern, to avoid a
+ `site.ts` ↔ `siteSchema.ts` cycle). And zod 4 OMITS an input-absent key whose transform
+ returns `undefined`, so the object step re-emits every key in `SITE_KEYS` order — the
+ always-emit shape `parseSite` has always had.
+6. *`migrateToSites.ts:101` is not on `readChannelConfigFile`*: it reads the legacy raw `group`
+ key the schema drops.
+7. *`patchChannelConfig` returns what it wrote (or null) and holds a per-path lock.* The two
+ callers that used to fall back to their own copy of a missing config still do, through
+ `writeChannelConfig`: relocateChannelMedia (move out) and renameChannel.
+8. *doNotClean / excludeTruncatedCheck keep their rule*: ANY parseable JSON — even `null` or
+ `3` — reads as a marker (`{setAt:""}`), so the coercion is not "non-null object".
+9. *storageLocations' rollback restores only `dataDir`* (the one field the job changed), not
+ the whole config it found at its start.
+10. *relocateChannelMedia's move-out does not throw on a null patch* (the review asked for a
+ throw at both sites; storageLocations throws). The move-out has no ledger to roll back and
+ the swap has already happened when the config is written, so a throw would leave a swapped
+ link and an unrecorded `dataDir`; it writes the job's own copy of the config instead — the
+ same thing its existing no-config branch did.
+
+**Behaviour changes (all intended).**
+- A social fetch no longer reverts Configure-form edits made while it ran (`fetchPosts` wrote
+ back the config it read at its start; it now patches `lastSyncedAt`).
+- A sync over a `config.json` that is not valid JSON used to throw at the stamp; the stamp is
+ now skipped (`readChannelConfig` already answered null, so the scheduler never picks such a
+ channel).
+- `resolveEffectiveAvailability`'s download-outcome side reads through `loadDownloadOutcome`,
+ so it now needs the full shape check. A read-only scan of all 57,897 live
+ `download-outcome.json` files: 1,120 carry an `availabilityClass`, **0** answer differently.
+- **Config writes now come out in SCHEMA key order.** Before, a form save or toggle wrote its
+ caller's spread order (form keys appended last); syncs already normalised it through
+ `updateConfigField`. Semantically nothing moves, but `transcripts/` is its own git repo:
+ the first form save or toggle per channel may show a one-time key-reorder diff there. The
+ numbers tool (parse → write) does not exercise this.
+
+**Review fixes (`973e59ee`).**
+- The jsonFile state (`tmpSeq`, write chains, file locks) lives on `globalThis.__yttJsonFile__`
+ — the house pattern (`jobs/registry.ts`, `controller/autoRunner.ts`) — because Next can load
+ the module once per bundle layer (instrumentation-armed runners vs server actions), and two
+ copies would each start the counter at 0 and hold separate locks. The temp name gains 4
+ random bytes (`${file}.tmp-${pid}-${seq}-${hex}`). A test imports a second module instance and
+ checks both share one chain.
+- storageLocations' re-point THROWS on a null patch, so the ledger rolls the link back instead
+ of recording `configWritten` with no `dataDir` on disk. relocateChannelMedia's move-out, on a
+ null patch, writes the job's own copy instead (deviation 10).
+- `writeChannelConfig` returns what it wrote, so the patch parses once; the site `z.object` is
+ built once at module load (`parseSite` fills `siteId`); a dead import removed.
+- SITE.md / CHANNEL.md prose corrected (the always-written site keys named; "absent means
+ inherit" only for the overrides; the two whole-config fallbacks named; the doubled
+ `downloadFilter`/`audioCheck` headings removed), with two new pinning tests.
+
+**Not every JSON writer is on the shared writer.** The 14 below are JSON writers still on the
+per-pid temp name `${file}.tmp-${process.pid}` — not "non-JSON", as a draft of this record
+said:
+`controller/failedTranscriptions.ts:33,54`, `controller/maybeMissingStore.ts:54`,
+`controller/rosterStore.ts:235`, `controller/duplicateShorts.ts:640,734`,
+`controller/scanCorruptMedia.ts:393,454`, `controller/shard.ts:56`,
+`controller/backupSavedVideos.ts:118`, `controller/relocateDir.ts:139`,
+`jobs/syncSchedulerState.ts:122`, `jobs/workerDefaults.ts:61`, `lib/widgetPresets.ts:83`,
+`lib/homepage.ts:129`, `bin/migrate-channel-priority.ts:206`.
+**`maybeMissingStore` and `rosterStore` are per-channel files written from several lanes — the
+same race class this slice fixes for `config.json`. Owed, later.** The non-JSON writers
+(normalizeTranscript/LiveChat cues, `runYtdlp` playlist, xSessionBroker, videoActions,
+cutReleaseAction, transcode, transcribeOne, savedVideo's copy, the umtool ones) are out of
+scope. The chain is per process: a CLI beside the live editor is not covered (the rename is
+still atomic).
+
+Also noted by review, accepted: `channelConfig.ts` and `channelGroups.ts` are value-imported by
+client modules, so the four channel/group `*_FIELD_DOCS` string records ship to the browser
+(a few KB, no zod — the grep below).
+
+**Numbers** (`plans/tools/phase3-files-numbers.ts`, one process, never writes the corpus). The
+inputs were FROZEN once (`FREEZE_TO`) into a scratch tree — 6 `site.json`, 71 `config.json`,
+and the first ≤300 dirs per sidecar filename in sorted slug × sorted id order: 1,763 sidecar
+files (attribution 259, diarization 300, availability 300, download-outcome 300,
+transcribe-outcome 300, do-not-clean 1, exclude-truncated-check 3, ai-digest 300,
+ai-digest.overrides 0 — none exist in the corpus) — because the live corpus moves under the
+running editor. The script prints each site parse + the file `writeSite(getSite())` writes,
+each config parse + md5 of `writeChannelConfig(readChannelConfig())`, and per sidecar md5 of
+the canonical load + md5 of `write(load())` (load only for the two markers, whose only writer
+stamps the clock). `main` (a detached worktree at `1e27f7c3`) vs the branch at `e75047f8`,
+`f6a08bd1` and — after the review fixes — `973e59ee` (`s4b-numbers-fix.txt`): **diff empty,
+3,832 lines**, each time. **Unknown-key report: empty** for all
+77 files. 0 null loads.
+
+Parity beyond the corpus (ad hoc, main's function vs the branch's): `parseSite` over 20,000
+random inputs — deepStrictEqual and key order equal; `writeSite` over 3,000 random sites —
+byte-identical files or the identical throw; `parseChannelConfig` over 20,000 random inputs —
+deepStrictEqual and key order equal.
+
+**`build:index` timing** (cold full build, scratch corpus as above; the build's own
+`Done in` figure — 77,624 transcripts, 6 sites built):
+
+| run | when | machine | time |
+|---|---|---|---|
+| before-main (`1e27f7c3`) #1 | 14:03–14:29 | I/O-loaded: the live editor's sync-all + metadata scan and an ffmpeg remux on the platter drive (`/proc/pressure/io` "some" 60–78 %) | 1542.74 s |
+| branch (`973e59ee`) | 14:29–14:57 | same load, partly | 1699.98 s |
+| before-main #2 | 15:54–16:22 | quiet | 1691.98 s |
+
+Branch vs before-main #2: **+0.5 %** — within the 5 % gate, **not a regression**; run #1 is the
+outlier (the load shifted which run it slowed; the numbers are not a controlled benchmark).
+No profiling was needed. Run #2 was taken by the coordinator with the same harness.
+
+**Gates on the branch tip before the merge (`973e59ee` + this commit):**
+
+| gate | result |
+|---|---|
+| `pnpm -r --workspace-concurrency=1 exec tsc --noEmit` | clean after every commit |
+| common tests | **1709/1709** (1663 before; +46: jsonFile 9, sidecar 8, siteSchema 11, channelConfigSchema 7, channels +5, fileSchemaDocs 6) |
+| editor unit | 67/67 |
+| `pnpm run test:scripts` | 156 pass / 1 skip |
+| mcp | 219/219 |
+| `next build` editor / export (at `7c03d7c9`) | green; `├ ƒ /api/view/[name]`; `grep -rl 'ZodError\|_zod'` over both `.next/static` prints nothing |
+| `file-schemas-docs --check` | clean |
+| numbers tool vs `main` | diff empty (above) |
+| e2e editor, the plan's 34 specs, at `7c03d7c9` | **190 passed**, 0 failed, 11.1 min, exit 0 — every spec the plan named exists; none dropped |
+| e2e editor after the review fixes (`973e59ee`): storage-locations, channel-storage, channel-rename, sites-crud, channels, digest, availability | **63 passed**, 0 failed, 3.7 min |
+| e2e export: site-branding, related-sites, pwa-search, ask-chat | **33 passed**, 0 failed, 1.0 min |
+
+**Merged with `main` `3241fed2` (slice 3a) as `7dfd7508`** — conflicts only in this file and
+`editor/CHANGELOG.md`, both sides kept, 3a first. Gates on the merged tip:
+
+| gate | result |
+|---|---|
+| `pnpm -r --workspace-concurrency=1 exec tsc --noEmit` | clean |
+| common tests | **1723/1723** (1709 + slice 3a's 14) |
+| editor unit | 67/67 |
+| `pnpm run test:scripts` | 156 pass / 1 skip |
+| mcp | 219/219 |
+| `next build` editor / export | both green; `├ ƒ /api/view/[name]`; `grep -rl 'ZodError\|_zod'` over both `.next/static` prints nothing |
+| numbers tool vs `main` (frozen inputs) | diff empty, 3,832 lines |
+| e2e editor, the plan's 34 specs | **190 passed**, 0 failed, 12.7 min, exit 0 |
+
+The export build first failed on the merged tip with `ENOENT export/public/tags.json`. Not this
+slice: the live editor's `build-deploy` job `01M3AADF600BDC99T2WPH1KBQB` (14:21–14:37, an
+incremental build:index on the real corpus plus a Jeralyzer deploy) regenerated the primary's
+`export/public/` without a `tags.json` (Jeralyzer has no curated tags), which left this
+worktree's per-path symlink to it dangling. The stale link was removed and the export build
+re-run green.
diff --git a/plans/tools/phase3-files-numbers.ts b/plans/tools/phase3-files-numbers.ts
@@ -0,0 +1,317 @@
+#!/usr/bin/env tsx
+// The one-core Phase 3 slice 4b measurement: what the site.json, channel
+// config.json and per-video sidecar readers ANSWER over the real corpus, and
+// what a write of that answer puts on disk — printed deterministically so a run
+// on `main` and a run on the branch can be diffed.
+//
+// Model: phase3-settings-numbers.ts next door, and the same rules.
+//
+// NEVER WRITES THE CORPUS. Every file is COPIED into a scratch tree under
+// os.tmpdir() first; every read-for-measurement and every write-back happens on
+// the copy, and the scratch tree is deleted at the end. The live corpus is only
+// ever read (readdir, readFile, copyFile source).
+//
+// NEVER BOOTS A SERVER. The readers are called in-process; instrumentation.ts
+// is not loaded, so nothing is armed.
+//
+// ONE PROCESS. `TRANSCRIPTS_DIR` is pointed at the scratch tree BEFORE any
+// module that memoises `getPaths()` is imported, and every reader used here
+// takes its location explicitly (a Paths, or a video dir).
+//
+// USES ONLY NAMES PRESENT ON BOTH SIDES of slice 4b (getSite, writeSite,
+// readChannelConfig, writeChannelConfig, the loadX/writeX sidecar functions),
+// so the SAME file runs on `main` and on the branch. The declared-key lists come
+// from the branch's docs records when they exist and from a literal otherwise;
+// the literal is checked against the records when both are present.
+//
+// Usage, from the repo root:
+// LIVE_TRANSCRIPTS_DIR=/abs/transcripts node_modules/.bin/tsx plans/tools/phase3-files-numbers.ts > out.txt
+// Default LIVE_TRANSCRIPTS_DIR: the primary checkout's, a sibling of this repo.
+// SAMPLE_N (default 300): sidecar dirs sampled per filename.
+//
+// FREEZING THE INPUTS. The live corpus moves while a slice is in flight (the
+// running editor stamps `lastSyncedAt`, writes outcomes), so a `main` run in the
+// morning and a branch run in the evening would differ for reasons that are not
+// this code. `FREEZE_TO=/abs/dir` copies exactly what a measurement reads —
+// every site.json, every config.json, and the sampled sidecar files, in the same
+// layout — into that directory and exits; both runs then take
+// `LIVE_TRANSCRIPTS_DIR=/abs/dir`, and a frozen tree samples itself.
+
+import { createHash } from "node:crypto";
+import fs from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const REPO = path.resolve(HERE, "..", "..");
+const LIVE =
+ process.env.LIVE_TRANSCRIPTS_DIR ??
+ path.join(path.dirname(REPO), "yt-dlp-transcript-browser", "transcripts");
+const SAMPLE_N = Number(process.env.SAMPLE_N ?? 300);
+
+const SCRATCH = fs.mkdtempSync(path.join(os.tmpdir(), "phase3-files-"));
+process.env.TRANSCRIPTS_DIR = SCRATCH;
+process.env.TZ = "UTC";
+
+// The keys each file may carry, as of 2026-09-24. On the branch these must
+// equal the docs records (asserted below).
+const SITE_KEYS_LITERAL = [
+ "siteId", "siteTitle", "siteDescription", "headerTitle", "homeTagline",
+ "accent", "socialLinks", "groups", "defaultGroupId", "channels",
+ "cloudflareProject", "siteUrl", "relatedSites", "pwa", "archives",
+ "archiveMaxBytes", "duplicates", "hubUrl",
+];
+const CHANNEL_KEYS_LITERAL = [
+ "handling", "sourceKind", "postFetcher", "socialHandle", "platform", "name",
+ "url", "audioFormat", "downloadFormat", "keepSourceVideo", "keepLatest",
+ "extractionMode", "savedVideosDir", "dataDir", "ytdlpExtraArgs", "subLangs",
+ "lastSyncedAt", "lastFullDownloadAt", "lastFullSweepAt", "excludeFromBuild",
+ "excludeFromCleanup", "syncIntervalMinutes", "fullSweepIntervalMinutes",
+ "skipLiveDownloads", "downloadFilter", "cookiesFromBrowser", "cookieMode",
+ "sleepBetweenDownloadsSeconds", "audioCheck",
+];
+
+function sortedKeys(_key: string, value: unknown): unknown {
+ if (!value || typeof value !== "object" || Array.isArray(value)) return value;
+ const src = value as Record<string, unknown>;
+ const out: Record<string, unknown> = {};
+ for (const k of Object.keys(src).sort()) out[k] = src[k];
+ return out;
+}
+
+function canonical(v: unknown): string {
+ // `undefined` members are dropped by JSON; that is also what a write drops.
+ return JSON.stringify(v, sortedKeys, 2) ?? "undefined";
+}
+
+function md5(text: string | Buffer): string {
+ return createHash("md5").update(text).digest("hex");
+}
+
+function sortedDir(dir: string): string[] {
+ try {
+ return fs.readdirSync(dir).sort();
+ } catch (e) {
+ return [];
+ }
+}
+
+function rawJson(file: string): unknown {
+ try {
+ return JSON.parse(fs.readFileSync(file, "utf8"));
+ } catch {
+ return undefined;
+ }
+}
+
+function rawKeys(raw: unknown): string[] {
+ return raw && typeof raw === "object" && !Array.isArray(raw)
+ ? Object.keys(raw as object)
+ : [];
+}
+
+async function declaredKeys(): Promise<{ site: string[]; channel: string[] }> {
+ const siteMod = (await import("../../common/lib/site")) as Record<string, unknown>;
+ const chMod = (await import("../../common/lib/channelConfig")) as Record<string, unknown>;
+ const siteDocs = siteMod.SITE_FIELD_DOCS as Record<string, string> | undefined;
+ const chKeys = chMod.CHANNEL_CONFIG_KEYS as readonly string[] | undefined;
+ const same = (a: readonly string[], b: readonly string[]) =>
+ [...a].sort().join() === [...b].sort().join();
+ if (siteDocs && !same(Object.keys(siteDocs), SITE_KEYS_LITERAL)) {
+ throw new Error("SITE_FIELD_DOCS disagrees with this script's literal list");
+ }
+ if (chKeys && !same(chKeys, CHANNEL_KEYS_LITERAL)) {
+ throw new Error("CHANNEL_CONFIG_KEYS disagrees with this script's literal list");
+ }
+ return { site: SITE_KEYS_LITERAL, channel: CHANNEL_KEYS_LITERAL };
+}
+
+async function sites(declared: string[]): Promise<void> {
+ const { getPaths } = await import("../../common/lib/paths");
+ const { getSite, writeSite, siteConfigFile } = await import("../../common/lib/site");
+ const paths = getPaths();
+ console.log("# site.json");
+ for (const id of sortedDir(path.join(LIVE, "sites"))) {
+ const live = path.join(LIVE, "sites", id, "site.json");
+ if (!fs.existsSync(live)) continue;
+ const file = siteConfigFile(paths, id);
+ fs.mkdirSync(path.dirname(file), { recursive: true });
+ fs.copyFileSync(live, file);
+ console.log(`## ${id}`);
+ const unknown = rawKeys(rawJson(file)).filter((k) => !declared.includes(k));
+ console.log(`unknown keys: ${JSON.stringify(unknown.sort())}`);
+ const site = getSite(id, paths);
+ console.log(canonical(site));
+ try {
+ await writeSite(site, paths);
+ console.log("### written by writeSite(getSite())");
+ console.log(fs.readFileSync(file, "utf8").trimEnd());
+ } catch (e) {
+ console.log(`WRITE THREW: ${(e as Error).message}`);
+ }
+ }
+}
+
+async function channels(declared: string[]): Promise<string[]> {
+ const { getPaths } = await import("../../common/lib/paths");
+ const { readChannelConfig, writeChannelConfig } = await import(
+ "../../common/controller/channels"
+ );
+ const paths = getPaths();
+ console.log("");
+ console.log("# channel config.json");
+ const slugs: string[] = [];
+ for (const slug of sortedDir(path.join(LIVE, "channels"))) {
+ const live = path.join(LIVE, "channels", slug, "config.json");
+ if (!fs.existsSync(live)) continue;
+ slugs.push(slug);
+ const file = path.join(paths.channelsDir, slug, "config.json");
+ fs.mkdirSync(path.dirname(file), { recursive: true });
+ fs.copyFileSync(live, file);
+ console.log(`## ${slug}`);
+ const unknown = rawKeys(rawJson(file)).filter((k) => !declared.includes(k));
+ console.log(`unknown keys: ${JSON.stringify(unknown.sort())}`);
+ const config = await readChannelConfig(paths, slug);
+ console.log(canonical(config));
+ if (!config) continue;
+ try {
+ await writeChannelConfig(paths, slug, config);
+ console.log(`written md5: ${md5(fs.readFileSync(file))}`);
+ } catch (e) {
+ console.log(`WRITE THREW: ${(e as Error).message}`);
+ }
+ }
+ return slugs;
+}
+
+type Pair = {
+ filename: string;
+ load: (dir: string) => Promise<unknown>;
+ // Absent for the two markers whose only writer stamps the clock.
+ write?: (dir: string, value: never) => Promise<void>;
+};
+
+async function sidecarPairs(): Promise<Pair[]> {
+ const attribution = await import("../../common/lib/attribution-server");
+ const diarization = await import("../../common/lib/diarization-server");
+ const availability = await import("../../common/lib/availability-server");
+ const dlo = await import("../../common/lib/downloadOutcome-server");
+ const tro = await import("../../common/lib/transcribeOutcome-server");
+ const dnc = await import("../../common/lib/doNotClean-server");
+ const etc = await import("../../common/lib/excludeTruncatedCheck-server");
+ const digest = await import("../../common/lib/digest-server");
+ const { ATTRIBUTION_FILENAME } = await import("../../common/lib/attribution");
+ const { DIARIZATION_FILENAME } = await import("../../common/lib/diarization");
+ const { AVAILABILITY_FILENAME } = await import("../../common/lib/availability");
+ const { DOWNLOAD_OUTCOME_FILENAME } = await import("../../common/lib/downloadOutcome");
+ const { TRANSCRIBE_OUTCOME_FILENAME } = await import("../../common/lib/transcribeOutcome");
+ const { DO_NOT_CLEAN_FILENAME } = await import("../../common/lib/doNotClean");
+ const { EXCLUDE_TRUNCATED_CHECK_FILENAME } = await import(
+ "../../common/lib/excludeTruncatedCheck"
+ );
+ const { DIGEST_FILENAME, DIGEST_OVERRIDES_FILENAME } = await import(
+ "../../common/lib/digest"
+ );
+ return [
+ { filename: ATTRIBUTION_FILENAME, load: attribution.loadAttribution, write: attribution.writeAttribution },
+ { filename: DIARIZATION_FILENAME, load: diarization.loadDiarization, write: diarization.writeDiarization },
+ { filename: AVAILABILITY_FILENAME, load: availability.loadAvailability, write: availability.writeAvailability },
+ { filename: DOWNLOAD_OUTCOME_FILENAME, load: dlo.loadDownloadOutcome, write: dlo.writeDownloadOutcome },
+ { filename: TRANSCRIBE_OUTCOME_FILENAME, load: tro.loadTranscribeOutcome, write: tro.writeTranscribeOutcome },
+ { filename: DO_NOT_CLEAN_FILENAME, load: dnc.loadDoNotClean },
+ { filename: EXCLUDE_TRUNCATED_CHECK_FILENAME, load: etc.loadExcludeTruncatedCheck },
+ { filename: DIGEST_FILENAME, load: digest.loadDigest, write: digest.writeDigest },
+ { filename: DIGEST_OVERRIDES_FILENAME, load: digest.loadDigestOverrides, write: digest.writeDigestOverrides },
+ ] as Pair[];
+}
+
+// A deterministic sample: channels in sorted order, video ids in sorted order,
+// the first SAMPLE_N dirs holding each filename.
+function sample(slugs: string[], filenames: string[]): Map<string, string[]> {
+ const out = new Map<string, string[]>(filenames.map((f) => [f, []]));
+ for (const slug of slugs) {
+ const data = path.join(LIVE, "channels", slug, "data");
+ for (const id of sortedDir(data)) {
+ const dir = path.join(data, id);
+ const present = new Set(sortedDir(dir));
+ for (const f of filenames) {
+ const list = out.get(f)!;
+ if (list.length < SAMPLE_N && present.has(f)) list.push(dir);
+ }
+ }
+ if ([...out.values()].every((l) => l.length >= SAMPLE_N)) break;
+ }
+ return out;
+}
+
+async function sidecars(slugs: string[]): Promise<void> {
+ const pairs = await sidecarPairs();
+ const picked = sample(slugs, pairs.map((p) => p.filename));
+ console.log("");
+ console.log(`# sidecars (first ${SAMPLE_N} per filename, sorted slugs × sorted ids)`);
+ let n = 0;
+ for (const pair of pairs) {
+ const dirs = picked.get(pair.filename)!;
+ console.log(`## ${pair.filename} — ${dirs.length} dirs`);
+ let nulls = 0;
+ for (const live of dirs) {
+ const rel = path.relative(path.join(LIVE, "channels"), live);
+ const dir = path.join(SCRATCH, "sidecars", String(n++));
+ fs.mkdirSync(dir, { recursive: true });
+ fs.copyFileSync(path.join(live, pair.filename), path.join(dir, pair.filename));
+ const loaded = await pair.load(dir);
+ if (loaded === null) nulls++;
+ let written = "-";
+ if (loaded !== null && pair.write) {
+ await pair.write(dir, loaded as never);
+ const file = path.join(dir, pair.filename);
+ written = fs.existsSync(file) ? md5(fs.readFileSync(file)) : "removed";
+ }
+ console.log(`${rel} load=${md5(canonical(loaded))} write=${written}`);
+ }
+ console.log(`null loads: ${nulls}`);
+ }
+}
+
+async function freeze(to: string): Promise<void> {
+ const copy = (rel: string) => {
+ const dest = path.join(to, rel);
+ fs.mkdirSync(path.dirname(dest), { recursive: true });
+ fs.copyFileSync(path.join(LIVE, rel), dest);
+ };
+ for (const id of sortedDir(path.join(LIVE, "sites"))) {
+ const rel = path.join("sites", id, "site.json");
+ if (fs.existsSync(path.join(LIVE, rel))) copy(rel);
+ }
+ const slugs: string[] = [];
+ for (const slug of sortedDir(path.join(LIVE, "channels"))) {
+ const rel = path.join("channels", slug, "config.json");
+ if (!fs.existsSync(path.join(LIVE, rel))) continue;
+ slugs.push(slug);
+ copy(rel);
+ }
+ const pairs = await sidecarPairs();
+ const picked = sample(slugs, pairs.map((p) => p.filename));
+ let files = 0;
+ for (const [filename, dirs] of picked) {
+ for (const dir of dirs) {
+ copy(path.relative(LIVE, path.join(dir, filename)));
+ files++;
+ }
+ }
+ console.error(`froze ${slugs.length} configs and ${files} sidecar files into ${to}`);
+}
+
+try {
+ if (process.env.FREEZE_TO) {
+ await freeze(path.resolve(process.env.FREEZE_TO));
+ process.exit(0);
+ }
+ const declared = await declaredKeys();
+ await sites(declared.site);
+ const slugs = await channels(declared.channel);
+ await sidecars(slugs);
+} finally {
+ fs.rmSync(SCRATCH, { recursive: true, force: true });
+}