commit 050c6202ad6be34834ae0f78f84c455a82933565
parent 8d6afff1b19e317844012eeffb8dd917fa27099a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 28 Sep 2026 01:48:51 -0400
common: every environment variable, declared once; ENVIRONMENT.md generated
`common/lib/envVars.ts` declares each variable the apps' code reads, by
audience: paths (the getPaths() override surface), runtime, port (generated
from ports.mjs), internal (set by the pipeline for a process it spawns),
docker (the container's ARCHILYZER_* set) and test. ENVIRONMENT.md is
generated from it (`archilyzer docs env [--check]`, `bin/env-docs.ts`).
`envVars.test.ts` holds the list to the code in both directions: every
`process.env.X` / `env.X` read under common/, editor/, export/, homepage/,
mcp/src and scripts/ is declared, every declared name is still mentioned by
the code, docker/ or a compose file, the paths audience is exactly what
getPaths() reads, and the committed ENVIRONMENT.md is fresh. umtool's own
knobs stay in umtool/docs until Phase 5.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
5 files changed, 575 insertions(+), 0 deletions(-)
diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md
@@ -0,0 +1,181 @@
+# Environment variables
+
+<!-- GENERATED by common/bin/env-docs.ts from common/lib/envVars.ts — do not edit by hand. -->
+
+Every environment variable the repo's code reads, by who it is for. The list is code (`common/lib/envVars.ts`), and a test fails when the code reads a variable the list does not declare, or the list declares one nothing reads. umtool's own knobs are documented in [umtool/docs](umtool/docs/README.md).
+
+Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/archilyzer.ts docs env`. `archilyzer doctor` prints which of the paths overrides are set on this machine.
+
+## Paths and binaries
+
+The one override surface for where things live and which binary runs. Every one is read by `getPaths()` (`common/lib/paths.ts`) and nowhere else; nothing hardcodes a location.
+
+| Variable | Default | What it does | Read by |
+|---|---|---|---|
+| `TRANSCRIPTS_DIR` | `<repo>/transcripts` | The corpus: channels, sites, the LMDB index, job logs, the saved-video store. | common/lib/paths.ts (getPaths) |
+| `SAVED_VIDEOS_DIR` | `<TRANSCRIPTS_DIR>/saved-videos` | The persisted source-video store, when it should live on another disk. | common/lib/paths.ts (getPaths) |
+| `SITES_DIR` | `<TRANSCRIPTS_DIR>/sites` | Per-site config (`<id>/site.json`, every key in [SITE.md](SITE.md)) and the homepage's `_homepage/`. | common/lib/paths.ts (getPaths) |
+| `SETTINGS_FILE` | `<repo>/settings.json` | The settings file (every key in [SETTINGS.md](SETTINGS.md)). | common/lib/paths.ts (getPaths) |
+| `EXPORT_PUBLIC_DIR` | `<repo>/export/public` | The dir the export site serves at `/`, composed one site at a time. | common/lib/paths.ts (getPaths) |
+| `EXPORT_INDEX_DIR` | `.export-index` beside `EXPORT_PUBLIC_DIR` | The build's staging area (not served): the shared index and per-site aggregates. | common/lib/paths.ts (getPaths) |
+| `EXPORT_BUILDS_DIR` | `.export-builds` beside `EXPORT_PUBLIC_DIR` | Per-site `out/` bundles from a docker-mode build. | common/lib/paths.ts (getPaths) |
+| `EDITOR_CHANGELOG_FILE` | `<repo>/editor/CHANGELOG.md` | The editor changelog the release cutter reads and rewrites. The e2e server points it at a gitignored copy. | common/lib/paths.ts (getPaths) |
+| `EXPORT_CHANGELOG_FILE` | `<repo>/export/CHANGELOG.md` | The export changelog, likewise. | common/lib/paths.ts (getPaths) |
+| `CHARTS_CONFIG_FILE` | `<repo>/chart-templates.json` | The legacy chart-templates file, read only by a migration. | common/lib/paths.ts (getPaths) |
+| `SEARCH_ALIASES_FILE` | `<TRANSCRIPTS_DIR>/search-aliases.json` | The corpus-wide search-alias dictionary. | common/lib/paths.ts (getPaths) |
+| `CURATED_TAGS_FILE` | `<TRANSCRIPTS_DIR>/tags.json` | Curated per-video tags. Written only through `applyTagAssignments`. | common/lib/paths.ts (getPaths) |
+| `YTDLP_BIN` | `yt-dlp` on PATH | The downloader. Every fetch goes through it. | common/lib/paths.ts (getPaths) |
+| `FFMPEG_BIN` | `ffmpeg` on PATH | Audio extraction for transcription and diarization. | common/lib/paths.ts (getPaths) |
+| `FFPROBE_BIN` | `ffprobe` on PATH | Duration checks (the short-audio guard, windowing). | common/lib/paths.ts (getPaths) |
+| `WHISPER_BIN` | `whisper-cli` on PATH | whisper.cpp, one of the three transcription engines (with chough and parakeet.cpp). | common/lib/paths.ts (getPaths) |
+| `WHISPER_MODEL` | `~/whispercpp/whisper.cpp/models/ggml-base.en.bin` | whisper.cpp's model, when a worker names none. | common/lib/paths.ts (getPaths) |
+| `PARAKEET_STITCH_BIN` | `<repo>/scripts/parakeet-stitch.mjs` | The parakeet.cpp engine's wrapper (overlapping windows, stitched). | common/lib/paths.ts (getPaths) |
+| `PARAKEET_CLI` | `parakeet-cli` on PATH | The parakeet.cpp binary the wrapper drives (the wrapper reads it too). | common/lib/paths.ts (getPaths) |
+| `PARAKEET_MODEL` | none | parakeet.cpp's `.gguf`, when a worker names none (the wrapper reads it too). | common/lib/paths.ts (getPaths) |
+| `DIARIZE_BIN` | `<repo>/scripts/diarize.mjs` | The speaker-diarization wrapper. The e2e suite swaps in a fake here. | common/lib/paths.ts (getPaths) |
+| `RSYNC_BIN` | `rsync` on PATH | Mirrors the saved-video store to a backup destination. | common/lib/paths.ts (getPaths) |
+| `FINDMNT_BIN` | `findmnt` on PATH | The read-only volume-identity probe behind storage locations. Optional. | common/lib/paths.ts (getPaths) |
+| `UDISKSCTL_BIN` | `udisksctl` on PATH | Mounts an attached volume from `/storage`. Optional. | common/lib/paths.ts (getPaths) |
+| `GALLERY_DL_BIN` | `gallery-dl` on PATH | The X/Twitter post fetcher, for social channels. | common/lib/paths.ts (getPaths) |
+| `OLLAMA_URL` | `http://127.0.0.1:11434` | The local ollama server, the local digest and attribution engine. | common/lib/paths.ts (getPaths) |
+| `CLAUDE_BIN` | `claude` on PATH | The `claude` CLI, driving the opt-in metered digest lane. | common/lib/paths.ts (getPaths) |
+
+## Runtime
+
+Tokens, credentials and knobs a running process reads. Most configuration is not here but in `settings.json` ([SETTINGS.md](SETTINGS.md)).
+
+| Variable | Default | What it does | Read by |
+|---|---|---|---|
+| `WORKER_TOKEN` | unset (both surfaces off) | Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`, the MCP's `fetch_clip`). Set the same value on both ends. | common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts |
+| `SYNC_HEARTBEAT_SECONDS` | `settings.syncScheduler.heartbeatSeconds` | Overrides the editor's in-process sync heartbeat. `0` = no internal timer (tick from cron instead). | editor/app/scheduler/heartbeat.ts |
+| `SYNC_TICK_URL` | `http://127.0.0.1:3001/api/scheduler/tick` | Where `archilyzer sync tick` (cron's heartbeat) posts. | common/bin/sync-tick.ts |
+| `SYNC_TICK_TOKEN` | unset (no auth) | Bearer token for the tick endpoint; set on both the editor and the cron job. | common/bin/sync-tick.ts, editor/app/scheduler/auth.ts |
+| `R2_ACCESS_KEY_ID` | — | R2 S3 credentials for uploading oversize archives at deploy time (with `R2_SECRET_ACCESS_KEY` and `CLOUDFLARE_ACCOUNT_ID`). See [PUBLISH.md](PUBLISH.md). | common/publish/build.ts |
+| `R2_SECRET_ACCESS_KEY` | — | See `R2_ACCESS_KEY_ID`. | common/publish/build.ts |
+| `CLOUDFLARE_ACCOUNT_ID` | — | The account the R2 endpoint belongs to. wrangler reads its own credentials. | common/publish/build.ts |
+| `DOCKER_BIN` | `docker` | The container engine for docker-mode builds (e.g. `podman`). | common/publish/build.ts |
+| `DOCKER_BUILD_MEMORY` | no cap | Per-container memory cap for a docker-mode build (`--memory`). | common/publish/build.ts |
+| `DOCKER_BUILD_CPUS` | no cap | Per-container CPU cap for a docker-mode build (`--cpus`). | common/publish/build.ts |
+| `ARCHIVE_CHANNEL_CONCURRENCY` | `4` | How many channels' archive zips `build archives` builds at once. | common/bin/build-archives.ts |
+| `MAX_ARCHIVE_BYTES` | the Cloudflare-safe cap | The served-file size cap for archives, in bytes; `0` = no cap. A site's own `archiveMaxBytes` wins. | common/bin/compose-site.ts |
+| `CHOUGH_BIN` | `chough` on PATH | The chough transcription engine, when a worker names no binary. | common/lib/transcriptionApps.ts |
+| `CHOUGH_MODEL` | chough's own | Passed to chough from a worker's model field; chough auto-downloads one when unset. | chough (set by common/lib/transcriptionApps.ts) |
+| `CHOUGH_URL` | local | Passed to chough from a worker's remote-server field. | chough (set by common/lib/transcriptionApps.ts) |
+| `OLLAMA_DIGEST_MODEL` | `qwen2.5:7b` | The ollama model the local digest lane asks for when settings name none. | common/lib/digestApps.ts |
+| `CLAUDE_DIGEST_MODEL` | the CLI's default | The model the metered digest lane asks `claude` for when settings name none. | common/lib/digestApps.ts |
+| `NITTER_INSTANCES` | a built-in list | Comma-separated Nitter instances for the X fallback fetcher, in order of preference. | common/social/xNitterFetcher.ts |
+| `UMTOOL_URL` | unset (no link) | umtool's front door; when set, the video page links to it. | editor/app/channels/[slug]/videos/[id]/page.tsx |
+| `TRANSCRIPT_SITE_URL` | — | MCP server: one published archive to read over HTTP. | mcp/src/sources.ts |
+| `TRANSCRIPT_HUB_URL` | — | MCP server: a hub, federating every archive it lists. | mcp/src/sources.ts |
+| `TRANSCRIPT_LOCAL_DIR` | — | MCP server: a composed public dir on disk. | mcp/src/sources.ts |
+| `TRANSCRIPT_PLATFORM_LINKS` | off | `1` cites platform watch pages instead of the archive's own pages. | common/lib/archive/reader-fs.ts |
+| `MCP_IO_STATS` | off | `1` turns on per-call I/O accounting, for `mcp/bench`. | common/lib/archive/io-stats.ts |
+| `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's `fetch_clip` talk to. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool |
+| `ARCHILYZER_AGENT` | `cli` | Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`. | scripts/archilyzer-ops.mjs |
+| `DIARIZE_ENGINE_KIND` | `sherpa-onnx` | The diarization engine: `sherpa-onnx` or `sortformer`. | scripts/diarize.mjs |
+| `DIARIZE_ENGINE_CMD` | the bundled sherpa script | The engine command the wrapper runs. | scripts/diarize.mjs |
+| `DIARIZE_PYTHON` | `python3` | The python for the default engine. | scripts/diarize.mjs |
+| `DIARIZE_SEG_MODEL` | — (required) | Segmentation model. The editor passes the settings' value as a flag. | scripts/diarize.mjs |
+| `DIARIZE_EMB_MODEL` | — (required) | Speaker-embedding model. The editor passes the settings' value as a flag. | scripts/diarize.mjs |
+| `DIARIZE_THRESHOLD` | `0.5` | Clustering threshold. | scripts/diarize.mjs |
+| `DIARIZE_THREADS` | `4` | Engine threads. | scripts/diarize.mjs |
+| `DIARIZE_WINDOW_MINUTES` | `45` | Window length for long files; `0` never windows. | scripts/diarize.mjs |
+| `DIARIZE_WINDOW_AFTER_MINUTES` | `90` | Only files longer than this are windowed. | scripts/diarize.mjs |
+| `SORTFORMER_BIN` | — (required for sortformer) | The sortformer engine binary. | scripts/diarize.mjs, scripts/diarize-sortformer.mjs |
+| `SORTFORMER_MODEL` | — (required for sortformer) | The sortformer `.gguf`. | scripts/diarize.mjs, scripts/diarize-sortformer.mjs |
+| `PARAKEET_SEGMENT_SEC` | `480` | parakeet window length, seconds (a worker's chunk size wins). | scripts/parakeet-stitch.mjs |
+| `PARAKEET_OVERLAP_SEC` | `6` | parakeet window overlap, seconds. | scripts/parakeet-stitch.mjs |
+| `PARAKEET_DECODER` | parakeet-cli's | `ctc` or `tdt`, passed through to parakeet-cli. | scripts/parakeet-stitch.mjs |
+| `PARAKEET_LANG` | parakeet-cli's | A locale, passed through to parakeet-cli. | scripts/parakeet-stitch.mjs |
+| `PARAKEET_DEVICE` | parakeet-cli's | Compute device (`cpu`, `CUDA0`, `Vulkan1`, …), exported to parakeet-cli. | scripts/parakeet-stitch.mjs |
+
+## Ports
+
+Every local server's default port, from `common/lib/ports.mjs`. The primary checkout uses these; worktree N adds N × 100 (`pnpm wt list`).
+
+| Variable | Default | What it does | Read by |
+|---|---|---|---|
+| `EDITOR_PORT` | `3001` | Editor real dev/start. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `PORT` | `3011` | Editor test server + Playwright editor baseURL. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `EXPORT_PORT` | `3010` | Export server launched by the editor e2e. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `EXPORT_DEV_PORT` | `3000` | Export real dev. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `EXPORT_E2E_PORT` | `3020` | Export's own Playwright suite. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `OLLAMA_STUB_PORT` | `11435` | Digest-lane stub server in the editor e2e suite. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `HOMEPAGE_DEV_PORT` | `3030` | Homepage real dev. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `HOMEPAGE_PORT` | `3031` | Homepage static `serve out` (start:homepage). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `HOMEPAGE_E2E_PORT` | `3040` | Homepage's own Playwright suite. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `HUB_PORT` | `3041` | Export's hub Playwright suite (e2e:hub). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `UMTOOL_PORT` | `3050` | Umtool real dev/start. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `UMTOOL_E2E_PORT` | `3051` | Umtool's own Playwright suite. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `EDITOR_STUB_PORT` | `3052` | Stub editor the umtool e2e suite fetches clips from. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `ORIGIN_B_PORT` | `4610` | Export's two-origin suite: the member site (e2e:2origin). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `HUB_A_PORT` | `4611` | Export's two-origin suite: the hub (e2e:2origin). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+
+## Set by the pipeline
+
+The publish pipeline sets these for a process it spawns. Listed so a reader knows what they are; nobody sets them by hand.
+
+| Variable | Default | What it does | Read by |
+|---|---|---|---|
+| `SITE_ID` | — | Which site a compose or an export build is for. `archilyzer build site <id>` sets it; `compose site` and `build site` fall back to it when no id is given. | common/bin/compose-site.ts, export/app/lib/site.ts |
+| `INSTANCE_MODE` | a site | `hub` makes the export build the hub. Set by `archilyzer build hub`. | export/app/lib/mode.ts, common/lib/archive/contract.ts |
+| `BUILD_ARCHIVES` | on | `0` skips archive-zip generation for one build (`--skip-archives`). | common/bin/compose-site.ts, common/bin/build-archives.ts |
+| `ARCHIVES_READONLY` | off | `1` inside a docker-mode build container: materialize archives, never write the shared cache. | common/bin/compose-site.ts |
+| `HOMEPAGE_PUBLIC_DIR` | `<repo>/homepage/public` | Where `compose homepage` writes. | common/bin/compose-homepage.ts |
+| `HOMEPAGE_SUMMARY_FILE` | `homepage/public/homepage-summary.json` | A dev-only summary file for the homepage; ignored by a production build. The homepage e2e sets it. | homepage/app/lib/summary.ts |
+
+## Docker
+
+The container's own set, read by `docker/*.sh`, the compose files and Caddy — not by the apps' code (except `ARCHILYZER_IDLE_BOOT`). See [RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md).
+
+| Variable | Default | What it does | Read by |
+|---|---|---|---|
+| `ARCHILYZER_TRANSCRIBER` | baked per image target (`whisper-cpp` in `runtime`) | `whisper-cpp` or `parakeet`: which worker the first boot seeds and which model it fetches. | docker/entrypoint.sh |
+| `ARCHILYZER_FETCH_MODEL` | per transcriber | Which model the first boot downloads; `none` skips it. | docker/entrypoint.sh |
+| `ARCHILYZER_MODELS_DIR` | `/data/models` | Where models live in the container. | docker/entrypoint.sh |
+| `ARCHILYZER_BUILDS_DIR` | `/data/builds` | Where the container keeps built sites. | docker/entrypoint.sh |
+| `ARCHILYZER_SITE_OUT` | `/data/builds/site` | The built export site the `site` service serves. | docker/entrypoint.sh, docker/publish-site.sh |
+| `ARCHILYZER_IDLE_BOOT` | off | `1` boots the editor without arming the heartbeat or any auto-queue runner. | common/lib/idleBoot.ts (the editor) |
+| `ARCHILYZER_AUTH_MODE` | `basic` | `basic`, `forward` or `none` — the only escape hatch from the exposure guard. | docker/guard-exposure.sh, docker/caddy-start.sh |
+| `ARCHILYZER_AUTH_USER` | `archilyzer` | Basic-auth user. | docker/Caddyfile |
+| `ARCHILYZER_AUTH_HASH` | — | Basic-auth bcrypt hash (`caddy hash-password`). | docker/Caddyfile, docker/guard-exposure.sh |
+| `ARCHILYZER_AUTH_IMPORT` | derived from the mode | Set by docker/caddy-start.sh from the mode: which auth snippet the private sites import. | docker/Caddyfile |
+| `ARCHILYZER_FORWARD_AUTH_UPSTREAM` | — | Forward-auth server (Authelia, tinyauth, …), `host:port`. | docker/Caddyfile |
+| `ARCHILYZER_FORWARD_AUTH_URI` | `/api/auth/caddy` | The forward-auth server's verify path. | docker/Caddyfile |
+| `ARCHILYZER_TAG` | `local` | The image tag the compose files build and run. | docker-compose*.yml |
+
+## Tests only
+
+Read only by a test harness, a fake binary or a test-mode branch. Never set one on a real instance.
+
+| Variable | Default | What it does | Read by |
+|---|---|---|---|
+| `EDITOR_TEST_ROUTES` | off | `1` opens the editor's `/api/test/*` routes. The e2e server sets it. | editor/app/api/test/_guard.ts, editor/instrumentation.ts |
+| `E2E_MODE` | dev | `start` runs the editor suite against `next start` instead of `next dev`. | editor/playwright.config.ts |
+| `E2E_QUEUE` | on | `0` skips the machine-global e2e queue (the port check still runs). | scripts/queue-lock.mjs |
+| `E2E_PORT_CHECK` | on | `0` skips the pre-run check that the suite's ports are free. | scripts/queue-lock.mjs |
+| `E2E_QUEUE_TIMEOUT` | wait forever | Seconds to wait for the queue before giving up. | scripts/queue-lock.mjs |
+| `E2E_PORT_GRACE_MS` | `3000` | How long the port check waits for a just-freed port. | scripts/queue-lock.mjs |
+| `E2E_QUEUE_LOCK_FILE` | one per machine | The queue's lock file; the queue's own tests point it elsewhere. | scripts/queue-lock.mjs |
+| `QUEUE_LOCK_HELD` | — | Set by the queue for the command it runs, so a nested wrapper passes through. | scripts/queue-lock.mjs |
+| `PLAYWRIGHT_BASE_URL` | `http://localhost:<PORT>` | The editor test server's URL; the worktree injector sets it. | editor/playwright.config.ts, editor/e2e/baseUrl.ts |
+| `AUDIO_CHECK_INTERVAL_MS_OVERRIDE` | the real cadence | Shrinks the mid-download audio check so the e2e suite sees it fire. | common/ytdlp/audioCheckedDownload.ts |
+| `AUDIO_CHECK_SIZE_GATE_OVERRIDE` | the real gate | Likewise, the size gate. | common/ytdlp/audioCheckedDownload.ts |
+| `AUDIO_CHECK_INTERVAL_FLOOR_MS_OVERRIDE` | the real floor | Likewise, the interval floor. | common/ytdlp/audioCheckedDownload.ts |
+| `AUDIO_CHECK_RECOVER_STEP_MS_OVERRIDE` | the real step | Likewise, the recovery step. | common/ytdlp/audioCheckedDownload.ts |
+| `AUDIO_CHECK_RECOVER_AFTER_OVERRIDE` | the real count | Likewise, the recovery count. | common/ytdlp/audioCheckedDownload.ts |
+| `AUDIO_CHECK_DEBUG_PAUSE_MS` | off | A debugging pause inside the audio check. | common/ytdlp/audioCheckedDownload.ts |
+| `FAKE_YTDLP_AUDIO_CHECK_MODE` | — | Fake yt-dlp: which audio-check scenario to act out. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_YTDLP_CHUNK_DELAY_MS` | — | Fake yt-dlp: delay between written chunks. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_YTDLP_CORRUPT_AFTER_CHUNK` | — | Fake yt-dlp: start corrupting after this chunk. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_YTDLP_CORRUPT_RUNS` | — | Fake yt-dlp: how many runs corrupt. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_YTDLP_DETERMINISTIC_CORRUPT` | — | Fake yt-dlp: corrupt deterministically. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_YTDLP_RECOVER_ON_RESUME` | — | Fake yt-dlp: a resumed run recovers. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_YTDLP_TOTAL_CHUNKS` | — | Fake yt-dlp: how many chunks a download has. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_GALLERY_DL_AUTH_FAIL` | — | Fake gallery-dl: fail as an auth error. | editor/e2e/fixtures/bin/fake-gallery-dl.mjs |
+| `FIXTURE_MAX_LIFETIME_MS` | the watchdog's | How long a fake binary may live before its watchdog kills it. | editor/e2e/fixtures/bin/_watchdog.mjs |
+| `OLLAMA_STUB_MODEL` | `qwen2.5:7b` | The model the ollama stub claims to serve. | editor/e2e/fixtures/ollama-stub.mjs |
+| `RACK_SHOTS` | off (spec skipped) | Runs the `/channels` rack screenshot audit. | editor/e2e/channels-rack-audit.spec.ts |
+| `TWO_ORIGIN_REBUILD` | off | `1` rebuilds the two-origin suite's cached hub bundle. | export/e2e-2origin/globalSetup.ts |
+| `IMAGE` | `yt-dlp-transcript-browser-e2e` | The sharded e2e run's image tag. | scripts/run-sharded-e2e.mjs |
+| `SKIP_BUILD` | off | `1` reuses the sharded e2e image instead of rebuilding it (`--no-build`). | scripts/run-sharded-e2e.mjs |
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -177,6 +177,13 @@ export const COMMANDS: Command[] = [
run: async () => (await import("./sync-tick")).tick(),
},
{
+ path: ["docs", "env"],
+ usage: "[--check] write ENVIRONMENT.md from the declared env-var list (lib/envVars.ts)",
+ flags: { check: "boolean" },
+ run: async ({ flags }) =>
+ (await import("./env-docs")).main({ check: flags.check === true }),
+ },
+ {
path: ["settings", "example"],
usage: "[--check] write settings.json.example + SETTINGS.md from the schema",
flags: { check: "boolean" },
diff --git a/common/bin/env-docs.ts b/common/bin/env-docs.ts
@@ -0,0 +1,34 @@
+// WRITE ENVIRONMENT.MD FROM THE DECLARED LIST (lib/envVars.ts).
+//
+// archilyzer docs env [--check]
+//
+// `--check` writes nothing and returns 1 when the committed file differs from
+// what the list generates (the same claim lib/envVars.test.ts makes). The
+// sibling of file-schemas-docs.ts (SITE.md, CHANNEL.md) and settings-example.ts
+// (SETTINGS.md).
+
+import { readFile, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { renderEnvironmentMarkdown } from "../lib/envVars";
+import { runIfEntryPoint } from "./_cli";
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+export async function main(opts: { check?: boolean } = {}): Promise<number> {
+ const file = path.join(REPO, "ENVIRONMENT.md");
+ const want = renderEnvironmentMarkdown();
+ if (opts.check) {
+ const have = await readFile(file, "utf8").catch(() => "");
+ if (have !== want) {
+ console.error("ENVIRONMENT.md is stale — regenerate it with `archilyzer docs env`");
+ return 1;
+ }
+ return 0;
+ }
+ await writeFile(file, want);
+ console.log("wrote ENVIRONMENT.md");
+ return 0;
+}
+
+runIfEntryPoint(import.meta.url, () => main({ check: process.argv.includes("--check") }));
diff --git a/common/lib/envVars.test.ts b/common/lib/envVars.test.ts
@@ -0,0 +1,128 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { readdirSync, readFileSync, statSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { ENV_AUDIENCES, ENV_VARS, renderEnvironmentMarkdown } from "./envVars";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common test
+//
+// envVars.ts is the one declared list of environment variables, and
+// ENVIRONMENT.md is generated from it. These tests are what keep the list true:
+// the code is read as text, in both directions.
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+// Where the apps' code lives. umtool is out of scope on purpose (envVars.ts
+// says why); tests are skipped because a test SETS variables for itself.
+const CODE_ROOTS = ["common", "editor", "export", "homepage", "mcp/src", "scripts"];
+const SKIP_DIRS = new Set(["node_modules", ".next", "out", "public", "test-results", "blob-report", "test-transcripts"]);
+
+// The platform's own variables: read here, documented by node, Next, a shell.
+const PLATFORM = new Set(["CI", "NODE_ENV", "NEXT_RUNTIME", "LD_LIBRARY_PATH", "PATH", "HOME", "NODE_OPTIONS"]);
+
+function codeFiles(): string[] {
+ const out: string[] = [];
+ const walk = (dir: string) => {
+ for (const n of readdirSync(dir)) {
+ if (SKIP_DIRS.has(n) || n.startsWith(".")) continue;
+ const p = path.join(dir, n);
+ if (statSync(p).isDirectory()) walk(p);
+ else if (/\.(ts|tsx|mts|mjs|js)$/.test(n) && !/\.test\.(ts|mjs)$/.test(n)) out.push(p);
+ }
+ };
+ for (const r of CODE_ROOTS) walk(path.join(REPO, r));
+ return out;
+}
+
+// Comment lines are dropped first: prose that names `process.env.NAME` as an
+// example is not a read.
+function codeText(file: string): string {
+ return readFileSync(file, "utf8")
+ .split("\n")
+ .filter((l) => !/^\s*(\/\/|\*|\/\*)/.test(l))
+ .join("\n");
+}
+
+const READ_PATTERNS = [
+ /process\.env\??\.([A-Z][A-Z0-9_]+)\b/g,
+ /process\.env\[["']([A-Z][A-Z0-9_]+)["']\]/g,
+ // `env.X` on an env object handed in (a spawn's env, a testable `env =
+ // process.env` parameter).
+ /\b[eE]nv\??\.([A-Z][A-Z0-9_]{2,})\b/g,
+ // audioCheckedDownload.ts reads its test overrides through a helper.
+ /envIntOverride\(["']([A-Z][A-Z0-9_]+)["']\)/g,
+];
+
+function reads(): Map<string, Set<string>> {
+ const byName = new Map<string, Set<string>>();
+ for (const file of codeFiles()) {
+ const text = codeText(file);
+ for (const re of READ_PATTERNS) {
+ for (const m of text.matchAll(re)) {
+ const name = m[1];
+ if (PLATFORM.has(name)) continue;
+ if (!byName.has(name)) byName.set(name, new Set());
+ byName.get(name)!.add(path.relative(REPO, file));
+ }
+ }
+ }
+ return byName;
+}
+
+test("ENVIRONMENT.md is what envVars.ts generates", () => {
+ assert.equal(
+ readFileSync(path.join(REPO, "ENVIRONMENT.md"), "utf8"),
+ renderEnvironmentMarkdown(),
+ "ENVIRONMENT.md is stale: run `archilyzer docs env`",
+ );
+});
+
+test("every variable the code reads is declared", () => {
+ const declared = new Set(ENV_VARS.map((v) => v.name));
+ const missing = [...reads()]
+ .filter(([name]) => !declared.has(name))
+ .map(([name, files]) => `${name} (read by ${[...files].join(", ")})`);
+ assert.deepEqual(missing, [], "declare these in common/lib/envVars.ts");
+});
+
+test("every declared variable is still mentioned by the code, a script or a compose file", () => {
+ const corpus = [
+ ...codeFiles().map((f) => readFileSync(f, "utf8")),
+ ...readdirSync(path.join(REPO, "docker"))
+ .filter((n) => statSync(path.join(REPO, "docker", n)).isFile())
+ .map((n) => readFileSync(path.join(REPO, "docker", n), "utf8")),
+ ...readdirSync(REPO)
+ .filter((n) => /^docker-compose.*\.yml$|^Dockerfile/.test(n))
+ .map((n) => readFileSync(path.join(REPO, n), "utf8")),
+ ...["", "editor", "export", "homepage"].map((d) =>
+ readFileSync(path.join(REPO, d, "package.json"), "utf8"),
+ ),
+ ].join("\n");
+ const stale = ENV_VARS.filter((v) => !new RegExp(`\\b${v.name}\\b`).test(corpus)).map((v) => v.name);
+ assert.deepEqual(stale, [], "nothing mentions these any more: delete their entries");
+});
+
+test("the paths audience is exactly what getPaths() reads", () => {
+ const pathsTs = codeText(path.join(REPO, "common/lib/paths.ts"));
+ const inPaths = new Set([...pathsTs.matchAll(/process\.env\.([A-Z][A-Z0-9_]+)/g)].map((m) => m[1]));
+ const declared = new Set(ENV_VARS.filter((v) => v.audience === "paths").map((v) => v.name));
+ assert.deepEqual([...declared].filter((n) => !inPaths.has(n)), [], "declared paths but not read by getPaths()");
+ assert.deepEqual([...inPaths].filter((n) => !declared.has(n)), [], "read by getPaths() but not declared as paths");
+});
+
+test("names are unique and every audience has a section", () => {
+ const names = ENV_VARS.map((v) => v.name);
+ assert.equal(new Set(names).size, names.length);
+ const audiences = new Set(ENV_AUDIENCES.map((a) => a.id));
+ for (const v of ENV_VARS) assert.ok(audiences.has(v.audience), v.name);
+ const md = renderEnvironmentMarkdown();
+ for (const v of ENV_VARS) assert.ok(md.includes(`| \`${v.name}\` |`), v.name);
+});
+
+test("the docker audience is the ARCHILYZER_ set", () => {
+ for (const v of ENV_VARS.filter((x) => x.audience === "docker")) {
+ assert.match(v.name, /^ARCHILYZER_/);
+ }
+});
diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts
@@ -0,0 +1,225 @@
+// EVERY ENVIRONMENT VARIABLE THE REPO READS — the one declared list.
+//
+// ENVIRONMENT.md is generated from this (common/bin/env-docs.ts, `--check` in
+// the common tests), and `envVars.test.ts` holds the list to the code: every
+// variable read by common/, editor/, export/, homepage/, mcp/src and scripts/ is
+// declared here, and every entry here is still read somewhere. A variable that
+// is added without an entry, or deleted with its entry left behind, fails the
+// build. umtool's own knobs are NOT here — its song and report scripts read
+// dozens, documented in umtool/docs, and fold into the core in one-core Phase 5.
+//
+// THE AUDIENCES, which are the point of the table:
+// paths — the ONE override surface for where things live and which binary
+// runs: every one is read by getPaths() (lib/paths.ts) and nowhere
+// else. The test checks both directions.
+// runtime — the other knobs, secrets and tokens a running process reads.
+// port — generated from lib/ports.mjs, never listed twice.
+// internal — set BY the pipeline for a process it spawns. Listed so a reader
+// knows what it is; nobody sets it by hand.
+// docker — the container's ARCHILYZER_* set, read by docker/*.sh, the
+// compose files and Caddy — a different process from the apps,
+// documented in RUNNING_IN_DOCKER.md.
+// test — read only by a test harness, a fake binary or a test-mode branch.
+//
+// Pure data (no imports but the port table), so a doc generator and a doctor can
+// both read it without loading anything else.
+
+import { PORTS } from "./ports.mjs";
+
+export type EnvAudience = "paths" | "runtime" | "port" | "internal" | "docker" | "test";
+
+export type EnvVarDecl = {
+ name: string;
+ audience: EnvAudience;
+ // What it does, one or two sentences.
+ doc: string;
+ // What an unset variable means, in words (a value, "off", "—").
+ default: string;
+ // Where it is read — a file, or a short list of them.
+ readBy: string;
+};
+
+const paths = (name: string, def: string, doc: string): EnvVarDecl => ({
+ name,
+ audience: "paths",
+ doc,
+ default: def,
+ readBy: "common/lib/paths.ts (getPaths)",
+});
+
+const DECLARED: EnvVarDecl[] = [
+ // ── paths: getPaths() ──────────────────────────────────────────────────
+ paths("TRANSCRIPTS_DIR", "`<repo>/transcripts`", "The corpus: channels, sites, the LMDB index, job logs, the saved-video store."),
+ paths("SAVED_VIDEOS_DIR", "`<TRANSCRIPTS_DIR>/saved-videos`", "The persisted source-video store, when it should live on another disk."),
+ paths("SITES_DIR", "`<TRANSCRIPTS_DIR>/sites`", "Per-site config (`<id>/site.json`, every key in [SITE.md](SITE.md)) and the homepage's `_homepage/`."),
+ paths("SETTINGS_FILE", "`<repo>/settings.json`", "The settings file (every key in [SETTINGS.md](SETTINGS.md))."),
+ paths("EXPORT_PUBLIC_DIR", "`<repo>/export/public`", "The dir the export site serves at `/`, composed one site at a time."),
+ paths("EXPORT_INDEX_DIR", "`.export-index` beside `EXPORT_PUBLIC_DIR`", "The build's staging area (not served): the shared index and per-site aggregates."),
+ paths("EXPORT_BUILDS_DIR", "`.export-builds` beside `EXPORT_PUBLIC_DIR`", "Per-site `out/` bundles from a docker-mode build."),
+ paths("EDITOR_CHANGELOG_FILE", "`<repo>/editor/CHANGELOG.md`", "The editor changelog the release cutter reads and rewrites. The e2e server points it at a gitignored copy."),
+ paths("EXPORT_CHANGELOG_FILE", "`<repo>/export/CHANGELOG.md`", "The export changelog, likewise."),
+ paths("CHARTS_CONFIG_FILE", "`<repo>/chart-templates.json`", "The legacy chart-templates file, read only by a migration."),
+ paths("SEARCH_ALIASES_FILE", "`<TRANSCRIPTS_DIR>/search-aliases.json`", "The corpus-wide search-alias dictionary."),
+ paths("CURATED_TAGS_FILE", "`<TRANSCRIPTS_DIR>/tags.json`", "Curated per-video tags. Written only through `applyTagAssignments`."),
+ paths("YTDLP_BIN", "`yt-dlp` on PATH", "The downloader. Every fetch goes through it."),
+ paths("FFMPEG_BIN", "`ffmpeg` on PATH", "Audio extraction for transcription and diarization."),
+ paths("FFPROBE_BIN", "`ffprobe` on PATH", "Duration checks (the short-audio guard, windowing)."),
+ paths("WHISPER_BIN", "`whisper-cli` on PATH", "whisper.cpp, one of the three transcription engines (with chough and parakeet.cpp)."),
+ paths("WHISPER_MODEL", "`~/whispercpp/whisper.cpp/models/ggml-base.en.bin`", "whisper.cpp's model, when a worker names none."),
+ paths("PARAKEET_STITCH_BIN", "`<repo>/scripts/parakeet-stitch.mjs`", "The parakeet.cpp engine's wrapper (overlapping windows, stitched)."),
+ paths("PARAKEET_CLI", "`parakeet-cli` on PATH", "The parakeet.cpp binary the wrapper drives (the wrapper reads it too)."),
+ paths("PARAKEET_MODEL", "none", "parakeet.cpp's `.gguf`, when a worker names none (the wrapper reads it too)."),
+ paths("DIARIZE_BIN", "`<repo>/scripts/diarize.mjs`", "The speaker-diarization wrapper. The e2e suite swaps in a fake here."),
+ paths("RSYNC_BIN", "`rsync` on PATH", "Mirrors the saved-video store to a backup destination."),
+ paths("FINDMNT_BIN", "`findmnt` on PATH", "The read-only volume-identity probe behind storage locations. Optional."),
+ paths("UDISKSCTL_BIN", "`udisksctl` on PATH", "Mounts an attached volume from `/storage`. Optional."),
+ paths("GALLERY_DL_BIN", "`gallery-dl` on PATH", "The X/Twitter post fetcher, for social channels."),
+ paths("OLLAMA_URL", "`http://127.0.0.1:11434`", "The local ollama server, the local digest and attribution engine."),
+ paths("CLAUDE_BIN", "`claude` on PATH", "The `claude` CLI, driving the opt-in metered digest lane."),
+
+ // ── runtime ────────────────────────────────────────────────────────────
+ { name: "WORKER_TOKEN", audience: "runtime", default: "unset (both surfaces off)", readBy: "common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts", doc: "Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`, the MCP's `fetch_clip`). Set the same value on both ends." },
+ { name: "SYNC_HEARTBEAT_SECONDS", audience: "runtime", default: "`settings.syncScheduler.heartbeatSeconds`", readBy: "editor/app/scheduler/heartbeat.ts", doc: "Overrides the editor's in-process sync heartbeat. `0` = no internal timer (tick from cron instead)." },
+ { name: "SYNC_TICK_URL", audience: "runtime", default: "`http://127.0.0.1:3001/api/scheduler/tick`", readBy: "common/bin/sync-tick.ts", doc: "Where `archilyzer sync tick` (cron's heartbeat) posts." },
+ { name: "SYNC_TICK_TOKEN", audience: "runtime", default: "unset (no auth)", readBy: "common/bin/sync-tick.ts, editor/app/scheduler/auth.ts", doc: "Bearer token for the tick endpoint; set on both the editor and the cron job." },
+ { name: "R2_ACCESS_KEY_ID", audience: "runtime", default: "—", readBy: "common/publish/build.ts", doc: "R2 S3 credentials for uploading oversize archives at deploy time (with `R2_SECRET_ACCESS_KEY` and `CLOUDFLARE_ACCOUNT_ID`). See [PUBLISH.md](PUBLISH.md)." },
+ { name: "R2_SECRET_ACCESS_KEY", audience: "runtime", default: "—", readBy: "common/publish/build.ts", doc: "See `R2_ACCESS_KEY_ID`." },
+ { name: "CLOUDFLARE_ACCOUNT_ID", audience: "runtime", default: "—", readBy: "common/publish/build.ts", doc: "The account the R2 endpoint belongs to. wrangler reads its own credentials." },
+ { name: "DOCKER_BIN", audience: "runtime", default: "`docker`", readBy: "common/publish/build.ts", doc: "The container engine for docker-mode builds (e.g. `podman`)." },
+ { name: "DOCKER_BUILD_MEMORY", audience: "runtime", default: "no cap", readBy: "common/publish/build.ts", doc: "Per-container memory cap for a docker-mode build (`--memory`)." },
+ { name: "DOCKER_BUILD_CPUS", audience: "runtime", default: "no cap", readBy: "common/publish/build.ts", doc: "Per-container CPU cap for a docker-mode build (`--cpus`)." },
+ { name: "ARCHIVE_CHANNEL_CONCURRENCY", audience: "runtime", default: "`4`", readBy: "common/bin/build-archives.ts", doc: "How many channels' archive zips `build archives` builds at once." },
+ { name: "MAX_ARCHIVE_BYTES", audience: "runtime", default: "the Cloudflare-safe cap", readBy: "common/bin/compose-site.ts", doc: "The served-file size cap for archives, in bytes; `0` = no cap. A site's own `archiveMaxBytes` wins." },
+ { name: "CHOUGH_BIN", audience: "runtime", default: "`chough` on PATH", readBy: "common/lib/transcriptionApps.ts", doc: "The chough transcription engine, when a worker names no binary." },
+ { name: "CHOUGH_MODEL", audience: "runtime", default: "chough's own", readBy: "chough (set by common/lib/transcriptionApps.ts)", doc: "Passed to chough from a worker's model field; chough auto-downloads one when unset." },
+ { name: "CHOUGH_URL", audience: "runtime", default: "local", readBy: "chough (set by common/lib/transcriptionApps.ts)", doc: "Passed to chough from a worker's remote-server field." },
+ { name: "OLLAMA_DIGEST_MODEL", audience: "runtime", default: "`qwen2.5:7b`", readBy: "common/lib/digestApps.ts", doc: "The ollama model the local digest lane asks for when settings name none." },
+ { name: "CLAUDE_DIGEST_MODEL", audience: "runtime", default: "the CLI's default", readBy: "common/lib/digestApps.ts", doc: "The model the metered digest lane asks `claude` for when settings name none." },
+ { name: "NITTER_INSTANCES", audience: "runtime", default: "a built-in list", readBy: "common/social/xNitterFetcher.ts", doc: "Comma-separated Nitter instances for the X fallback fetcher, in order of preference." },
+ { name: "UMTOOL_URL", audience: "runtime", default: "unset (no link)", readBy: "editor/app/channels/[slug]/videos/[id]/page.tsx", doc: "umtool's front door; when set, the video page links to it." },
+ { name: "TRANSCRIPT_SITE_URL", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: one published archive to read over HTTP." },
+ { name: "TRANSCRIPT_HUB_URL", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: a hub, federating every archive it lists." },
+ { name: "TRANSCRIPT_LOCAL_DIR", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: a composed public dir on disk." },
+ { name: "TRANSCRIPT_PLATFORM_LINKS", audience: "runtime", default: "off", readBy: "common/lib/archive/reader-fs.ts", doc: "`1` cites platform watch pages instead of the archive's own pages." },
+ { name: "MCP_IO_STATS", audience: "runtime", default: "off", readBy: "common/lib/archive/io-stats.ts", doc: "`1` turns on per-call I/O accounting, for `mcp/bench`." },
+ { name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool", doc: "Which editor `pnpm ops` and the MCP's `fetch_clip` talk to." },
+ { name: "ARCHILYZER_AGENT", audience: "runtime", default: "`cli`", readBy: "scripts/archilyzer-ops.mjs", doc: "Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`." },
+ { name: "DIARIZE_ENGINE_KIND", audience: "runtime", default: "`sherpa-onnx`", readBy: "scripts/diarize.mjs", doc: "The diarization engine: `sherpa-onnx` or `sortformer`." },
+ { name: "DIARIZE_ENGINE_CMD", audience: "runtime", default: "the bundled sherpa script", readBy: "scripts/diarize.mjs", doc: "The engine command the wrapper runs." },
+ { name: "DIARIZE_PYTHON", audience: "runtime", default: "`python3`", readBy: "scripts/diarize.mjs", doc: "The python for the default engine." },
+ { name: "DIARIZE_SEG_MODEL", audience: "runtime", default: "— (required)", readBy: "scripts/diarize.mjs", doc: "Segmentation model. The editor passes the settings' value as a flag." },
+ { name: "DIARIZE_EMB_MODEL", audience: "runtime", default: "— (required)", readBy: "scripts/diarize.mjs", doc: "Speaker-embedding model. The editor passes the settings' value as a flag." },
+ { name: "DIARIZE_THRESHOLD", audience: "runtime", default: "`0.5`", readBy: "scripts/diarize.mjs", doc: "Clustering threshold." },
+ { name: "DIARIZE_THREADS", audience: "runtime", default: "`4`", readBy: "scripts/diarize.mjs", doc: "Engine threads." },
+ { name: "DIARIZE_WINDOW_MINUTES", audience: "runtime", default: "`45`", readBy: "scripts/diarize.mjs", doc: "Window length for long files; `0` never windows." },
+ { name: "DIARIZE_WINDOW_AFTER_MINUTES", audience: "runtime", default: "`90`", readBy: "scripts/diarize.mjs", doc: "Only files longer than this are windowed." },
+ { name: "SORTFORMER_BIN", audience: "runtime", default: "— (required for sortformer)", readBy: "scripts/diarize.mjs, scripts/diarize-sortformer.mjs", doc: "The sortformer engine binary." },
+ { name: "SORTFORMER_MODEL", audience: "runtime", default: "— (required for sortformer)", readBy: "scripts/diarize.mjs, scripts/diarize-sortformer.mjs", doc: "The sortformer `.gguf`." },
+ { name: "PARAKEET_SEGMENT_SEC", audience: "runtime", default: "`480`", readBy: "scripts/parakeet-stitch.mjs", doc: "parakeet window length, seconds (a worker's chunk size wins)." },
+ { name: "PARAKEET_OVERLAP_SEC", audience: "runtime", default: "`6`", readBy: "scripts/parakeet-stitch.mjs", doc: "parakeet window overlap, seconds." },
+ { name: "PARAKEET_DECODER", audience: "runtime", default: "parakeet-cli's", readBy: "scripts/parakeet-stitch.mjs", doc: "`ctc` or `tdt`, passed through to parakeet-cli." },
+ { name: "PARAKEET_LANG", audience: "runtime", default: "parakeet-cli's", readBy: "scripts/parakeet-stitch.mjs", doc: "A locale, passed through to parakeet-cli." },
+ { name: "PARAKEET_DEVICE", audience: "runtime", default: "parakeet-cli's", readBy: "scripts/parakeet-stitch.mjs", doc: "Compute device (`cpu`, `CUDA0`, `Vulkan1`, …), exported to parakeet-cli." },
+
+ // ── internal: the pipeline sets these for a process it spawns ──────────
+ { name: "SITE_ID", audience: "internal", default: "—", readBy: "common/bin/compose-site.ts, export/app/lib/site.ts", doc: "Which site a compose or an export build is for. `archilyzer build site <id>` sets it; `compose site` and `build site` fall back to it when no id is given." },
+ { name: "INSTANCE_MODE", audience: "internal", default: "a site", readBy: "export/app/lib/mode.ts, common/lib/archive/contract.ts", doc: "`hub` makes the export build the hub. Set by `archilyzer build hub`." },
+ { name: "BUILD_ARCHIVES", audience: "internal", default: "on", readBy: "common/bin/compose-site.ts, common/bin/build-archives.ts", doc: "`0` skips archive-zip generation for one build (`--skip-archives`)." },
+ { name: "ARCHIVES_READONLY", audience: "internal", default: "off", readBy: "common/bin/compose-site.ts", doc: "`1` inside a docker-mode build container: materialize archives, never write the shared cache." },
+ { name: "HOMEPAGE_PUBLIC_DIR", audience: "internal", default: "`<repo>/homepage/public`", readBy: "common/bin/compose-homepage.ts", doc: "Where `compose homepage` writes." },
+ { name: "HOMEPAGE_SUMMARY_FILE", audience: "internal", default: "`homepage/public/homepage-summary.json`", readBy: "homepage/app/lib/summary.ts", doc: "A dev-only summary file for the homepage; ignored by a production build. The homepage e2e sets it." },
+
+ // ── docker: the container's set ────────────────────────────────────────
+ { name: "ARCHILYZER_TRANSCRIBER", audience: "docker", default: "baked per image target (`whisper-cpp` in `runtime`)", readBy: "docker/entrypoint.sh", doc: "`whisper-cpp` or `parakeet`: which worker the first boot seeds and which model it fetches." },
+ { name: "ARCHILYZER_FETCH_MODEL", audience: "docker", default: "per transcriber", readBy: "docker/entrypoint.sh", doc: "Which model the first boot downloads; `none` skips it." },
+ { name: "ARCHILYZER_MODELS_DIR", audience: "docker", default: "`/data/models`", readBy: "docker/entrypoint.sh", doc: "Where models live in the container." },
+ { name: "ARCHILYZER_BUILDS_DIR", audience: "docker", default: "`/data/builds`", readBy: "docker/entrypoint.sh", doc: "Where the container keeps built sites." },
+ { name: "ARCHILYZER_SITE_OUT", audience: "docker", default: "`/data/builds/site`", readBy: "docker/entrypoint.sh, docker/publish-site.sh", doc: "The built export site the `site` service serves." },
+ { name: "ARCHILYZER_IDLE_BOOT", audience: "docker", default: "off", readBy: "common/lib/idleBoot.ts (the editor)", doc: "`1` boots the editor without arming the heartbeat or any auto-queue runner." },
+ { name: "ARCHILYZER_AUTH_MODE", audience: "docker", default: "`basic`", readBy: "docker/guard-exposure.sh, docker/caddy-start.sh", doc: "`basic`, `forward` or `none` — the only escape hatch from the exposure guard." },
+ { name: "ARCHILYZER_AUTH_USER", audience: "docker", default: "`archilyzer`", readBy: "docker/Caddyfile", doc: "Basic-auth user." },
+ { name: "ARCHILYZER_AUTH_HASH", audience: "docker", default: "—", readBy: "docker/Caddyfile, docker/guard-exposure.sh", doc: "Basic-auth bcrypt hash (`caddy hash-password`)." },
+ { name: "ARCHILYZER_AUTH_IMPORT", audience: "docker", default: "derived from the mode", readBy: "docker/Caddyfile", doc: "Set by docker/caddy-start.sh from the mode: which auth snippet the private sites import." },
+ { name: "ARCHILYZER_FORWARD_AUTH_UPSTREAM", audience: "docker", default: "—", readBy: "docker/Caddyfile", doc: "Forward-auth server (Authelia, tinyauth, …), `host:port`." },
+ { name: "ARCHILYZER_FORWARD_AUTH_URI", audience: "docker", default: "`/api/auth/caddy`", readBy: "docker/Caddyfile", doc: "The forward-auth server's verify path." },
+ { name: "ARCHILYZER_TAG", audience: "docker", default: "`local`", readBy: "docker-compose*.yml", doc: "The image tag the compose files build and run." },
+
+ // ── test: harnesses, fakes and test-mode branches ──────────────────────
+ { name: "EDITOR_TEST_ROUTES", audience: "test", default: "off", readBy: "editor/app/api/test/_guard.ts, editor/instrumentation.ts", doc: "`1` opens the editor's `/api/test/*` routes. The e2e server sets it." },
+ { name: "E2E_MODE", audience: "test", default: "dev", readBy: "editor/playwright.config.ts", doc: "`start` runs the editor suite against `next start` instead of `next dev`." },
+ { name: "E2E_QUEUE", audience: "test", default: "on", readBy: "scripts/queue-lock.mjs", doc: "`0` skips the machine-global e2e queue (the port check still runs)." },
+ { name: "E2E_PORT_CHECK", audience: "test", default: "on", readBy: "scripts/queue-lock.mjs", doc: "`0` skips the pre-run check that the suite's ports are free." },
+ { name: "E2E_QUEUE_TIMEOUT", audience: "test", default: "wait forever", readBy: "scripts/queue-lock.mjs", doc: "Seconds to wait for the queue before giving up." },
+ { name: "E2E_PORT_GRACE_MS", audience: "test", default: "`3000`", readBy: "scripts/queue-lock.mjs", doc: "How long the port check waits for a just-freed port." },
+ { name: "E2E_QUEUE_LOCK_FILE", audience: "test", default: "one per machine", readBy: "scripts/queue-lock.mjs", doc: "The queue's lock file; the queue's own tests point it elsewhere." },
+ { name: "QUEUE_LOCK_HELD", audience: "test", default: "—", readBy: "scripts/queue-lock.mjs", doc: "Set by the queue for the command it runs, so a nested wrapper passes through." },
+ { name: "PLAYWRIGHT_BASE_URL", audience: "test", default: "`http://localhost:<PORT>`", readBy: "editor/playwright.config.ts, editor/e2e/baseUrl.ts", doc: "The editor test server's URL; the worktree injector sets it." },
+ { name: "AUDIO_CHECK_INTERVAL_MS_OVERRIDE", audience: "test", default: "the real cadence", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Shrinks the mid-download audio check so the e2e suite sees it fire." },
+ { name: "AUDIO_CHECK_SIZE_GATE_OVERRIDE", audience: "test", default: "the real gate", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Likewise, the size gate." },
+ { name: "AUDIO_CHECK_INTERVAL_FLOOR_MS_OVERRIDE", audience: "test", default: "the real floor", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Likewise, the interval floor." },
+ { name: "AUDIO_CHECK_RECOVER_STEP_MS_OVERRIDE", audience: "test", default: "the real step", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Likewise, the recovery step." },
+ { name: "AUDIO_CHECK_RECOVER_AFTER_OVERRIDE", audience: "test", default: "the real count", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Likewise, the recovery count." },
+ { name: "AUDIO_CHECK_DEBUG_PAUSE_MS", audience: "test", default: "off", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "A debugging pause inside the audio check." },
+ { name: "FAKE_YTDLP_AUDIO_CHECK_MODE", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: which audio-check scenario to act out." },
+ { name: "FAKE_YTDLP_CHUNK_DELAY_MS", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: delay between written chunks." },
+ { name: "FAKE_YTDLP_CORRUPT_AFTER_CHUNK", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: start corrupting after this chunk." },
+ { name: "FAKE_YTDLP_CORRUPT_RUNS", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: how many runs corrupt." },
+ { name: "FAKE_YTDLP_DETERMINISTIC_CORRUPT", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: corrupt deterministically." },
+ { name: "FAKE_YTDLP_RECOVER_ON_RESUME", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: a resumed run recovers." },
+ { name: "FAKE_YTDLP_TOTAL_CHUNKS", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: how many chunks a download has." },
+ { name: "FAKE_GALLERY_DL_AUTH_FAIL", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-gallery-dl.mjs", doc: "Fake gallery-dl: fail as an auth error." },
+ { name: "FIXTURE_MAX_LIFETIME_MS", audience: "test", default: "the watchdog's", readBy: "editor/e2e/fixtures/bin/_watchdog.mjs", doc: "How long a fake binary may live before its watchdog kills it." },
+ { name: "OLLAMA_STUB_MODEL", audience: "test", default: "`qwen2.5:7b`", readBy: "editor/e2e/fixtures/ollama-stub.mjs", doc: "The model the ollama stub claims to serve." },
+ { name: "RACK_SHOTS", audience: "test", default: "off (spec skipped)", readBy: "editor/e2e/channels-rack-audit.spec.ts", doc: "Runs the `/channels` rack screenshot audit." },
+ { name: "TWO_ORIGIN_REBUILD", audience: "test", default: "off", readBy: "export/e2e-2origin/globalSetup.ts", doc: "`1` rebuilds the two-origin suite's cached hub bundle." },
+ { name: "IMAGE", audience: "test", default: "`yt-dlp-transcript-browser-e2e`", readBy: "scripts/run-sharded-e2e.mjs", doc: "The sharded e2e run's image tag." },
+ { name: "SKIP_BUILD", audience: "test", default: "off", readBy: "scripts/run-sharded-e2e.mjs", doc: "`1` reuses the sharded e2e image instead of rebuilding it (`--no-build`)." },
+];
+
+// The port rows come from the port table, so a port is declared once.
+const PORT_ROWS: EnvVarDecl[] = Object.entries(PORTS).map(([name, decl]) => ({
+ name,
+ audience: "port",
+ doc: `${decl.what[0].toUpperCase()}${decl.what.slice(1)}. A worktree adds its offset (\`pnpm wt list\`).`,
+ default: `\`${decl.base}\``,
+ readBy: "common/lib/ports.mjs",
+}));
+
+export const ENV_VARS: readonly EnvVarDecl[] = [...DECLARED, ...PORT_ROWS];
+
+export const ENV_AUDIENCES: ReadonlyArray<{ id: EnvAudience; title: string; intro: string }> = [
+ { id: "paths", title: "Paths and binaries", intro: "The one override surface for where things live and which binary runs. Every one is read by `getPaths()` (`common/lib/paths.ts`) and nowhere else; nothing hardcodes a location." },
+ { id: "runtime", title: "Runtime", intro: "Tokens, credentials and knobs a running process reads. Most configuration is not here but in `settings.json` ([SETTINGS.md](SETTINGS.md))." },
+ { id: "port", title: "Ports", intro: "Every local server's default port, from `common/lib/ports.mjs`. The primary checkout uses these; worktree N adds N × 100 (`pnpm wt list`)." },
+ { id: "internal", title: "Set by the pipeline", intro: "The publish pipeline sets these for a process it spawns. Listed so a reader knows what they are; nobody sets them by hand." },
+ { id: "docker", title: "Docker", intro: "The container's own set, read by `docker/*.sh`, the compose files and Caddy — not by the apps' code (except `ARCHILYZER_IDLE_BOOT`). See [RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md)." },
+ { id: "test", title: "Tests only", intro: "Read only by a test harness, a fake binary or a test-mode branch. Never set one on a real instance." },
+];
+
+export function envVar(name: string): EnvVarDecl | undefined {
+ return ENV_VARS.find((v) => v.name === name);
+}
+
+// ENVIRONMENT.md — one table per audience.
+export function renderEnvironmentMarkdown(): string {
+ const cell = (s: string) => s.replace(/\|/g, "\\|");
+ const out: string[] = [
+ "# Environment variables",
+ "",
+ "<!-- GENERATED by common/bin/env-docs.ts from common/lib/envVars.ts — do not edit by hand. -->",
+ "",
+ "Every environment variable the repo's code reads, by who it is for. The list is code (`common/lib/envVars.ts`), and a test fails when the code reads a variable the list does not declare, or the list declares one nothing reads. umtool's own knobs are documented in [umtool/docs](umtool/docs/README.md).",
+ "",
+ "Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/archilyzer.ts docs env`. `archilyzer doctor` prints which of the paths overrides are set on this machine.",
+ "",
+ ];
+ for (const a of ENV_AUDIENCES) {
+ const rows = ENV_VARS.filter((v) => v.audience === a.id);
+ out.push(`## ${a.title}`, "", a.intro, "", "| Variable | Default | What it does | Read by |", "|---|---|---|---|");
+ for (const v of rows) {
+ out.push(`| \`${v.name}\` | ${cell(v.default)} | ${cell(v.doc)} | ${cell(v.readBy)} |`);
+ }
+ out.push("");
+ }
+ return out.join("\n");
+}