Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 050c6202ad6be34834ae0f78f84c455a82933565
parent 8d6afff1b19e317844012eeffb8dd917fa27099a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 28 Sep 2026 01:48:51 -0400

common: every environment variable, declared once; ENVIRONMENT.md generated

`common/lib/envVars.ts` declares each variable the apps' code reads, by
audience: paths (the getPaths() override surface), runtime, port (generated
from ports.mjs), internal (set by the pipeline for a process it spawns),
docker (the container's ARCHILYZER_* set) and test. ENVIRONMENT.md is
generated from it (`archilyzer docs env [--check]`, `bin/env-docs.ts`).

`envVars.test.ts` holds the list to the code in both directions: every
`process.env.X` / `env.X` read under common/, editor/, export/, homepage/,
mcp/src and scripts/ is declared, every declared name is still mentioned by
the code, docker/ or a compose file, the paths audience is exactly what
getPaths() reads, and the committed ENVIRONMENT.md is fresh. umtool's own
knobs stay in umtool/docs until Phase 5.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
AENVIRONMENT.md | 181+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/bin/archilyzer.ts | 7+++++++
Acommon/bin/env-docs.ts | 34++++++++++++++++++++++++++++++++++
Acommon/lib/envVars.test.ts | 128+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/envVars.ts | 225+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
5 files changed, 575 insertions(+), 0 deletions(-)

diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md @@ -0,0 +1,181 @@ +# Environment variables + +<!-- GENERATED by common/bin/env-docs.ts from common/lib/envVars.ts — do not edit by hand. --> + +Every environment variable the repo's code reads, by who it is for. The list is code (`common/lib/envVars.ts`), and a test fails when the code reads a variable the list does not declare, or the list declares one nothing reads. umtool's own knobs are documented in [umtool/docs](umtool/docs/README.md). + +Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/archilyzer.ts docs env`. `archilyzer doctor` prints which of the paths overrides are set on this machine. + +## Paths and binaries + +The one override surface for where things live and which binary runs. Every one is read by `getPaths()` (`common/lib/paths.ts`) and nowhere else; nothing hardcodes a location. + +| Variable | Default | What it does | Read by | +|---|---|---|---| +| `TRANSCRIPTS_DIR` | `<repo>/transcripts` | The corpus: channels, sites, the LMDB index, job logs, the saved-video store. | common/lib/paths.ts (getPaths) | +| `SAVED_VIDEOS_DIR` | `<TRANSCRIPTS_DIR>/saved-videos` | The persisted source-video store, when it should live on another disk. | common/lib/paths.ts (getPaths) | +| `SITES_DIR` | `<TRANSCRIPTS_DIR>/sites` | Per-site config (`<id>/site.json`, every key in [SITE.md](SITE.md)) and the homepage's `_homepage/`. | common/lib/paths.ts (getPaths) | +| `SETTINGS_FILE` | `<repo>/settings.json` | The settings file (every key in [SETTINGS.md](SETTINGS.md)). | common/lib/paths.ts (getPaths) | +| `EXPORT_PUBLIC_DIR` | `<repo>/export/public` | The dir the export site serves at `/`, composed one site at a time. | common/lib/paths.ts (getPaths) | +| `EXPORT_INDEX_DIR` | `.export-index` beside `EXPORT_PUBLIC_DIR` | The build's staging area (not served): the shared index and per-site aggregates. | common/lib/paths.ts (getPaths) | +| `EXPORT_BUILDS_DIR` | `.export-builds` beside `EXPORT_PUBLIC_DIR` | Per-site `out/` bundles from a docker-mode build. | common/lib/paths.ts (getPaths) | +| `EDITOR_CHANGELOG_FILE` | `<repo>/editor/CHANGELOG.md` | The editor changelog the release cutter reads and rewrites. The e2e server points it at a gitignored copy. | common/lib/paths.ts (getPaths) | +| `EXPORT_CHANGELOG_FILE` | `<repo>/export/CHANGELOG.md` | The export changelog, likewise. | common/lib/paths.ts (getPaths) | +| `CHARTS_CONFIG_FILE` | `<repo>/chart-templates.json` | The legacy chart-templates file, read only by a migration. | common/lib/paths.ts (getPaths) | +| `SEARCH_ALIASES_FILE` | `<TRANSCRIPTS_DIR>/search-aliases.json` | The corpus-wide search-alias dictionary. | common/lib/paths.ts (getPaths) | +| `CURATED_TAGS_FILE` | `<TRANSCRIPTS_DIR>/tags.json` | Curated per-video tags. Written only through `applyTagAssignments`. | common/lib/paths.ts (getPaths) | +| `YTDLP_BIN` | `yt-dlp` on PATH | The downloader. Every fetch goes through it. | common/lib/paths.ts (getPaths) | +| `FFMPEG_BIN` | `ffmpeg` on PATH | Audio extraction for transcription and diarization. | common/lib/paths.ts (getPaths) | +| `FFPROBE_BIN` | `ffprobe` on PATH | Duration checks (the short-audio guard, windowing). | common/lib/paths.ts (getPaths) | +| `WHISPER_BIN` | `whisper-cli` on PATH | whisper.cpp, one of the three transcription engines (with chough and parakeet.cpp). | common/lib/paths.ts (getPaths) | +| `WHISPER_MODEL` | `~/whispercpp/whisper.cpp/models/ggml-base.en.bin` | whisper.cpp's model, when a worker names none. | common/lib/paths.ts (getPaths) | +| `PARAKEET_STITCH_BIN` | `<repo>/scripts/parakeet-stitch.mjs` | The parakeet.cpp engine's wrapper (overlapping windows, stitched). | common/lib/paths.ts (getPaths) | +| `PARAKEET_CLI` | `parakeet-cli` on PATH | The parakeet.cpp binary the wrapper drives (the wrapper reads it too). | common/lib/paths.ts (getPaths) | +| `PARAKEET_MODEL` | none | parakeet.cpp's `.gguf`, when a worker names none (the wrapper reads it too). | common/lib/paths.ts (getPaths) | +| `DIARIZE_BIN` | `<repo>/scripts/diarize.mjs` | The speaker-diarization wrapper. The e2e suite swaps in a fake here. | common/lib/paths.ts (getPaths) | +| `RSYNC_BIN` | `rsync` on PATH | Mirrors the saved-video store to a backup destination. | common/lib/paths.ts (getPaths) | +| `FINDMNT_BIN` | `findmnt` on PATH | The read-only volume-identity probe behind storage locations. Optional. | common/lib/paths.ts (getPaths) | +| `UDISKSCTL_BIN` | `udisksctl` on PATH | Mounts an attached volume from `/storage`. Optional. | common/lib/paths.ts (getPaths) | +| `GALLERY_DL_BIN` | `gallery-dl` on PATH | The X/Twitter post fetcher, for social channels. | common/lib/paths.ts (getPaths) | +| `OLLAMA_URL` | `http://127.0.0.1:11434` | The local ollama server, the local digest and attribution engine. | common/lib/paths.ts (getPaths) | +| `CLAUDE_BIN` | `claude` on PATH | The `claude` CLI, driving the opt-in metered digest lane. | common/lib/paths.ts (getPaths) | + +## Runtime + +Tokens, credentials and knobs a running process reads. Most configuration is not here but in `settings.json` ([SETTINGS.md](SETTINGS.md)). + +| Variable | Default | What it does | Read by | +|---|---|---|---| +| `WORKER_TOKEN` | unset (both surfaces off) | Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`, the MCP's `fetch_clip`). Set the same value on both ends. | common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts | +| `SYNC_HEARTBEAT_SECONDS` | `settings.syncScheduler.heartbeatSeconds` | Overrides the editor's in-process sync heartbeat. `0` = no internal timer (tick from cron instead). | editor/app/scheduler/heartbeat.ts | +| `SYNC_TICK_URL` | `http://127.0.0.1:3001/api/scheduler/tick` | Where `archilyzer sync tick` (cron's heartbeat) posts. | common/bin/sync-tick.ts | +| `SYNC_TICK_TOKEN` | unset (no auth) | Bearer token for the tick endpoint; set on both the editor and the cron job. | common/bin/sync-tick.ts, editor/app/scheduler/auth.ts | +| `R2_ACCESS_KEY_ID` | — | R2 S3 credentials for uploading oversize archives at deploy time (with `R2_SECRET_ACCESS_KEY` and `CLOUDFLARE_ACCOUNT_ID`). See [PUBLISH.md](PUBLISH.md). | common/publish/build.ts | +| `R2_SECRET_ACCESS_KEY` | — | See `R2_ACCESS_KEY_ID`. | common/publish/build.ts | +| `CLOUDFLARE_ACCOUNT_ID` | — | The account the R2 endpoint belongs to. wrangler reads its own credentials. | common/publish/build.ts | +| `DOCKER_BIN` | `docker` | The container engine for docker-mode builds (e.g. `podman`). | common/publish/build.ts | +| `DOCKER_BUILD_MEMORY` | no cap | Per-container memory cap for a docker-mode build (`--memory`). | common/publish/build.ts | +| `DOCKER_BUILD_CPUS` | no cap | Per-container CPU cap for a docker-mode build (`--cpus`). | common/publish/build.ts | +| `ARCHIVE_CHANNEL_CONCURRENCY` | `4` | How many channels' archive zips `build archives` builds at once. | common/bin/build-archives.ts | +| `MAX_ARCHIVE_BYTES` | the Cloudflare-safe cap | The served-file size cap for archives, in bytes; `0` = no cap. A site's own `archiveMaxBytes` wins. | common/bin/compose-site.ts | +| `CHOUGH_BIN` | `chough` on PATH | The chough transcription engine, when a worker names no binary. | common/lib/transcriptionApps.ts | +| `CHOUGH_MODEL` | chough's own | Passed to chough from a worker's model field; chough auto-downloads one when unset. | chough (set by common/lib/transcriptionApps.ts) | +| `CHOUGH_URL` | local | Passed to chough from a worker's remote-server field. | chough (set by common/lib/transcriptionApps.ts) | +| `OLLAMA_DIGEST_MODEL` | `qwen2.5:7b` | The ollama model the local digest lane asks for when settings name none. | common/lib/digestApps.ts | +| `CLAUDE_DIGEST_MODEL` | the CLI's default | The model the metered digest lane asks `claude` for when settings name none. | common/lib/digestApps.ts | +| `NITTER_INSTANCES` | a built-in list | Comma-separated Nitter instances for the X fallback fetcher, in order of preference. | common/social/xNitterFetcher.ts | +| `UMTOOL_URL` | unset (no link) | umtool's front door; when set, the video page links to it. | editor/app/channels/[slug]/videos/[id]/page.tsx | +| `TRANSCRIPT_SITE_URL` | — | MCP server: one published archive to read over HTTP. | mcp/src/sources.ts | +| `TRANSCRIPT_HUB_URL` | — | MCP server: a hub, federating every archive it lists. | mcp/src/sources.ts | +| `TRANSCRIPT_LOCAL_DIR` | — | MCP server: a composed public dir on disk. | mcp/src/sources.ts | +| `TRANSCRIPT_PLATFORM_LINKS` | off | `1` cites platform watch pages instead of the archive's own pages. | common/lib/archive/reader-fs.ts | +| `MCP_IO_STATS` | off | `1` turns on per-call I/O accounting, for `mcp/bench`. | common/lib/archive/io-stats.ts | +| `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's `fetch_clip` talk to. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool | +| `ARCHILYZER_AGENT` | `cli` | Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`. | scripts/archilyzer-ops.mjs | +| `DIARIZE_ENGINE_KIND` | `sherpa-onnx` | The diarization engine: `sherpa-onnx` or `sortformer`. | scripts/diarize.mjs | +| `DIARIZE_ENGINE_CMD` | the bundled sherpa script | The engine command the wrapper runs. | scripts/diarize.mjs | +| `DIARIZE_PYTHON` | `python3` | The python for the default engine. | scripts/diarize.mjs | +| `DIARIZE_SEG_MODEL` | — (required) | Segmentation model. The editor passes the settings' value as a flag. | scripts/diarize.mjs | +| `DIARIZE_EMB_MODEL` | — (required) | Speaker-embedding model. The editor passes the settings' value as a flag. | scripts/diarize.mjs | +| `DIARIZE_THRESHOLD` | `0.5` | Clustering threshold. | scripts/diarize.mjs | +| `DIARIZE_THREADS` | `4` | Engine threads. | scripts/diarize.mjs | +| `DIARIZE_WINDOW_MINUTES` | `45` | Window length for long files; `0` never windows. | scripts/diarize.mjs | +| `DIARIZE_WINDOW_AFTER_MINUTES` | `90` | Only files longer than this are windowed. | scripts/diarize.mjs | +| `SORTFORMER_BIN` | — (required for sortformer) | The sortformer engine binary. | scripts/diarize.mjs, scripts/diarize-sortformer.mjs | +| `SORTFORMER_MODEL` | — (required for sortformer) | The sortformer `.gguf`. | scripts/diarize.mjs, scripts/diarize-sortformer.mjs | +| `PARAKEET_SEGMENT_SEC` | `480` | parakeet window length, seconds (a worker's chunk size wins). | scripts/parakeet-stitch.mjs | +| `PARAKEET_OVERLAP_SEC` | `6` | parakeet window overlap, seconds. | scripts/parakeet-stitch.mjs | +| `PARAKEET_DECODER` | parakeet-cli's | `ctc` or `tdt`, passed through to parakeet-cli. | scripts/parakeet-stitch.mjs | +| `PARAKEET_LANG` | parakeet-cli's | A locale, passed through to parakeet-cli. | scripts/parakeet-stitch.mjs | +| `PARAKEET_DEVICE` | parakeet-cli's | Compute device (`cpu`, `CUDA0`, `Vulkan1`, …), exported to parakeet-cli. | scripts/parakeet-stitch.mjs | + +## Ports + +Every local server's default port, from `common/lib/ports.mjs`. The primary checkout uses these; worktree N adds N × 100 (`pnpm wt list`). + +| Variable | Default | What it does | Read by | +|---|---|---|---| +| `EDITOR_PORT` | `3001` | Editor real dev/start. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `PORT` | `3011` | Editor test server + Playwright editor baseURL. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `EXPORT_PORT` | `3010` | Export server launched by the editor e2e. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `EXPORT_DEV_PORT` | `3000` | Export real dev. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `EXPORT_E2E_PORT` | `3020` | Export's own Playwright suite. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `OLLAMA_STUB_PORT` | `11435` | Digest-lane stub server in the editor e2e suite. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `HOMEPAGE_DEV_PORT` | `3030` | Homepage real dev. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `HOMEPAGE_PORT` | `3031` | Homepage static `serve out` (start:homepage). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `HOMEPAGE_E2E_PORT` | `3040` | Homepage's own Playwright suite. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `HUB_PORT` | `3041` | Export's hub Playwright suite (e2e:hub). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `UMTOOL_PORT` | `3050` | Umtool real dev/start. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `UMTOOL_E2E_PORT` | `3051` | Umtool's own Playwright suite. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `EDITOR_STUB_PORT` | `3052` | Stub editor the umtool e2e suite fetches clips from. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `ORIGIN_B_PORT` | `4610` | Export's two-origin suite: the member site (e2e:2origin). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | +| `HUB_A_PORT` | `4611` | Export's two-origin suite: the hub (e2e:2origin). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs | + +## Set by the pipeline + +The publish pipeline sets these for a process it spawns. Listed so a reader knows what they are; nobody sets them by hand. + +| Variable | Default | What it does | Read by | +|---|---|---|---| +| `SITE_ID` | — | Which site a compose or an export build is for. `archilyzer build site <id>` sets it; `compose site` and `build site` fall back to it when no id is given. | common/bin/compose-site.ts, export/app/lib/site.ts | +| `INSTANCE_MODE` | a site | `hub` makes the export build the hub. Set by `archilyzer build hub`. | export/app/lib/mode.ts, common/lib/archive/contract.ts | +| `BUILD_ARCHIVES` | on | `0` skips archive-zip generation for one build (`--skip-archives`). | common/bin/compose-site.ts, common/bin/build-archives.ts | +| `ARCHIVES_READONLY` | off | `1` inside a docker-mode build container: materialize archives, never write the shared cache. | common/bin/compose-site.ts | +| `HOMEPAGE_PUBLIC_DIR` | `<repo>/homepage/public` | Where `compose homepage` writes. | common/bin/compose-homepage.ts | +| `HOMEPAGE_SUMMARY_FILE` | `homepage/public/homepage-summary.json` | A dev-only summary file for the homepage; ignored by a production build. The homepage e2e sets it. | homepage/app/lib/summary.ts | + +## Docker + +The container's own set, read by `docker/*.sh`, the compose files and Caddy — not by the apps' code (except `ARCHILYZER_IDLE_BOOT`). See [RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md). + +| Variable | Default | What it does | Read by | +|---|---|---|---| +| `ARCHILYZER_TRANSCRIBER` | baked per image target (`whisper-cpp` in `runtime`) | `whisper-cpp` or `parakeet`: which worker the first boot seeds and which model it fetches. | docker/entrypoint.sh | +| `ARCHILYZER_FETCH_MODEL` | per transcriber | Which model the first boot downloads; `none` skips it. | docker/entrypoint.sh | +| `ARCHILYZER_MODELS_DIR` | `/data/models` | Where models live in the container. | docker/entrypoint.sh | +| `ARCHILYZER_BUILDS_DIR` | `/data/builds` | Where the container keeps built sites. | docker/entrypoint.sh | +| `ARCHILYZER_SITE_OUT` | `/data/builds/site` | The built export site the `site` service serves. | docker/entrypoint.sh, docker/publish-site.sh | +| `ARCHILYZER_IDLE_BOOT` | off | `1` boots the editor without arming the heartbeat or any auto-queue runner. | common/lib/idleBoot.ts (the editor) | +| `ARCHILYZER_AUTH_MODE` | `basic` | `basic`, `forward` or `none` — the only escape hatch from the exposure guard. | docker/guard-exposure.sh, docker/caddy-start.sh | +| `ARCHILYZER_AUTH_USER` | `archilyzer` | Basic-auth user. | docker/Caddyfile | +| `ARCHILYZER_AUTH_HASH` | — | Basic-auth bcrypt hash (`caddy hash-password`). | docker/Caddyfile, docker/guard-exposure.sh | +| `ARCHILYZER_AUTH_IMPORT` | derived from the mode | Set by docker/caddy-start.sh from the mode: which auth snippet the private sites import. | docker/Caddyfile | +| `ARCHILYZER_FORWARD_AUTH_UPSTREAM` | — | Forward-auth server (Authelia, tinyauth, …), `host:port`. | docker/Caddyfile | +| `ARCHILYZER_FORWARD_AUTH_URI` | `/api/auth/caddy` | The forward-auth server's verify path. | docker/Caddyfile | +| `ARCHILYZER_TAG` | `local` | The image tag the compose files build and run. | docker-compose*.yml | + +## Tests only + +Read only by a test harness, a fake binary or a test-mode branch. Never set one on a real instance. + +| Variable | Default | What it does | Read by | +|---|---|---|---| +| `EDITOR_TEST_ROUTES` | off | `1` opens the editor's `/api/test/*` routes. The e2e server sets it. | editor/app/api/test/_guard.ts, editor/instrumentation.ts | +| `E2E_MODE` | dev | `start` runs the editor suite against `next start` instead of `next dev`. | editor/playwright.config.ts | +| `E2E_QUEUE` | on | `0` skips the machine-global e2e queue (the port check still runs). | scripts/queue-lock.mjs | +| `E2E_PORT_CHECK` | on | `0` skips the pre-run check that the suite's ports are free. | scripts/queue-lock.mjs | +| `E2E_QUEUE_TIMEOUT` | wait forever | Seconds to wait for the queue before giving up. | scripts/queue-lock.mjs | +| `E2E_PORT_GRACE_MS` | `3000` | How long the port check waits for a just-freed port. | scripts/queue-lock.mjs | +| `E2E_QUEUE_LOCK_FILE` | one per machine | The queue's lock file; the queue's own tests point it elsewhere. | scripts/queue-lock.mjs | +| `QUEUE_LOCK_HELD` | — | Set by the queue for the command it runs, so a nested wrapper passes through. | scripts/queue-lock.mjs | +| `PLAYWRIGHT_BASE_URL` | `http://localhost:<PORT>` | The editor test server's URL; the worktree injector sets it. | editor/playwright.config.ts, editor/e2e/baseUrl.ts | +| `AUDIO_CHECK_INTERVAL_MS_OVERRIDE` | the real cadence | Shrinks the mid-download audio check so the e2e suite sees it fire. | common/ytdlp/audioCheckedDownload.ts | +| `AUDIO_CHECK_SIZE_GATE_OVERRIDE` | the real gate | Likewise, the size gate. | common/ytdlp/audioCheckedDownload.ts | +| `AUDIO_CHECK_INTERVAL_FLOOR_MS_OVERRIDE` | the real floor | Likewise, the interval floor. | common/ytdlp/audioCheckedDownload.ts | +| `AUDIO_CHECK_RECOVER_STEP_MS_OVERRIDE` | the real step | Likewise, the recovery step. | common/ytdlp/audioCheckedDownload.ts | +| `AUDIO_CHECK_RECOVER_AFTER_OVERRIDE` | the real count | Likewise, the recovery count. | common/ytdlp/audioCheckedDownload.ts | +| `AUDIO_CHECK_DEBUG_PAUSE_MS` | off | A debugging pause inside the audio check. | common/ytdlp/audioCheckedDownload.ts | +| `FAKE_YTDLP_AUDIO_CHECK_MODE` | — | Fake yt-dlp: which audio-check scenario to act out. | editor/e2e/fixtures/bin/fake-ytdlp.mjs | +| `FAKE_YTDLP_CHUNK_DELAY_MS` | — | Fake yt-dlp: delay between written chunks. | editor/e2e/fixtures/bin/fake-ytdlp.mjs | +| `FAKE_YTDLP_CORRUPT_AFTER_CHUNK` | — | Fake yt-dlp: start corrupting after this chunk. | editor/e2e/fixtures/bin/fake-ytdlp.mjs | +| `FAKE_YTDLP_CORRUPT_RUNS` | — | Fake yt-dlp: how many runs corrupt. | editor/e2e/fixtures/bin/fake-ytdlp.mjs | +| `FAKE_YTDLP_DETERMINISTIC_CORRUPT` | — | Fake yt-dlp: corrupt deterministically. | editor/e2e/fixtures/bin/fake-ytdlp.mjs | +| `FAKE_YTDLP_RECOVER_ON_RESUME` | — | Fake yt-dlp: a resumed run recovers. | editor/e2e/fixtures/bin/fake-ytdlp.mjs | +| `FAKE_YTDLP_TOTAL_CHUNKS` | — | Fake yt-dlp: how many chunks a download has. | editor/e2e/fixtures/bin/fake-ytdlp.mjs | +| `FAKE_GALLERY_DL_AUTH_FAIL` | — | Fake gallery-dl: fail as an auth error. | editor/e2e/fixtures/bin/fake-gallery-dl.mjs | +| `FIXTURE_MAX_LIFETIME_MS` | the watchdog's | How long a fake binary may live before its watchdog kills it. | editor/e2e/fixtures/bin/_watchdog.mjs | +| `OLLAMA_STUB_MODEL` | `qwen2.5:7b` | The model the ollama stub claims to serve. | editor/e2e/fixtures/ollama-stub.mjs | +| `RACK_SHOTS` | off (spec skipped) | Runs the `/channels` rack screenshot audit. | editor/e2e/channels-rack-audit.spec.ts | +| `TWO_ORIGIN_REBUILD` | off | `1` rebuilds the two-origin suite's cached hub bundle. | export/e2e-2origin/globalSetup.ts | +| `IMAGE` | `yt-dlp-transcript-browser-e2e` | The sharded e2e run's image tag. | scripts/run-sharded-e2e.mjs | +| `SKIP_BUILD` | off | `1` reuses the sharded e2e image instead of rebuilding it (`--no-build`). | scripts/run-sharded-e2e.mjs | diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts @@ -177,6 +177,13 @@ export const COMMANDS: Command[] = [ run: async () => (await import("./sync-tick")).tick(), }, { + path: ["docs", "env"], + usage: "[--check] write ENVIRONMENT.md from the declared env-var list (lib/envVars.ts)", + flags: { check: "boolean" }, + run: async ({ flags }) => + (await import("./env-docs")).main({ check: flags.check === true }), + }, + { path: ["settings", "example"], usage: "[--check] write settings.json.example + SETTINGS.md from the schema", flags: { check: "boolean" }, diff --git a/common/bin/env-docs.ts b/common/bin/env-docs.ts @@ -0,0 +1,34 @@ +// WRITE ENVIRONMENT.MD FROM THE DECLARED LIST (lib/envVars.ts). +// +// archilyzer docs env [--check] +// +// `--check` writes nothing and returns 1 when the committed file differs from +// what the list generates (the same claim lib/envVars.test.ts makes). The +// sibling of file-schemas-docs.ts (SITE.md, CHANNEL.md) and settings-example.ts +// (SETTINGS.md). + +import { readFile, writeFile } from "node:fs/promises"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { renderEnvironmentMarkdown } from "../lib/envVars"; +import { runIfEntryPoint } from "./_cli"; + +const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", ".."); + +export async function main(opts: { check?: boolean } = {}): Promise<number> { + const file = path.join(REPO, "ENVIRONMENT.md"); + const want = renderEnvironmentMarkdown(); + if (opts.check) { + const have = await readFile(file, "utf8").catch(() => ""); + if (have !== want) { + console.error("ENVIRONMENT.md is stale — regenerate it with `archilyzer docs env`"); + return 1; + } + return 0; + } + await writeFile(file, want); + console.log("wrote ENVIRONMENT.md"); + return 0; +} + +runIfEntryPoint(import.meta.url, () => main({ check: process.argv.includes("--check") })); diff --git a/common/lib/envVars.test.ts b/common/lib/envVars.test.ts @@ -0,0 +1,128 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { readdirSync, readFileSync, statSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { ENV_AUDIENCES, ENV_VARS, renderEnvironmentMarkdown } from "./envVars"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common test +// +// envVars.ts is the one declared list of environment variables, and +// ENVIRONMENT.md is generated from it. These tests are what keep the list true: +// the code is read as text, in both directions. + +const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", ".."); + +// Where the apps' code lives. umtool is out of scope on purpose (envVars.ts +// says why); tests are skipped because a test SETS variables for itself. +const CODE_ROOTS = ["common", "editor", "export", "homepage", "mcp/src", "scripts"]; +const SKIP_DIRS = new Set(["node_modules", ".next", "out", "public", "test-results", "blob-report", "test-transcripts"]); + +// The platform's own variables: read here, documented by node, Next, a shell. +const PLATFORM = new Set(["CI", "NODE_ENV", "NEXT_RUNTIME", "LD_LIBRARY_PATH", "PATH", "HOME", "NODE_OPTIONS"]); + +function codeFiles(): string[] { + const out: string[] = []; + const walk = (dir: string) => { + for (const n of readdirSync(dir)) { + if (SKIP_DIRS.has(n) || n.startsWith(".")) continue; + const p = path.join(dir, n); + if (statSync(p).isDirectory()) walk(p); + else if (/\.(ts|tsx|mts|mjs|js)$/.test(n) && !/\.test\.(ts|mjs)$/.test(n)) out.push(p); + } + }; + for (const r of CODE_ROOTS) walk(path.join(REPO, r)); + return out; +} + +// Comment lines are dropped first: prose that names `process.env.NAME` as an +// example is not a read. +function codeText(file: string): string { + return readFileSync(file, "utf8") + .split("\n") + .filter((l) => !/^\s*(\/\/|\*|\/\*)/.test(l)) + .join("\n"); +} + +const READ_PATTERNS = [ + /process\.env\??\.([A-Z][A-Z0-9_]+)\b/g, + /process\.env\[["']([A-Z][A-Z0-9_]+)["']\]/g, + // `env.X` on an env object handed in (a spawn's env, a testable `env = + // process.env` parameter). + /\b[eE]nv\??\.([A-Z][A-Z0-9_]{2,})\b/g, + // audioCheckedDownload.ts reads its test overrides through a helper. + /envIntOverride\(["']([A-Z][A-Z0-9_]+)["']\)/g, +]; + +function reads(): Map<string, Set<string>> { + const byName = new Map<string, Set<string>>(); + for (const file of codeFiles()) { + const text = codeText(file); + for (const re of READ_PATTERNS) { + for (const m of text.matchAll(re)) { + const name = m[1]; + if (PLATFORM.has(name)) continue; + if (!byName.has(name)) byName.set(name, new Set()); + byName.get(name)!.add(path.relative(REPO, file)); + } + } + } + return byName; +} + +test("ENVIRONMENT.md is what envVars.ts generates", () => { + assert.equal( + readFileSync(path.join(REPO, "ENVIRONMENT.md"), "utf8"), + renderEnvironmentMarkdown(), + "ENVIRONMENT.md is stale: run `archilyzer docs env`", + ); +}); + +test("every variable the code reads is declared", () => { + const declared = new Set(ENV_VARS.map((v) => v.name)); + const missing = [...reads()] + .filter(([name]) => !declared.has(name)) + .map(([name, files]) => `${name} (read by ${[...files].join(", ")})`); + assert.deepEqual(missing, [], "declare these in common/lib/envVars.ts"); +}); + +test("every declared variable is still mentioned by the code, a script or a compose file", () => { + const corpus = [ + ...codeFiles().map((f) => readFileSync(f, "utf8")), + ...readdirSync(path.join(REPO, "docker")) + .filter((n) => statSync(path.join(REPO, "docker", n)).isFile()) + .map((n) => readFileSync(path.join(REPO, "docker", n), "utf8")), + ...readdirSync(REPO) + .filter((n) => /^docker-compose.*\.yml$|^Dockerfile/.test(n)) + .map((n) => readFileSync(path.join(REPO, n), "utf8")), + ...["", "editor", "export", "homepage"].map((d) => + readFileSync(path.join(REPO, d, "package.json"), "utf8"), + ), + ].join("\n"); + const stale = ENV_VARS.filter((v) => !new RegExp(`\\b${v.name}\\b`).test(corpus)).map((v) => v.name); + assert.deepEqual(stale, [], "nothing mentions these any more: delete their entries"); +}); + +test("the paths audience is exactly what getPaths() reads", () => { + const pathsTs = codeText(path.join(REPO, "common/lib/paths.ts")); + const inPaths = new Set([...pathsTs.matchAll(/process\.env\.([A-Z][A-Z0-9_]+)/g)].map((m) => m[1])); + const declared = new Set(ENV_VARS.filter((v) => v.audience === "paths").map((v) => v.name)); + assert.deepEqual([...declared].filter((n) => !inPaths.has(n)), [], "declared paths but not read by getPaths()"); + assert.deepEqual([...inPaths].filter((n) => !declared.has(n)), [], "read by getPaths() but not declared as paths"); +}); + +test("names are unique and every audience has a section", () => { + const names = ENV_VARS.map((v) => v.name); + assert.equal(new Set(names).size, names.length); + const audiences = new Set(ENV_AUDIENCES.map((a) => a.id)); + for (const v of ENV_VARS) assert.ok(audiences.has(v.audience), v.name); + const md = renderEnvironmentMarkdown(); + for (const v of ENV_VARS) assert.ok(md.includes(`| \`${v.name}\` |`), v.name); +}); + +test("the docker audience is the ARCHILYZER_ set", () => { + for (const v of ENV_VARS.filter((x) => x.audience === "docker")) { + assert.match(v.name, /^ARCHILYZER_/); + } +}); diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts @@ -0,0 +1,225 @@ +// EVERY ENVIRONMENT VARIABLE THE REPO READS — the one declared list. +// +// ENVIRONMENT.md is generated from this (common/bin/env-docs.ts, `--check` in +// the common tests), and `envVars.test.ts` holds the list to the code: every +// variable read by common/, editor/, export/, homepage/, mcp/src and scripts/ is +// declared here, and every entry here is still read somewhere. A variable that +// is added without an entry, or deleted with its entry left behind, fails the +// build. umtool's own knobs are NOT here — its song and report scripts read +// dozens, documented in umtool/docs, and fold into the core in one-core Phase 5. +// +// THE AUDIENCES, which are the point of the table: +// paths — the ONE override surface for where things live and which binary +// runs: every one is read by getPaths() (lib/paths.ts) and nowhere +// else. The test checks both directions. +// runtime — the other knobs, secrets and tokens a running process reads. +// port — generated from lib/ports.mjs, never listed twice. +// internal — set BY the pipeline for a process it spawns. Listed so a reader +// knows what it is; nobody sets it by hand. +// docker — the container's ARCHILYZER_* set, read by docker/*.sh, the +// compose files and Caddy — a different process from the apps, +// documented in RUNNING_IN_DOCKER.md. +// test — read only by a test harness, a fake binary or a test-mode branch. +// +// Pure data (no imports but the port table), so a doc generator and a doctor can +// both read it without loading anything else. + +import { PORTS } from "./ports.mjs"; + +export type EnvAudience = "paths" | "runtime" | "port" | "internal" | "docker" | "test"; + +export type EnvVarDecl = { + name: string; + audience: EnvAudience; + // What it does, one or two sentences. + doc: string; + // What an unset variable means, in words (a value, "off", "—"). + default: string; + // Where it is read — a file, or a short list of them. + readBy: string; +}; + +const paths = (name: string, def: string, doc: string): EnvVarDecl => ({ + name, + audience: "paths", + doc, + default: def, + readBy: "common/lib/paths.ts (getPaths)", +}); + +const DECLARED: EnvVarDecl[] = [ + // ── paths: getPaths() ────────────────────────────────────────────────── + paths("TRANSCRIPTS_DIR", "`<repo>/transcripts`", "The corpus: channels, sites, the LMDB index, job logs, the saved-video store."), + paths("SAVED_VIDEOS_DIR", "`<TRANSCRIPTS_DIR>/saved-videos`", "The persisted source-video store, when it should live on another disk."), + paths("SITES_DIR", "`<TRANSCRIPTS_DIR>/sites`", "Per-site config (`<id>/site.json`, every key in [SITE.md](SITE.md)) and the homepage's `_homepage/`."), + paths("SETTINGS_FILE", "`<repo>/settings.json`", "The settings file (every key in [SETTINGS.md](SETTINGS.md))."), + paths("EXPORT_PUBLIC_DIR", "`<repo>/export/public`", "The dir the export site serves at `/`, composed one site at a time."), + paths("EXPORT_INDEX_DIR", "`.export-index` beside `EXPORT_PUBLIC_DIR`", "The build's staging area (not served): the shared index and per-site aggregates."), + paths("EXPORT_BUILDS_DIR", "`.export-builds` beside `EXPORT_PUBLIC_DIR`", "Per-site `out/` bundles from a docker-mode build."), + paths("EDITOR_CHANGELOG_FILE", "`<repo>/editor/CHANGELOG.md`", "The editor changelog the release cutter reads and rewrites. The e2e server points it at a gitignored copy."), + paths("EXPORT_CHANGELOG_FILE", "`<repo>/export/CHANGELOG.md`", "The export changelog, likewise."), + paths("CHARTS_CONFIG_FILE", "`<repo>/chart-templates.json`", "The legacy chart-templates file, read only by a migration."), + paths("SEARCH_ALIASES_FILE", "`<TRANSCRIPTS_DIR>/search-aliases.json`", "The corpus-wide search-alias dictionary."), + paths("CURATED_TAGS_FILE", "`<TRANSCRIPTS_DIR>/tags.json`", "Curated per-video tags. Written only through `applyTagAssignments`."), + paths("YTDLP_BIN", "`yt-dlp` on PATH", "The downloader. Every fetch goes through it."), + paths("FFMPEG_BIN", "`ffmpeg` on PATH", "Audio extraction for transcription and diarization."), + paths("FFPROBE_BIN", "`ffprobe` on PATH", "Duration checks (the short-audio guard, windowing)."), + paths("WHISPER_BIN", "`whisper-cli` on PATH", "whisper.cpp, one of the three transcription engines (with chough and parakeet.cpp)."), + paths("WHISPER_MODEL", "`~/whispercpp/whisper.cpp/models/ggml-base.en.bin`", "whisper.cpp's model, when a worker names none."), + paths("PARAKEET_STITCH_BIN", "`<repo>/scripts/parakeet-stitch.mjs`", "The parakeet.cpp engine's wrapper (overlapping windows, stitched)."), + paths("PARAKEET_CLI", "`parakeet-cli` on PATH", "The parakeet.cpp binary the wrapper drives (the wrapper reads it too)."), + paths("PARAKEET_MODEL", "none", "parakeet.cpp's `.gguf`, when a worker names none (the wrapper reads it too)."), + paths("DIARIZE_BIN", "`<repo>/scripts/diarize.mjs`", "The speaker-diarization wrapper. The e2e suite swaps in a fake here."), + paths("RSYNC_BIN", "`rsync` on PATH", "Mirrors the saved-video store to a backup destination."), + paths("FINDMNT_BIN", "`findmnt` on PATH", "The read-only volume-identity probe behind storage locations. Optional."), + paths("UDISKSCTL_BIN", "`udisksctl` on PATH", "Mounts an attached volume from `/storage`. Optional."), + paths("GALLERY_DL_BIN", "`gallery-dl` on PATH", "The X/Twitter post fetcher, for social channels."), + paths("OLLAMA_URL", "`http://127.0.0.1:11434`", "The local ollama server, the local digest and attribution engine."), + paths("CLAUDE_BIN", "`claude` on PATH", "The `claude` CLI, driving the opt-in metered digest lane."), + + // ── runtime ──────────────────────────────────────────────────────────── + { name: "WORKER_TOKEN", audience: "runtime", default: "unset (both surfaces off)", readBy: "common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts", doc: "Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`, the MCP's `fetch_clip`). Set the same value on both ends." }, + { name: "SYNC_HEARTBEAT_SECONDS", audience: "runtime", default: "`settings.syncScheduler.heartbeatSeconds`", readBy: "editor/app/scheduler/heartbeat.ts", doc: "Overrides the editor's in-process sync heartbeat. `0` = no internal timer (tick from cron instead)." }, + { name: "SYNC_TICK_URL", audience: "runtime", default: "`http://127.0.0.1:3001/api/scheduler/tick`", readBy: "common/bin/sync-tick.ts", doc: "Where `archilyzer sync tick` (cron's heartbeat) posts." }, + { name: "SYNC_TICK_TOKEN", audience: "runtime", default: "unset (no auth)", readBy: "common/bin/sync-tick.ts, editor/app/scheduler/auth.ts", doc: "Bearer token for the tick endpoint; set on both the editor and the cron job." }, + { name: "R2_ACCESS_KEY_ID", audience: "runtime", default: "—", readBy: "common/publish/build.ts", doc: "R2 S3 credentials for uploading oversize archives at deploy time (with `R2_SECRET_ACCESS_KEY` and `CLOUDFLARE_ACCOUNT_ID`). See [PUBLISH.md](PUBLISH.md)." }, + { name: "R2_SECRET_ACCESS_KEY", audience: "runtime", default: "—", readBy: "common/publish/build.ts", doc: "See `R2_ACCESS_KEY_ID`." }, + { name: "CLOUDFLARE_ACCOUNT_ID", audience: "runtime", default: "—", readBy: "common/publish/build.ts", doc: "The account the R2 endpoint belongs to. wrangler reads its own credentials." }, + { name: "DOCKER_BIN", audience: "runtime", default: "`docker`", readBy: "common/publish/build.ts", doc: "The container engine for docker-mode builds (e.g. `podman`)." }, + { name: "DOCKER_BUILD_MEMORY", audience: "runtime", default: "no cap", readBy: "common/publish/build.ts", doc: "Per-container memory cap for a docker-mode build (`--memory`)." }, + { name: "DOCKER_BUILD_CPUS", audience: "runtime", default: "no cap", readBy: "common/publish/build.ts", doc: "Per-container CPU cap for a docker-mode build (`--cpus`)." }, + { name: "ARCHIVE_CHANNEL_CONCURRENCY", audience: "runtime", default: "`4`", readBy: "common/bin/build-archives.ts", doc: "How many channels' archive zips `build archives` builds at once." }, + { name: "MAX_ARCHIVE_BYTES", audience: "runtime", default: "the Cloudflare-safe cap", readBy: "common/bin/compose-site.ts", doc: "The served-file size cap for archives, in bytes; `0` = no cap. A site's own `archiveMaxBytes` wins." }, + { name: "CHOUGH_BIN", audience: "runtime", default: "`chough` on PATH", readBy: "common/lib/transcriptionApps.ts", doc: "The chough transcription engine, when a worker names no binary." }, + { name: "CHOUGH_MODEL", audience: "runtime", default: "chough's own", readBy: "chough (set by common/lib/transcriptionApps.ts)", doc: "Passed to chough from a worker's model field; chough auto-downloads one when unset." }, + { name: "CHOUGH_URL", audience: "runtime", default: "local", readBy: "chough (set by common/lib/transcriptionApps.ts)", doc: "Passed to chough from a worker's remote-server field." }, + { name: "OLLAMA_DIGEST_MODEL", audience: "runtime", default: "`qwen2.5:7b`", readBy: "common/lib/digestApps.ts", doc: "The ollama model the local digest lane asks for when settings name none." }, + { name: "CLAUDE_DIGEST_MODEL", audience: "runtime", default: "the CLI's default", readBy: "common/lib/digestApps.ts", doc: "The model the metered digest lane asks `claude` for when settings name none." }, + { name: "NITTER_INSTANCES", audience: "runtime", default: "a built-in list", readBy: "common/social/xNitterFetcher.ts", doc: "Comma-separated Nitter instances for the X fallback fetcher, in order of preference." }, + { name: "UMTOOL_URL", audience: "runtime", default: "unset (no link)", readBy: "editor/app/channels/[slug]/videos/[id]/page.tsx", doc: "umtool's front door; when set, the video page links to it." }, + { name: "TRANSCRIPT_SITE_URL", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: one published archive to read over HTTP." }, + { name: "TRANSCRIPT_HUB_URL", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: a hub, federating every archive it lists." }, + { name: "TRANSCRIPT_LOCAL_DIR", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: a composed public dir on disk." }, + { name: "TRANSCRIPT_PLATFORM_LINKS", audience: "runtime", default: "off", readBy: "common/lib/archive/reader-fs.ts", doc: "`1` cites platform watch pages instead of the archive's own pages." }, + { name: "MCP_IO_STATS", audience: "runtime", default: "off", readBy: "common/lib/archive/io-stats.ts", doc: "`1` turns on per-call I/O accounting, for `mcp/bench`." }, + { name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool", doc: "Which editor `pnpm ops` and the MCP's `fetch_clip` talk to." }, + { name: "ARCHILYZER_AGENT", audience: "runtime", default: "`cli`", readBy: "scripts/archilyzer-ops.mjs", doc: "Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`." }, + { name: "DIARIZE_ENGINE_KIND", audience: "runtime", default: "`sherpa-onnx`", readBy: "scripts/diarize.mjs", doc: "The diarization engine: `sherpa-onnx` or `sortformer`." }, + { name: "DIARIZE_ENGINE_CMD", audience: "runtime", default: "the bundled sherpa script", readBy: "scripts/diarize.mjs", doc: "The engine command the wrapper runs." }, + { name: "DIARIZE_PYTHON", audience: "runtime", default: "`python3`", readBy: "scripts/diarize.mjs", doc: "The python for the default engine." }, + { name: "DIARIZE_SEG_MODEL", audience: "runtime", default: "— (required)", readBy: "scripts/diarize.mjs", doc: "Segmentation model. The editor passes the settings' value as a flag." }, + { name: "DIARIZE_EMB_MODEL", audience: "runtime", default: "— (required)", readBy: "scripts/diarize.mjs", doc: "Speaker-embedding model. The editor passes the settings' value as a flag." }, + { name: "DIARIZE_THRESHOLD", audience: "runtime", default: "`0.5`", readBy: "scripts/diarize.mjs", doc: "Clustering threshold." }, + { name: "DIARIZE_THREADS", audience: "runtime", default: "`4`", readBy: "scripts/diarize.mjs", doc: "Engine threads." }, + { name: "DIARIZE_WINDOW_MINUTES", audience: "runtime", default: "`45`", readBy: "scripts/diarize.mjs", doc: "Window length for long files; `0` never windows." }, + { name: "DIARIZE_WINDOW_AFTER_MINUTES", audience: "runtime", default: "`90`", readBy: "scripts/diarize.mjs", doc: "Only files longer than this are windowed." }, + { name: "SORTFORMER_BIN", audience: "runtime", default: "— (required for sortformer)", readBy: "scripts/diarize.mjs, scripts/diarize-sortformer.mjs", doc: "The sortformer engine binary." }, + { name: "SORTFORMER_MODEL", audience: "runtime", default: "— (required for sortformer)", readBy: "scripts/diarize.mjs, scripts/diarize-sortformer.mjs", doc: "The sortformer `.gguf`." }, + { name: "PARAKEET_SEGMENT_SEC", audience: "runtime", default: "`480`", readBy: "scripts/parakeet-stitch.mjs", doc: "parakeet window length, seconds (a worker's chunk size wins)." }, + { name: "PARAKEET_OVERLAP_SEC", audience: "runtime", default: "`6`", readBy: "scripts/parakeet-stitch.mjs", doc: "parakeet window overlap, seconds." }, + { name: "PARAKEET_DECODER", audience: "runtime", default: "parakeet-cli's", readBy: "scripts/parakeet-stitch.mjs", doc: "`ctc` or `tdt`, passed through to parakeet-cli." }, + { name: "PARAKEET_LANG", audience: "runtime", default: "parakeet-cli's", readBy: "scripts/parakeet-stitch.mjs", doc: "A locale, passed through to parakeet-cli." }, + { name: "PARAKEET_DEVICE", audience: "runtime", default: "parakeet-cli's", readBy: "scripts/parakeet-stitch.mjs", doc: "Compute device (`cpu`, `CUDA0`, `Vulkan1`, …), exported to parakeet-cli." }, + + // ── internal: the pipeline sets these for a process it spawns ────────── + { name: "SITE_ID", audience: "internal", default: "—", readBy: "common/bin/compose-site.ts, export/app/lib/site.ts", doc: "Which site a compose or an export build is for. `archilyzer build site <id>` sets it; `compose site` and `build site` fall back to it when no id is given." }, + { name: "INSTANCE_MODE", audience: "internal", default: "a site", readBy: "export/app/lib/mode.ts, common/lib/archive/contract.ts", doc: "`hub` makes the export build the hub. Set by `archilyzer build hub`." }, + { name: "BUILD_ARCHIVES", audience: "internal", default: "on", readBy: "common/bin/compose-site.ts, common/bin/build-archives.ts", doc: "`0` skips archive-zip generation for one build (`--skip-archives`)." }, + { name: "ARCHIVES_READONLY", audience: "internal", default: "off", readBy: "common/bin/compose-site.ts", doc: "`1` inside a docker-mode build container: materialize archives, never write the shared cache." }, + { name: "HOMEPAGE_PUBLIC_DIR", audience: "internal", default: "`<repo>/homepage/public`", readBy: "common/bin/compose-homepage.ts", doc: "Where `compose homepage` writes." }, + { name: "HOMEPAGE_SUMMARY_FILE", audience: "internal", default: "`homepage/public/homepage-summary.json`", readBy: "homepage/app/lib/summary.ts", doc: "A dev-only summary file for the homepage; ignored by a production build. The homepage e2e sets it." }, + + // ── docker: the container's set ──────────────────────────────────────── + { name: "ARCHILYZER_TRANSCRIBER", audience: "docker", default: "baked per image target (`whisper-cpp` in `runtime`)", readBy: "docker/entrypoint.sh", doc: "`whisper-cpp` or `parakeet`: which worker the first boot seeds and which model it fetches." }, + { name: "ARCHILYZER_FETCH_MODEL", audience: "docker", default: "per transcriber", readBy: "docker/entrypoint.sh", doc: "Which model the first boot downloads; `none` skips it." }, + { name: "ARCHILYZER_MODELS_DIR", audience: "docker", default: "`/data/models`", readBy: "docker/entrypoint.sh", doc: "Where models live in the container." }, + { name: "ARCHILYZER_BUILDS_DIR", audience: "docker", default: "`/data/builds`", readBy: "docker/entrypoint.sh", doc: "Where the container keeps built sites." }, + { name: "ARCHILYZER_SITE_OUT", audience: "docker", default: "`/data/builds/site`", readBy: "docker/entrypoint.sh, docker/publish-site.sh", doc: "The built export site the `site` service serves." }, + { name: "ARCHILYZER_IDLE_BOOT", audience: "docker", default: "off", readBy: "common/lib/idleBoot.ts (the editor)", doc: "`1` boots the editor without arming the heartbeat or any auto-queue runner." }, + { name: "ARCHILYZER_AUTH_MODE", audience: "docker", default: "`basic`", readBy: "docker/guard-exposure.sh, docker/caddy-start.sh", doc: "`basic`, `forward` or `none` — the only escape hatch from the exposure guard." }, + { name: "ARCHILYZER_AUTH_USER", audience: "docker", default: "`archilyzer`", readBy: "docker/Caddyfile", doc: "Basic-auth user." }, + { name: "ARCHILYZER_AUTH_HASH", audience: "docker", default: "—", readBy: "docker/Caddyfile, docker/guard-exposure.sh", doc: "Basic-auth bcrypt hash (`caddy hash-password`)." }, + { name: "ARCHILYZER_AUTH_IMPORT", audience: "docker", default: "derived from the mode", readBy: "docker/Caddyfile", doc: "Set by docker/caddy-start.sh from the mode: which auth snippet the private sites import." }, + { name: "ARCHILYZER_FORWARD_AUTH_UPSTREAM", audience: "docker", default: "—", readBy: "docker/Caddyfile", doc: "Forward-auth server (Authelia, tinyauth, …), `host:port`." }, + { name: "ARCHILYZER_FORWARD_AUTH_URI", audience: "docker", default: "`/api/auth/caddy`", readBy: "docker/Caddyfile", doc: "The forward-auth server's verify path." }, + { name: "ARCHILYZER_TAG", audience: "docker", default: "`local`", readBy: "docker-compose*.yml", doc: "The image tag the compose files build and run." }, + + // ── test: harnesses, fakes and test-mode branches ────────────────────── + { name: "EDITOR_TEST_ROUTES", audience: "test", default: "off", readBy: "editor/app/api/test/_guard.ts, editor/instrumentation.ts", doc: "`1` opens the editor's `/api/test/*` routes. The e2e server sets it." }, + { name: "E2E_MODE", audience: "test", default: "dev", readBy: "editor/playwright.config.ts", doc: "`start` runs the editor suite against `next start` instead of `next dev`." }, + { name: "E2E_QUEUE", audience: "test", default: "on", readBy: "scripts/queue-lock.mjs", doc: "`0` skips the machine-global e2e queue (the port check still runs)." }, + { name: "E2E_PORT_CHECK", audience: "test", default: "on", readBy: "scripts/queue-lock.mjs", doc: "`0` skips the pre-run check that the suite's ports are free." }, + { name: "E2E_QUEUE_TIMEOUT", audience: "test", default: "wait forever", readBy: "scripts/queue-lock.mjs", doc: "Seconds to wait for the queue before giving up." }, + { name: "E2E_PORT_GRACE_MS", audience: "test", default: "`3000`", readBy: "scripts/queue-lock.mjs", doc: "How long the port check waits for a just-freed port." }, + { name: "E2E_QUEUE_LOCK_FILE", audience: "test", default: "one per machine", readBy: "scripts/queue-lock.mjs", doc: "The queue's lock file; the queue's own tests point it elsewhere." }, + { name: "QUEUE_LOCK_HELD", audience: "test", default: "—", readBy: "scripts/queue-lock.mjs", doc: "Set by the queue for the command it runs, so a nested wrapper passes through." }, + { name: "PLAYWRIGHT_BASE_URL", audience: "test", default: "`http://localhost:<PORT>`", readBy: "editor/playwright.config.ts, editor/e2e/baseUrl.ts", doc: "The editor test server's URL; the worktree injector sets it." }, + { name: "AUDIO_CHECK_INTERVAL_MS_OVERRIDE", audience: "test", default: "the real cadence", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Shrinks the mid-download audio check so the e2e suite sees it fire." }, + { name: "AUDIO_CHECK_SIZE_GATE_OVERRIDE", audience: "test", default: "the real gate", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Likewise, the size gate." }, + { name: "AUDIO_CHECK_INTERVAL_FLOOR_MS_OVERRIDE", audience: "test", default: "the real floor", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Likewise, the interval floor." }, + { name: "AUDIO_CHECK_RECOVER_STEP_MS_OVERRIDE", audience: "test", default: "the real step", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Likewise, the recovery step." }, + { name: "AUDIO_CHECK_RECOVER_AFTER_OVERRIDE", audience: "test", default: "the real count", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Likewise, the recovery count." }, + { name: "AUDIO_CHECK_DEBUG_PAUSE_MS", audience: "test", default: "off", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "A debugging pause inside the audio check." }, + { name: "FAKE_YTDLP_AUDIO_CHECK_MODE", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: which audio-check scenario to act out." }, + { name: "FAKE_YTDLP_CHUNK_DELAY_MS", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: delay between written chunks." }, + { name: "FAKE_YTDLP_CORRUPT_AFTER_CHUNK", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: start corrupting after this chunk." }, + { name: "FAKE_YTDLP_CORRUPT_RUNS", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: how many runs corrupt." }, + { name: "FAKE_YTDLP_DETERMINISTIC_CORRUPT", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: corrupt deterministically." }, + { name: "FAKE_YTDLP_RECOVER_ON_RESUME", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: a resumed run recovers." }, + { name: "FAKE_YTDLP_TOTAL_CHUNKS", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: how many chunks a download has." }, + { name: "FAKE_GALLERY_DL_AUTH_FAIL", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-gallery-dl.mjs", doc: "Fake gallery-dl: fail as an auth error." }, + { name: "FIXTURE_MAX_LIFETIME_MS", audience: "test", default: "the watchdog's", readBy: "editor/e2e/fixtures/bin/_watchdog.mjs", doc: "How long a fake binary may live before its watchdog kills it." }, + { name: "OLLAMA_STUB_MODEL", audience: "test", default: "`qwen2.5:7b`", readBy: "editor/e2e/fixtures/ollama-stub.mjs", doc: "The model the ollama stub claims to serve." }, + { name: "RACK_SHOTS", audience: "test", default: "off (spec skipped)", readBy: "editor/e2e/channels-rack-audit.spec.ts", doc: "Runs the `/channels` rack screenshot audit." }, + { name: "TWO_ORIGIN_REBUILD", audience: "test", default: "off", readBy: "export/e2e-2origin/globalSetup.ts", doc: "`1` rebuilds the two-origin suite's cached hub bundle." }, + { name: "IMAGE", audience: "test", default: "`yt-dlp-transcript-browser-e2e`", readBy: "scripts/run-sharded-e2e.mjs", doc: "The sharded e2e run's image tag." }, + { name: "SKIP_BUILD", audience: "test", default: "off", readBy: "scripts/run-sharded-e2e.mjs", doc: "`1` reuses the sharded e2e image instead of rebuilding it (`--no-build`)." }, +]; + +// The port rows come from the port table, so a port is declared once. +const PORT_ROWS: EnvVarDecl[] = Object.entries(PORTS).map(([name, decl]) => ({ + name, + audience: "port", + doc: `${decl.what[0].toUpperCase()}${decl.what.slice(1)}. A worktree adds its offset (\`pnpm wt list\`).`, + default: `\`${decl.base}\``, + readBy: "common/lib/ports.mjs", +})); + +export const ENV_VARS: readonly EnvVarDecl[] = [...DECLARED, ...PORT_ROWS]; + +export const ENV_AUDIENCES: ReadonlyArray<{ id: EnvAudience; title: string; intro: string }> = [ + { id: "paths", title: "Paths and binaries", intro: "The one override surface for where things live and which binary runs. Every one is read by `getPaths()` (`common/lib/paths.ts`) and nowhere else; nothing hardcodes a location." }, + { id: "runtime", title: "Runtime", intro: "Tokens, credentials and knobs a running process reads. Most configuration is not here but in `settings.json` ([SETTINGS.md](SETTINGS.md))." }, + { id: "port", title: "Ports", intro: "Every local server's default port, from `common/lib/ports.mjs`. The primary checkout uses these; worktree N adds N × 100 (`pnpm wt list`)." }, + { id: "internal", title: "Set by the pipeline", intro: "The publish pipeline sets these for a process it spawns. Listed so a reader knows what they are; nobody sets them by hand." }, + { id: "docker", title: "Docker", intro: "The container's own set, read by `docker/*.sh`, the compose files and Caddy — not by the apps' code (except `ARCHILYZER_IDLE_BOOT`). See [RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md)." }, + { id: "test", title: "Tests only", intro: "Read only by a test harness, a fake binary or a test-mode branch. Never set one on a real instance." }, +]; + +export function envVar(name: string): EnvVarDecl | undefined { + return ENV_VARS.find((v) => v.name === name); +} + +// ENVIRONMENT.md — one table per audience. +export function renderEnvironmentMarkdown(): string { + const cell = (s: string) => s.replace(/\|/g, "\\|"); + const out: string[] = [ + "# Environment variables", + "", + "<!-- GENERATED by common/bin/env-docs.ts from common/lib/envVars.ts — do not edit by hand. -->", + "", + "Every environment variable the repo's code reads, by who it is for. The list is code (`common/lib/envVars.ts`), and a test fails when the code reads a variable the list does not declare, or the list declares one nothing reads. umtool's own knobs are documented in [umtool/docs](umtool/docs/README.md).", + "", + "Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/archilyzer.ts docs env`. `archilyzer doctor` prints which of the paths overrides are set on this machine.", + "", + ]; + for (const a of ENV_AUDIENCES) { + const rows = ENV_VARS.filter((v) => v.audience === a.id); + out.push(`## ${a.title}`, "", a.intro, "", "| Variable | Default | What it does | Read by |", "|---|---|---|---|"); + for (const v of rows) { + out.push(`| \`${v.name}\` | ${cell(v.default)} | ${cell(v.doc)} | ${cell(v.readBy)} |`); + } + out.push(""); + } + return out.join("\n"); +}