commit d481ba6a7a1c83e36ad0ec21fac25917ca456956
parent 54cb4f2c228c680e023c141aa4b4e972c3ffc089
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 28 Sep 2026 02:54:13 -0400
Merge r11/phase-4-s3 (checkpoint A) — release 11 slice O6, one-core Phase 4 slice 3: archilyzer doctor / run <operation> / mcp, every bin a subcommand, the ports and the env vars declared once (ENVIRONMENT.md generated), PUBLISH.md absorbs DEPLOY_CLOUDFLARE.md + DEPLOY_DOCKER.md; the build mode's copy says it is a label; the E2E_ rename (checkpoint B) still to come
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
55 files changed, 3259 insertions(+), 688 deletions(-)
diff --git a/AGENTS.md b/AGENTS.md
@@ -195,8 +195,12 @@ Two things the image cannot bake, and the reasons matter:
the `site` service serves.
- **whisper models.** 142 MB to 3 GB, and the choice is the operator's.
`docker/entrypoint.sh` fetches one on first boot — and seeds a `settings.json`
- carrying one enabled worker, because `defaults()` returns `workers: []` and zero
- workers means auto-transcribe silently does nothing.
+ carrying one enabled worker for the image's engine: with no `workers` key (or no
+ file), `getSettings()` synthesizes `parallelTranscriptions` (default 2) enabled
+ workers of the default app, whisper.cpp — two CPU whisper slots, and never
+ parakeet in the Vulkan image. (`defaults()` alone has `workers: []`; zero workers
+ — auto-transcribe silently doing nothing — only happens for a file that says
+ `"workers": []`.)
`ARCHILYZER_IDLE_BOOT=1` boots the editor without arming the heartbeat or any of
the four auto-queue lane runners (`common/lib/idleBoot.ts`) — for pointing a fresh
@@ -208,7 +212,7 @@ back to the serial host build it already handles. Do not try to make
docker-in-docker work.
See [RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md) — which is about *running the
-apps*, not [DEPLOY_DOCKER.md](DEPLOY_DOCKER.md), which is about *building sites*.
+apps*, not [PUBLISH.md](PUBLISH.md), which is about *building and publishing sites*.
# The corpus, and where the live sites are configured
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
@@ -34,7 +34,8 @@ pnpm install
pnpm dev:editor # editor at http://localhost:3001
pnpm build # build the static site under export/out/
pnpm start:export # serve export/out/ at http://localhost:3000
-pnpm build:index # rebuild the search index in-process
+pnpm build:index # rebuild the search index in-process (= pnpm archilyzer index)
+pnpm archilyzer doctor # what this machine has: tools, corpus, settings, ports
```
`pnpm lint` at the root only lints `export/`. **The editor has no eslint config**, so
@@ -128,18 +129,29 @@ buttons" is red under `next dev` (which injects a Dev Tools button) and green un
**The editor e2e suite runs in dev mode and therefore never prerenders.** A change to
a layout or a client component is not verified until `pnpm build` passes too.
-## CLI shims
+## The `archilyzer` CLI
-The same controllers the editor uses are exposed as terminal shims under
-`common/bin/`, which is how you drive the pipeline headlessly or from cron:
+The same controllers the editor uses are one command line, `common/bin/archilyzer.ts`,
+which is how you drive the pipeline headlessly or from cron. `pnpm archilyzer
+<command>` from the repo root is the short form of `pnpm --filter
+yt-dlp-transcript-common exec tsx bin/archilyzer.ts <command>`; `pnpm archilyzer
+--help` lists every command.
```bash
-pnpm --filter yt-dlp-transcript-common exec tsx bin/build-index.ts
-pnpm --filter yt-dlp-transcript-common exec tsx bin/transform.ts --channel <slug>
-pnpm --filter yt-dlp-transcript-common exec tsx bin/retry-failures.ts --channel <slug>
-pnpm --filter yt-dlp-transcript-common exec tsx bin/verify-transcripts.ts --channel <slug>
+pnpm archilyzer doctor # read-only: can this machine do what it is configured to?
+pnpm archilyzer index # the LMDB index
+pnpm archilyzer run diarization <channel> [ids…] # one catalogued operation, offline, as the editor's job
+pnpm archilyzer build site <id> # publish: see PUBLISH.md
+pnpm archilyzer verify transcripts --channel <slug>
+pnpm archilyzer mcp # the MCP server on stdio
```
+The table is `archilyzer.ts`; the machinery (parser, lookup, usage) is `_cli.ts`. Every
+file in `common/bin/` is reachable from a row — a test fails otherwise. A bin that
+parses its own flags is a *passthrough* row, run as a child with its argv untouched.
+`run` refuses sync, the metadata scan, downloads and transcription: they run on the
+editor's paced download queue and worker pool, which a second process must not race.
+
## Internals worth knowing
**This is not the Next.js you may know.** The workspace tracks a recent major and its
@@ -155,7 +167,11 @@ deprecation notices.
`export/public/{summaries,transcripts}/` and renders a search/filter UI into a
static `out/`.
- `getPaths()` (`common/lib/paths.ts`) is the single resolver for every path and
- binary. Nothing should hardcode a location; add an env override there instead.
+ binary. Nothing should hardcode a location; add an env override there instead, and
+ declare it in `common/lib/envVars.ts` (a test fails until you do, and
+ `pnpm archilyzer docs env` regenerates [ENVIRONMENT.md](ENVIRONMENT.md)). A new
+ local server's port goes in `common/lib/ports.mjs`, which `pnpm wt` offsets per
+ worktree.
- `common/lib/project.ts` holds **product** identity (the name, the project URL) and
is deliberately import-free. **Operator** identity — what a given deployment calls
itself — lives in settings and per-site config. A string that should change when
diff --git a/DEPLOY_CLOUDFLARE.md b/DEPLOY_CLOUDFLARE.md
@@ -1,332 +0,0 @@
-# Deploying to Cloudflare (Pages + R2 archive overflow)
-
-An **Archilyzer** site is deployed to **Cloudflare Pages**; large download archives that
-exceed Pages' per-file limit overflow to **Cloudflare R2**. This guide covers the R2
-setup and — importantly — how to configure Cloudflare so that **public archive
-downloads can't be abused to drive up your bill**.
-
-Everything here fits inside Cloudflare's **free tier**.
-
-- [How archives are served](#how-archives-are-served)
-- [Preview deployments](#preview-deployments)
-- [One-time R2 setup](#one-time-r2-setup)
-- [Securing downloads against cost-abuse](#securing-downloads-against-cost-abuse)
-- [Cost expectations](#cost-expectations)
-
----
-
-## How archives are served
-
-At build time, `compose-site.ts` generates one transcript zip and one live-chat zip
-**per channel** into `export/public/archives/`, and records them in
-`public/archives/manifest.json`. The `/downloads` page and the header **Downloads**
-link read that manifest.
-
-Cloudflare Pages rejects any single asset larger than **25 MB**. Real channels blow
-past that easily (a channel's live-chat zip can be hundreds of MB). So the pipeline
-splits archives by size:
-
-| Archive size | Where it's served from | Manifest entry |
-|---|---|---|
-| ≤ 25 MB | Cloudflare **Pages** (shipped in `out/`, free) | `filename`, no `url` |
-| > 25 MB, R2 configured | Cloudflare **R2**, uploaded on deploy | `url` → R2 |
-| > 25 MB, R2 **not** configured | not served | `oversize: true`, shown as "Too large to host" |
-
-Oversize archives are staged during compose into `export/.r2-staging/<siteId>/archives/`
-(gitignored, kept out of `public/`), then uploaded by the editor's **Deploy** /
-**Build & deploy** actions to `<bucket>/<siteId>/archives/<file>.zip` **before** the
-Pages deploy runs, so the manifest URLs resolve immediately.
-
-> **Note:** every deploy path uploads oversize archives to R2 before the Pages
-> deploy: the editor's **Deploy** / **Build & deploy**, `pnpm ops deploy-site`, and
-> `archilyzer deploy site <id>` (which `pnpm run deploy` in `export/` runs, with
-> `SITE_ID`). A build alone (`archilyzer build site`, `pnpm run build`) only
-> *stages* them in `export/.r2-staging/`. The bucket is read from `settings.json`;
-> the R2 credentials come from the environment (see "3. Authenticate" below).
-
-Uploads go through R2's **S3 API** (via the AWS SDK's multipart uploader), not
-`wrangler r2 object put` — wrangler caps a single upload at **300 MiB**, and real
-live-chat archives are larger (multipart has no such limit). That's why archive
-uploads need S3 credentials (step 3) in addition to the wrangler auth your Pages
-deploy already uses.
-
----
-
-## Preview deployments
-
-A **preview** is the same built bundle deployed to a branch that is not the Pages
-project's production branch. Cloudflare publishes it at a **branch alias** —
-
-```
-https://<branch>.<project>.pages.dev
-```
-
-— and leaves the live site alone. Each deploy also gets an immutable
-per-deployment URL (`https://<hash>.<project>.pages.dev`), which wrangler prints as
-"Deployment complete! Take a peek over at …"; the editor repeats both on a
-`[preview]` line at the end of the job log, because the streamed log scrolls.
-
-The alias is a function of the project and the branch and nothing else, so it is
-known *before* the deploy runs — which is why the editor can link it while you are
-still typing the name.
-
-**Branch names** must be 1–28 lowercase letters, digits and dashes, starting and
-ending with a letter or digit. That is exactly what Cloudflare's alias sanitizer
-preserves verbatim, so the alias shown is the alias that resolves. `main`, `master`
-and `production` are refused: a deploy to the production branch is not a preview, it
-is the live site.
-
-### The three surfaces
-
-| Surface | How |
-|---|---|
-| Editor | A site's **Publish** tab → *Individual steps* → **Deploy a preview**: type a branch, press **Deploy preview**. The production button beside it says **Deploy to production**. |
-| Ops API | `POST /api/ops/deploy-site` `{ "siteId": "...", "preview": "<branch>" }` — deploy-only, of the already-built `export/out`. `POST /api/ops/build-deploy` takes `preview` too (build *then* preview-deploy). Both answer with `previewUrl`. |
-| CLI | `pnpm ops deploy-site --json '{"siteId":"anilyzer","preview":"tags-exclude"}' --wait` — the alias is printed on its own line after the log. |
-
-The high-value loop is **build once, preview, then promote**: `build-site` (or the
-Publish tab's *Build static export*), then `deploy-site` with a `preview`, look at
-it, then `deploy-site` again with no `preview` — the same `export/out`, unrebuilt.
-
-Deploy-only ships whatever is in `export/out`, which the basic build composes one
-site at a time into a single shared directory — so it **refuses, before starting a
-job, if `export/out` holds a build of another site** (or no build at all), naming
-the site to build first. `build-deploy` cannot hit this: it builds.
-
-### Two things to know
-
-**A preview shares the production R2 archive bucket.** R2 has no per-branch
-namespace, and the keys are `<siteId>/archives/<file>.zip` either way. In practice
-this is cheap and harmless — the upload skips any object R2 already holds at the same
-size, and an unchanged channel re-zips byte-stable — but a *changed* archive
-replaces the one production's manifest links to. The editor says so once at the top
-of every preview deploy.
-
-**Who can open a preview is a Cloudflare setting, not ours.** Pages projects have a
-*preview deployment access* setting (Settings → General): **public** by default, or
-restricted to Cloudflare Access. A default-configured project's preview URL is
-world-readable by anyone who has the link.
-
-> **A production deploy inherits the checkout's git branch.** `wrangler pages deploy`
-> with no `--branch` infers one from the git repository it runs in, so running a
-> *production* deploy from a feature-branch checkout silently produces a preview
-> instead. This predates the preview feature and is unchanged: only the preview path
-> passes `--branch`. If a "production" deploy did not go live, check what branch the
-> editor's checkout is on.
-
----
-
-## One-time R2 setup
-
-### 1. Create the bucket
-
-One bucket serves **all** your sites — objects are namespaced by `siteId`, so you do
-**not** need one bucket per site. In the dashboard (**R2 → Create bucket**) or via the
-same `wrangler` CLI the deploy already uses:
-
-```sh
-wrangler r2 bucket create my-archives-bucket
-```
-
-### 2. Expose it publicly
-
-R2 buckets are private by default, and the Downloads page links to plain
-`https://…/<file>.zip` URLs, so the bucket needs public read access. There are three
-ways to expose it; the first needs **no domain purchase** and is the recommended one:
-
-- **A Worker on `*.workers.dev` (recommended — no domain, no WHOIS):** deploy the
- bundled `r2-proxy/` Worker to a free `<name>.workers.dev` subdomain (just like Pages'
- `*.pages.dev`). It streams objects from the bucket and gives you edge caching **and**
- tunable rate limiting in code — the same cost defenses a custom domain would, without
- owning a domain. One Worker serves **every** site. See
- [Serving via a Worker on workers.dev](#option-a--serving-via-a-worker-on-workersdev-no-domain)
- below. Set the editor's public URL to `https://<name>.workers.dev`.
-- **Custom domain:** bucket **Settings → Public access → Custom Domains → Connect
- Domain**, e.g. `archives.example.com` (the domain must be on your Cloudflare account).
- Routes downloads through Cloudflare's CDN, WAF, caching, and dashboard rate-limiting
- rules. See [Securing via a custom domain](#option-b--securing-via-a-custom-domain).
-- **`r2.dev` subdomain (quick test only):** bucket **Settings → Public access → Allow
- Access**. You get `https://pub-abc123.r2.dev` — free and domain-free, but Cloudflare
- throttles `r2.dev` and gives you **no** cache / rate-limit control of your own. Fine
- for a smoke test; use the Worker for a real instance.
-
-### 3. Authenticate — two credentials
-
-**a) `wrangler` (for the Pages deploy).** The `wrangler pages deploy` step reuses
-whatever auth you already use. If it works today, nothing to do. Otherwise run
-`wrangler login`, or set `CLOUDFLARE_API_TOKEN`.
-
-**b) R2 S3 API keys (for the archive uploads).** Archive objects are uploaded over
-R2's S3-compatible API, which needs an Access Key ID + Secret. In the dashboard:
-**R2 → Manage R2 API Tokens → Create API token**, permission **Object Read & Write**,
-scoped to your bucket. Then set three environment variables where the editor runs:
-
-```sh
-export R2_ACCESS_KEY_ID=<access key id>
-export R2_SECRET_ACCESS_KEY=<secret access key>
-export CLOUDFLARE_ACCOUNT_ID=<your account id> # used to build the S3 endpoint
-```
-
-The account id is on the R2 overview page; the S3 endpoint is derived as
-`https://<CLOUDFLARE_ACCOUNT_ID>.r2.cloudflarestorage.com`. Keep these in the
-environment (a shell profile, a systemd unit, a `.env` the editor loads) — **not** in
-the settings JSON, which isn't a place for secrets. If a deploy has oversize archives
-to upload but these are unset, it fails *before* the Pages deploy (so the site never
-links to a missing file) with a message pointing back here.
-
-### 4. Point the editor at the bucket
-
-In the editor, open **Settings** and set:
-
-| Field | Value |
-|---|---|
-| **Archive overflow storage (R2 bucket)** | the bucket name, e.g. `my-archives-bucket` |
-| **Archive overflow public URL** | the public base from step 2, e.g. `https://archives.example.com` (no trailing slash needed) |
-
-Leave both blank to keep the old behavior (oversize archives dropped, shown as "Too
-large to host").
-
-On the next **Build & deploy**, oversize archives upload to R2 and the Downloads page
-(and the header link, which reappears once anything is hostable) point at them.
-
----
-
-## Securing downloads against cost-abuse
-
-The threat: someone scripts repeated downloads of large archives to run up your bill.
-
-**The reassuring part — R2 egress is free.** Unlike S3, Cloudflare R2 charges **$0**
-for bandwidth/egress. An attacker looping downloads of a 480 MB zip cannot run up a
-bandwidth bill. The *only* metered cost from reads is **Class B operations** (10M free
-per month, then $0.36/M) — and the defenses below make even that hard to reach.
-
-You get these controls one of two ways — **a Worker on `workers.dev`** (no domain) or
-**a custom domain**. Pick one; both are covered below. The Worker path is recommended
-if you don't want to own a domain.
-
-## Option A — Serving via a Worker on workers.dev (no domain)
-
-The bundled **`r2-proxy/`** Worker is a small, self-contained project that binds the R2
-bucket and serves archive objects on a free `<name>.workers.dev` subdomain — the same
-domain-free model as Pages' `*.pages.dev`. It's a **pure passthrough**: the request path
-`<siteId>/archives/<file>.zip` maps straight to the bucket key, so **one Worker serves
-every site** (deploy it once, not per site — that's the whole point of keying objects by
-site id). What it gives you, all on the free tier:
-
-- **Edge caching** — full downloads are cached with the Cache API, so repeat pulls skip
- R2 (no billable Class B op). It honors the `Cache-Control` we set on each object.
-- **Rate limiting** — Cloudflare's native, free rate-limit binding caps requests per
- client IP + file (dashboard rate-limit rules need a paid zone; this doesn't).
-- **Path allow-listing** — it only serves `*/archives/*.zip`, never arbitrary keys.
-- **Range / resumable downloads** — honors `Range` requests so big zips can resume.
-
-**Deploy it (once for the whole instance):**
-
-1. Edit `r2-proxy/wrangler.toml` and set `bucket_name` to the **same bucket** you use in
- the editor's Settings. Optionally rename the Worker (`name`) and tune the rate limit
- (`limit` / `period`).
-2. From the repo root:
- ```sh
- cd r2-proxy
- pnpm install # first time only
- pnpm run deploy # = wrangler deploy, reusing your host wrangler auth
- ```
- wrangler prints the deployed URL, e.g. `https://ytdlp-archive-proxy.<you>.workers.dev`.
- (An "unsafe fields are experimental" warning for the rate-limit binding is expected.)
-3. In the editor's **Settings**, set **Archive overflow public URL** to that
- `workers.dev` URL. Re-deploy a site and its Downloads links resolve through the Worker.
-
-The Worker code lives in `r2-proxy/src/index.ts` — the caching and rate-limit logic are
-small and commented if you want to adjust them.
-
-> **Free-tier limit:** Workers Free allows **100,000 requests/day** (resets daily). Far
-> more than a downloads endpoint needs; if you ever exceed it, requests get a `429`
-> (fail closed — no surprise bill) until the next day, or upgrade to Workers Paid ($5/mo).
-
-The archive `Cache-Control` is also set at upload time (`Cache-Control: public,
-max-age=3600`, constant `ARCHIVE_CACHE_CONTROL` in
-`common/publish/build.ts`); the Worker reads it back when caching. Archive
-filenames are stable and overwritten in place on re-deploy, so this 1-hour bound is what
-keeps a re-uploaded archive from being served stale for long — raise it if your archives
-rarely change.
-
-## Option B — Securing via a custom domain
-
-If you'd rather use a custom domain (its own upsides: dashboard WAF, managed bot rules,
-and rate-limiting rules without touching code), connect it per
-[step 2 above](#2-expose-it-publicly) and add these, in order of impact:
-
-### 1. Edge caching
-
-Served through a custom domain, Cloudflare's CDN caches each archive at the edge (it
-honors the `Cache-Control: public, max-age=3600` we set at upload), so repeated
-downloads of the same file are served from cache and **never hit R2**. To make caching
-aggressive, add a **Cache Rule** (dashboard: **Caching → Cache Rules → Create**):
-
-- **When:** `URI Path` contains `/archives/`
-- **Then:** *Eligible for cache*, **Edge TTL → Override → 1 day** (or longer).
-
-If you raise the TTL a lot, **purge the cache on deploy** (dashboard **Caching → Purge**,
-or `wrangler`/API) so a re-uploaded archive isn't served stale.
-
-### 2. Rate limiting — the hard backstop
-
-A Rate Limiting rule caps how fast any single client can pull archives, stopping a
-flood that misses cache. Dashboard: **Security → WAF → Rate limiting rules → Create**
-(the free plan includes one rule):
-
-- **When incoming requests match:** `URI Path` contains `/archives/`
-- **Rate:** e.g. **20 requests per 1 minute** per client IP
-- **Then:** *Block* for 10 minutes (or *Managed Challenge*).
-
-Tune the threshold to real usage — legitimate users download a handful of files, not
-dozens per minute.
-
-### 3. Bot Fight Mode + WAF managed rules
-
-Dashboard: **Security → Bots → Bot Fight Mode** (free). Blocks the low-effort scripted
-abuse that makes up most of this traffic. The free **WAF managed ruleset** adds a
-baseline of protection at no cost.
-
-### 4. Hotlink protection (optional)
-
-Stops other sites embedding your archives and spending your ops budget serving their
-audience. A WAF custom rule (**Security → WAF → Custom rules**):
-
-- **When:** `URI Path` contains `/archives/` **and** `Referer` does not contain your
- domain **and** `Referer` is not empty
-- **Then:** *Block*.
-
-(Allow an empty `Referer` so direct clicks and privacy-conscious browsers still work.)
-
-## Billing / usage alerts (either option)
-
-R2 has no hard spend cap, but Cloudflare **Notifications** (dashboard: **Notifications
-→ Add**) can email you when R2 storage or Class A/B operations cross a threshold —
-cheap insurance so nothing surprises you.
-
-## What to skip
-
-**Signed URLs / token-gated downloads** are the heavyweight option — they add key
-management and friction for legitimate users. Given egress is free and caching
-neutralizes the ops cost, they're overkill for *cost* defense (the `r2-proxy` Worker is
-a plain passthrough, not an access gate). Only reach for signed URLs if you want
-*access control* (private archives), not cost control.
-
----
-
-## Cost expectations
-
-Measured across all sites in this project, total compressed archives are **≈ 2–4.5 GB**.
-Against R2's free tier:
-
-| Resource | Free tier / month | This project's usage |
-|---|---|---|
-| **Egress / bandwidth** | unlimited, **$0** | irrelevant — no egress charge exists |
-| **Storage** | 10 GB-month | ~2–4.5 GB — comfortable |
-| **Class A ops** (writes) | 1,000,000 | ~one PUT per archive per deploy — negligible |
-| **Class B ops** (reads) | 10,000,000 | mostly absorbed by CDN cache |
-
-The realistic bill for hosting these archives is **$0**. The configuration above exists
-to keep it that way under adversarial traffic, not because normal usage is close to any
-limit.
diff --git a/DEPLOY_DOCKER.md b/DEPLOY_DOCKER.md
@@ -1,82 +0,0 @@
-# Docker export build pipeline
-
-> **Not the document you want if you are trying to *run* the apps in containers.**
-> That is [RUNNING_IN_DOCKER.md](./RUNNING_IN_DOCKER.md) — `docker compose up` and a
-> working archive server, built from the root `Dockerfile`. This page is about
-> *building sites*: fanning per-site export builds out across containers, using
-> `Dockerfile.build`. The two share nothing but the word "docker".
-
-Archilyzer's Docker build mode builds **every site in parallel** in isolated containers, then
-deploys them serially — a large speedup when you host several sites, and stronger
-isolation than the basic single-process build. This is opt-in: set **Build
-pipeline → Docker** in Settings (or the toggle on the Deploy page). Basic mode is
-unchanged and remains the default.
-
-## Prerequisites
-
-- A container engine: **Docker**, or **podman** (set `DOCKER_BIN=podman`). Rootless
- podman is a good fit — it maps container files to your host user automatically.
-- The editor host still needs Node + pnpm (Phase A and deploy run on the host) and
- your Cloudflare/R2 credentials in the environment (see
- [DEPLOY_CLOUDFLARE.md](./DEPLOY_CLOUDFLARE.md)). Credentials are **never** passed
- into a container — deploy runs on the host.
-
-The build image is built (and cached) automatically from `Dockerfile.build` the
-first time you run; edit **Build image** / **Dockerfile** in Settings to override
-the tag/path.
-
-## How it works
-
-Trigger it with **Build all sites** on the Deploy page. One managed job runs three
-ordered phases:
-
-1. **Phase A — shared, on the host, once.** `build:data` (search index +
- `.export-index` staging) then `build:archives` (warm the shared archive-zip
- cache for the union of all sites' channels). Only the host writes this shared
- state, so containers never race it. This phase is serial and is the long pole on
- a cold build; on a warm rebuild it's near-instant (unchanged channels are
- skipped).
-2. **Phase B — per-site, in parallel containers.** Each site's `compose:site +
- next build` runs in its own container, capped by **Max parallel builds**. Each
- writes an isolated `out/` under `export/.export-builds/<siteId>/`. Containers
- mount the corpus/index/staging/archive cache **read-only**. (Network is left on:
- `next build` fetches the site's fonts from Google via `next/font/google`;
- isolation comes from the read-only mounts, per-site output dir, and non-root
- user.)
-3. **Phase C — deploy, on the host, serially.** After every build finishes, each
- built site is deployed in turn (oversize-archive R2 upload, then `wrangler pages
- deploy`). A single site failing to build or deploy is reported and skipped; the
- rest still ship.
-
-If no container engine is available, the action logs a notice and falls back to a
-serial host build+deploy (one site at a time).
-
-## Mounts (per Phase-B container)
-
-| Host | Container | Mode |
-|---|---|---|
-| `transcripts/` (corpus + `index.mdb` + archive cache) | `/data/transcripts` | ro |
-| `export/.export-index` (shared + per-site staging) | `/data/export/.export-index` | ro |
-| `export/.export-builds/<siteId>` (public/out/.next/caches) | `/site` | rw |
-| `settings.json` (build config, mounted fresh — not baked) | `/data/settings.json` | ro |
-
-The per-site `/site` mount is persistent, so incremental `next build` (`.next`) and
-incremental compose (`.compose-cache`) stay warm across builds.
-
-## Tuning & environment
-
-- **Max parallel builds** (setting) — how many site containers run at once. Each
- `next build` can use up to ~8 GB; a safe starting point is `floor(RAM_GB / 9)`.
-- `DOCKER_BIN` — container binary (default `docker`; e.g. `podman`).
-- `DOCKER_BUILD_MEMORY`, `DOCKER_BUILD_CPUS` — optional per-container `--memory` /
- `--cpus` caps so a fan-out can't OOM/peg the host.
-- `BUILD_ARCHIVES=0` (or the **Skip archive zips** checkbox) — skip the archive
- warm + per-site archive materialize for a faster build with no download bundles.
-
-## Notes
-
-- Containers run as your host uid/gid (`-u`), so files under `.export-builds/` are
- host-owned, not root-owned.
-- The image bakes the repo source + deps; a code change rebuilds it, but Docker
- layer caching keeps that cheap (deps re-install only when the lockfile moves).
-- `.export-builds/` is gitignored and excluded from the image build context.
diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md
@@ -0,0 +1,181 @@
+# Environment variables
+
+<!-- GENERATED by common/bin/env-docs.ts from common/lib/envVars.ts — do not edit by hand. -->
+
+Every environment variable the repo's code reads, by who it is for. The list is code (`common/lib/envVars.ts`), and a test fails when the code reads a variable the list does not declare, or the list declares one that nothing outside the list names any more. umtool's own knobs are documented in [umtool/docs](umtool/docs/README.md).
+
+Regenerate this file with `pnpm archilyzer docs env`. `pnpm archilyzer doctor` prints which of the paths overrides are set on this machine.
+
+## Paths and binaries
+
+The one override surface for where things live and which binary runs. Every one is read by `getPaths()` (`common/lib/paths.ts`) and nowhere else; nothing hardcodes a location.
+
+| Variable | Default | What it does | Read by |
+|---|---|---|---|
+| `TRANSCRIPTS_DIR` | `<repo>/transcripts` | The corpus: channels, sites, the LMDB index, job logs, the saved-video store. | common/lib/paths.ts (getPaths) |
+| `SAVED_VIDEOS_DIR` | `<TRANSCRIPTS_DIR>/saved-videos` | The persisted source-video store, when it should live on another disk. | common/lib/paths.ts (getPaths) |
+| `SITES_DIR` | `<TRANSCRIPTS_DIR>/sites` | Per-site config (`<id>/site.json`, every key in [SITE.md](SITE.md)) and the homepage's `_homepage/`. | common/lib/paths.ts (getPaths) |
+| `SETTINGS_FILE` | `<repo>/settings.json` | The settings file (every key in [SETTINGS.md](SETTINGS.md)). | common/lib/paths.ts (getPaths) |
+| `EXPORT_PUBLIC_DIR` | `<repo>/export/public` | The dir the export site serves at `/`, composed one site at a time. | common/lib/paths.ts (getPaths) |
+| `EXPORT_INDEX_DIR` | `.export-index` beside `EXPORT_PUBLIC_DIR` | The build's staging area (not served): the shared index and per-site aggregates. | common/lib/paths.ts (getPaths) |
+| `EXPORT_BUILDS_DIR` | `.export-builds` beside `EXPORT_PUBLIC_DIR` | Per-site `out/` bundles from a docker-mode build. | common/lib/paths.ts (getPaths) |
+| `EDITOR_CHANGELOG_FILE` | `<repo>/editor/CHANGELOG.md` | The editor changelog the release cutter reads and rewrites. The e2e server points it at a gitignored copy. | common/lib/paths.ts (getPaths) |
+| `EXPORT_CHANGELOG_FILE` | `<repo>/export/CHANGELOG.md` | The export changelog, likewise. | common/lib/paths.ts (getPaths) |
+| `CHARTS_CONFIG_FILE` | `<repo>/chart-templates.json` | The legacy chart-templates file, read only by a migration. | common/lib/paths.ts (getPaths) |
+| `SEARCH_ALIASES_FILE` | `<TRANSCRIPTS_DIR>/search-aliases.json` | The corpus-wide search-alias dictionary. | common/lib/paths.ts (getPaths) |
+| `CURATED_TAGS_FILE` | `<TRANSCRIPTS_DIR>/tags.json` | Curated per-video tags. Written only through `applyTagAssignments`. | common/lib/paths.ts (getPaths) |
+| `YTDLP_BIN` | `yt-dlp` on PATH | The downloader. Every fetch goes through it. | common/lib/paths.ts (getPaths) |
+| `FFMPEG_BIN` | `ffmpeg` on PATH | Audio extraction for transcription and diarization. | common/lib/paths.ts (getPaths) |
+| `FFPROBE_BIN` | `ffprobe` on PATH | Duration checks (the short-audio guard, windowing). | common/lib/paths.ts (getPaths) |
+| `WHISPER_BIN` | `whisper-cli` on PATH | whisper.cpp, one of the three transcription engines (with chough and parakeet.cpp). | common/lib/paths.ts (getPaths) |
+| `WHISPER_MODEL` | `~/whispercpp/whisper.cpp/models/ggml-base.en.bin` | whisper.cpp's model, when a worker names none. | common/lib/paths.ts (getPaths) |
+| `PARAKEET_STITCH_BIN` | `<repo>/scripts/parakeet-stitch.mjs` | The parakeet.cpp engine's wrapper (overlapping windows, stitched). | common/lib/paths.ts (getPaths) |
+| `PARAKEET_CLI` | `parakeet-cli` on PATH | The parakeet.cpp binary the wrapper drives (the wrapper reads it too). | common/lib/paths.ts (getPaths) |
+| `PARAKEET_MODEL` | none | parakeet.cpp's `.gguf`, when a worker names none (the wrapper reads it too). | common/lib/paths.ts (getPaths) |
+| `DIARIZE_BIN` | `<repo>/scripts/diarize.mjs` | The speaker-diarization wrapper. The e2e suite swaps in a fake here. | common/lib/paths.ts (getPaths) |
+| `RSYNC_BIN` | `rsync` on PATH | Mirrors the saved-video store to a backup destination. | common/lib/paths.ts (getPaths) |
+| `FINDMNT_BIN` | `findmnt` on PATH | The read-only volume-identity probe behind storage locations. Optional. | common/lib/paths.ts (getPaths) |
+| `UDISKSCTL_BIN` | `udisksctl` on PATH | Mounts an attached volume from `/storage`. Optional. | common/lib/paths.ts (getPaths) |
+| `GALLERY_DL_BIN` | `gallery-dl` on PATH | The X/Twitter post fetcher, for social channels. | common/lib/paths.ts (getPaths) |
+| `OLLAMA_URL` | `http://127.0.0.1:11434` | The local ollama server, the local digest and attribution engine. | common/lib/paths.ts (getPaths) |
+| `CLAUDE_BIN` | `claude` on PATH | The `claude` CLI, driving the opt-in metered digest lane. | common/lib/paths.ts (getPaths) |
+
+## Runtime
+
+Tokens, credentials and knobs a running process reads. Most configuration is not here but in `settings.json` ([SETTINGS.md](SETTINGS.md)).
+
+| Variable | Default | What it does | Read by |
+|---|---|---|---|
+| `WORKER_TOKEN` | unset (both surfaces off) | Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`, the MCP's `fetch_clip`). Set the same value on both ends. | common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts |
+| `SYNC_HEARTBEAT_SECONDS` | `settings.syncScheduler.heartbeatSeconds` | Overrides the editor's in-process sync heartbeat. `0` = no internal timer (tick from cron instead). | editor/app/scheduler/heartbeat.ts |
+| `SYNC_TICK_URL` | `http://127.0.0.1:3001/api/scheduler/tick` | Where `archilyzer sync tick` (cron's heartbeat) posts. | common/bin/sync-tick.ts |
+| `SYNC_TICK_TOKEN` | unset (no auth) | Bearer token for the tick endpoint; set on both the editor and the cron job. | common/bin/sync-tick.ts, editor/app/scheduler/auth.ts |
+| `R2_ACCESS_KEY_ID` | — | R2 S3 credentials for uploading oversize archives at deploy time (with `R2_SECRET_ACCESS_KEY` and `CLOUDFLARE_ACCOUNT_ID`). See [PUBLISH.md](PUBLISH.md). | common/publish/build.ts |
+| `R2_SECRET_ACCESS_KEY` | — | See `R2_ACCESS_KEY_ID`. | common/publish/build.ts |
+| `CLOUDFLARE_ACCOUNT_ID` | — | The account the R2 endpoint belongs to. wrangler reads its own credentials. | common/publish/build.ts |
+| `DOCKER_BIN` | `docker` | The container engine for docker-mode builds (e.g. `podman`). | common/publish/build.ts |
+| `DOCKER_BUILD_MEMORY` | no cap | Per-container memory cap for a docker-mode build (`--memory`). | common/publish/build.ts |
+| `DOCKER_BUILD_CPUS` | no cap | Per-container CPU cap for a docker-mode build (`--cpus`). | common/publish/build.ts |
+| `ARCHIVE_CHANNEL_CONCURRENCY` | `4` | How many channels' archive zips `build archives` builds at once. | common/bin/build-archives.ts |
+| `MAX_ARCHIVE_BYTES` | the Cloudflare-safe cap | The served-file size cap for archives, in bytes; `0` = no cap. A site's own `archiveMaxBytes` wins. | common/bin/compose-site.ts |
+| `CHOUGH_BIN` | `chough` on PATH | The chough transcription engine, when a worker names no binary. | common/lib/transcriptionApps.ts |
+| `CHOUGH_MODEL` | chough's own | Passed to chough from a worker's model field; chough auto-downloads one when unset. | chough (set by common/lib/transcriptionApps.ts) |
+| `CHOUGH_URL` | local | Passed to chough from a worker's remote-server field. | chough (set by common/lib/transcriptionApps.ts) |
+| `OLLAMA_DIGEST_MODEL` | `qwen2.5:7b` | The ollama model the local digest lane asks for when settings name none. | common/lib/digestApps.ts |
+| `CLAUDE_DIGEST_MODEL` | the CLI's default | The model the metered digest lane asks `claude` for when settings name none. | common/lib/digestApps.ts |
+| `NITTER_INSTANCES` | a built-in list | Comma-separated Nitter instances for the X fallback fetcher, in order of preference. | common/social/xNitterFetcher.ts |
+| `UMTOOL_URL` | unset (no link) | umtool's front door; when set, the video page links to it. | editor/app/channels/[slug]/videos/[id]/page.tsx |
+| `TRANSCRIPT_SITE_URL` | — | MCP server: one published archive to read over HTTP. | mcp/src/sources.ts |
+| `TRANSCRIPT_HUB_URL` | — | MCP server: a hub, federating every archive it lists. | mcp/src/sources.ts |
+| `TRANSCRIPT_LOCAL_DIR` | — | MCP server: a composed public dir on disk. | mcp/src/sources.ts |
+| `TRANSCRIPT_PLATFORM_LINKS` | off | `1` cites platform watch pages instead of the archive's own pages. | common/lib/archive/reader-fs.ts |
+| `MCP_IO_STATS` | off | `1` turns on per-call I/O accounting, for `mcp/bench`. | common/lib/archive/io-stats.ts |
+| `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's `fetch_clip` talk to. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool |
+| `ARCHILYZER_AGENT` | `cli` | Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`. | scripts/archilyzer-ops.mjs |
+| `DIARIZE_ENGINE_KIND` | `sherpa-onnx` | The diarization engine: `sherpa-onnx` or `sortformer`. | scripts/diarize.mjs |
+| `DIARIZE_ENGINE_CMD` | the bundled sherpa script | The engine command the wrapper runs. | scripts/diarize.mjs |
+| `DIARIZE_PYTHON` | `python3` | The python for the default engine. | scripts/diarize.mjs |
+| `DIARIZE_SEG_MODEL` | — (required) | Segmentation model. The editor passes the settings' value as a flag. | scripts/diarize.mjs |
+| `DIARIZE_EMB_MODEL` | — (required) | Speaker-embedding model. The editor passes the settings' value as a flag. | scripts/diarize.mjs |
+| `DIARIZE_THRESHOLD` | `0.5` | Clustering threshold. | scripts/diarize.mjs |
+| `DIARIZE_THREADS` | `4` | Engine threads. | scripts/diarize.mjs |
+| `DIARIZE_WINDOW_MINUTES` | `45` | Window length for long files; `0` never windows. | scripts/diarize.mjs |
+| `DIARIZE_WINDOW_AFTER_MINUTES` | `90` | Only files longer than this are windowed. | scripts/diarize.mjs |
+| `SORTFORMER_BIN` | — (required for sortformer) | The sortformer engine binary. | scripts/diarize.mjs, scripts/diarize-sortformer.mjs |
+| `SORTFORMER_MODEL` | — (required for sortformer) | The sortformer `.gguf`. | scripts/diarize.mjs, scripts/diarize-sortformer.mjs |
+| `PARAKEET_SEGMENT_SEC` | `480` | parakeet window length, seconds (a worker's chunk size wins). | scripts/parakeet-stitch.mjs |
+| `PARAKEET_OVERLAP_SEC` | `6` | parakeet window overlap, seconds. | scripts/parakeet-stitch.mjs |
+| `PARAKEET_DECODER` | parakeet-cli's | `ctc` or `tdt`, passed through to parakeet-cli. | scripts/parakeet-stitch.mjs |
+| `PARAKEET_LANG` | parakeet-cli's | A locale, passed through to parakeet-cli. | scripts/parakeet-stitch.mjs |
+| `PARAKEET_DEVICE` | parakeet-cli's | Compute device (`cpu`, `CUDA0`, `Vulkan1`, …), exported to parakeet-cli. | scripts/parakeet-stitch.mjs |
+
+## Ports
+
+Every local server's default port, from `common/lib/ports.mjs`. The primary checkout uses these; worktree N adds N × 100 (`pnpm wt list`).
+
+| Variable | Default | What it does | Read by |
+|---|---|---|---|
+| `EDITOR_PORT` | `3001` | Editor real dev/start (`pnpm dev:editor`). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `PORT` | `3011` | Editor test server + Playwright editor baseURL. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `EXPORT_PORT` | `3010` | Export server launched by the editor e2e. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `EXPORT_DEV_PORT` | `3000` | Export real dev (`pnpm dev:export`). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `EXPORT_E2E_PORT` | `3020` | Export's own Playwright suite. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `OLLAMA_STUB_PORT` | `11435` | Digest-lane stub server in the editor e2e suite. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `HOMEPAGE_DEV_PORT` | `3030` | Homepage real dev (`pnpm dev:homepage`). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `HOMEPAGE_PORT` | `3031` | Homepage static `serve out` (start:homepage). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `HOMEPAGE_E2E_PORT` | `3040` | Homepage's own Playwright suite. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `HUB_PORT` | `3041` | Export's hub Playwright suite (e2e:hub). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `UMTOOL_PORT` | `3050` | Umtool real dev/start (`pnpm dev:umtool`). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `UMTOOL_E2E_PORT` | `3051` | Umtool's own Playwright suite. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `EDITOR_STUB_PORT` | `3052` | Stub editor the umtool e2e suite fetches clips from. A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `ORIGIN_B_PORT` | `4610` | Export's two-origin suite: the member site (e2e:2origin). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+| `HUB_A_PORT` | `4611` | Export's two-origin suite: the hub (e2e:2origin). A worktree adds its offset (`pnpm wt list`). | common/lib/ports.mjs |
+
+## Set by the pipeline
+
+The publish pipeline sets these for a process it spawns. Listed so a reader knows what they are; nobody sets them by hand.
+
+| Variable | Default | What it does | Read by |
+|---|---|---|---|
+| `SITE_ID` | — | Which site a compose or an export build is for. `archilyzer build site <id>` sets it; `compose site` and `build site` fall back to it when no id is given. | common/bin/compose-site.ts, export/app/lib/site.ts |
+| `INSTANCE_MODE` | a site | `hub` makes the export build the hub. Set by `archilyzer build hub`. | export/app/lib/mode.ts, common/lib/archive/contract.ts |
+| `BUILD_ARCHIVES` | on | `0` skips archive-zip generation for one build (`--skip-archives`). | common/bin/compose-site.ts, common/bin/build-archives.ts |
+| `ARCHIVES_READONLY` | off | `1` inside a docker-mode build container: materialize archives, never write the shared cache. | common/bin/compose-site.ts |
+| `HOMEPAGE_PUBLIC_DIR` | `<repo>/homepage/public` | Where `compose homepage` writes. | common/bin/compose-homepage.ts |
+| `HOMEPAGE_SUMMARY_FILE` | `homepage/public/homepage-summary.json` | A dev-only summary file for the homepage; ignored by a production build. The homepage e2e sets it. | homepage/app/lib/summary.ts |
+
+## Docker
+
+The container's own set, read by `docker/*.sh`, the compose files and Caddy — not by the apps' code (except `ARCHILYZER_IDLE_BOOT`). See [RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md).
+
+| Variable | Default | What it does | Read by |
+|---|---|---|---|
+| `ARCHILYZER_TRANSCRIBER` | baked per image target (`whisper-cpp` in `runtime`) | `whisper-cpp` or `parakeet`: which worker the first boot seeds and which model it fetches. | docker/entrypoint.sh |
+| `ARCHILYZER_FETCH_MODEL` | per transcriber | Which model the first boot downloads; `none` skips it. | docker/entrypoint.sh |
+| `ARCHILYZER_MODELS_DIR` | `/data/models` | Where models live in the container. | docker/entrypoint.sh |
+| `ARCHILYZER_BUILDS_DIR` | `/data/builds` | Where the container keeps built sites. | docker/entrypoint.sh |
+| `ARCHILYZER_SITE_OUT` | `/data/builds/site` | The built export site the `site` service serves. | docker/entrypoint.sh, docker/publish-site.sh |
+| `ARCHILYZER_IDLE_BOOT` | off | `1` boots the editor without arming the heartbeat or any auto-queue runner. | common/lib/idleBoot.ts (the editor) |
+| `ARCHILYZER_AUTH_MODE` | `basic` | `basic`, `forward` or `none` — the only escape hatch from the exposure guard. | docker/guard-exposure.sh, docker/caddy-start.sh |
+| `ARCHILYZER_AUTH_USER` | `archilyzer` | Basic-auth user. | docker/Caddyfile |
+| `ARCHILYZER_AUTH_HASH` | — | Basic-auth bcrypt hash (`caddy hash-password`). | docker/Caddyfile, docker/guard-exposure.sh |
+| `ARCHILYZER_AUTH_IMPORT` | derived from the mode | Set by docker/caddy-start.sh from the mode: which auth snippet the private sites import. | docker/Caddyfile |
+| `ARCHILYZER_FORWARD_AUTH_UPSTREAM` | — | Forward-auth server (Authelia, tinyauth, …), `host:port`. | docker/Caddyfile |
+| `ARCHILYZER_FORWARD_AUTH_URI` | `/api/auth/caddy` | The forward-auth server's verify path. | docker/Caddyfile |
+| `ARCHILYZER_TAG` | `local` | The image tag the compose files build and run. | docker-compose*.yml |
+
+## Tests only
+
+Read only by a test harness, a fake binary or a test-mode branch. Never set one on a real instance.
+
+| Variable | Default | What it does | Read by |
+|---|---|---|---|
+| `EDITOR_TEST_ROUTES` | off | `1` opens the editor's `/api/test/*` routes. The e2e server sets it. | editor/app/api/test/_guard.ts, editor/instrumentation.ts |
+| `E2E_MODE` | dev | `start` runs the editor suite against `next start` instead of `next dev`. | editor/playwright.config.ts |
+| `E2E_QUEUE` | on | `0` skips the machine-global e2e queue (the port check still runs). | scripts/queue-lock.mjs |
+| `E2E_PORT_CHECK` | on | `0` skips the pre-run check that the suite's ports are free. | scripts/queue-lock.mjs |
+| `E2E_QUEUE_TIMEOUT` | wait forever | Seconds to wait for the queue before giving up. | scripts/queue-lock.mjs |
+| `E2E_PORT_GRACE_MS` | `3000` | How long the port check waits for a just-freed port. | scripts/queue-lock.mjs |
+| `E2E_QUEUE_LOCK_FILE` | one per machine | The queue's lock file; the queue's own tests point it elsewhere. | scripts/queue-lock.mjs |
+| `QUEUE_LOCK_HELD` | — | Set by the queue for the command it runs, so a nested wrapper passes through. | scripts/queue-lock.mjs |
+| `PLAYWRIGHT_BASE_URL` | `http://localhost:<PORT>` | The editor test server's URL; the worktree injector sets it. | editor/playwright.config.ts, editor/e2e/baseUrl.ts |
+| `AUDIO_CHECK_INTERVAL_MS_OVERRIDE` | the real cadence | Shrinks the mid-download audio check so the e2e suite sees it fire. | common/ytdlp/audioCheckedDownload.ts |
+| `AUDIO_CHECK_SIZE_GATE_OVERRIDE` | the real gate | Likewise, the size gate. | common/ytdlp/audioCheckedDownload.ts |
+| `AUDIO_CHECK_INTERVAL_FLOOR_MS_OVERRIDE` | the real floor | Likewise, the interval floor. | common/ytdlp/audioCheckedDownload.ts |
+| `AUDIO_CHECK_RECOVER_STEP_MS_OVERRIDE` | the real step | Likewise, the recovery step. | common/ytdlp/audioCheckedDownload.ts |
+| `AUDIO_CHECK_RECOVER_AFTER_OVERRIDE` | the real count | Likewise, the recovery count. | common/ytdlp/audioCheckedDownload.ts |
+| `AUDIO_CHECK_DEBUG_PAUSE_MS` | off | A debugging pause inside the audio check. | common/ytdlp/audioCheckedDownload.ts |
+| `FAKE_YTDLP_AUDIO_CHECK_MODE` | — | Fake yt-dlp: which audio-check scenario to act out. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_YTDLP_CHUNK_DELAY_MS` | — | Fake yt-dlp: delay between written chunks. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_YTDLP_CORRUPT_AFTER_CHUNK` | — | Fake yt-dlp: start corrupting after this chunk. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_YTDLP_CORRUPT_RUNS` | — | Fake yt-dlp: how many runs corrupt. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_YTDLP_DETERMINISTIC_CORRUPT` | — | Fake yt-dlp: corrupt deterministically. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_YTDLP_RECOVER_ON_RESUME` | — | Fake yt-dlp: a resumed run recovers. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_YTDLP_TOTAL_CHUNKS` | — | Fake yt-dlp: how many chunks a download has. | editor/e2e/fixtures/bin/fake-ytdlp.mjs |
+| `FAKE_GALLERY_DL_AUTH_FAIL` | — | Fake gallery-dl: fail as an auth error. | editor/e2e/fixtures/bin/fake-gallery-dl.mjs |
+| `FIXTURE_MAX_LIFETIME_MS` | the watchdog's | How long a fake binary may live before its watchdog kills it. | editor/e2e/fixtures/bin/_watchdog.mjs |
+| `OLLAMA_STUB_MODEL` | `qwen2.5:7b` | The model the ollama stub claims to serve. | editor/e2e/fixtures/ollama-stub.mjs |
+| `RACK_SHOTS` | off (spec skipped) | Runs the `/channels` rack screenshot audit. | editor/e2e/channels-rack-audit.spec.ts |
+| `TWO_ORIGIN_REBUILD` | off | `1` rebuilds the two-origin suite's cached hub bundle. | export/e2e-2origin/globalSetup.ts |
+| `IMAGE` | `yt-dlp-transcript-browser-e2e` | The sharded e2e run's image tag. | scripts/run-sharded-e2e.mjs |
+| `SKIP_BUILD` | off | `1` reuses the sharded e2e image instead of rebuilding it (`--no-build`). | scripts/run-sharded-e2e.mjs |
diff --git a/PUBLISH.md b/PUBLISH.md
@@ -0,0 +1,448 @@
+# Publishing
+
+What gets published, how to build and deploy it, and how to keep a public archive
+cheap and safe. For *running* the apps in containers see
+[RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md); every environment variable named here
+is in [ENVIRONMENT.md](ENVIRONMENT.md).
+
+- [What gets published](#what-gets-published)
+- [The three ways to drive it](#the-three-ways-to-drive-it)
+- [Cloudflare Pages](#cloudflare-pages)
+- [Preview deployments](#preview-deployments)
+- [Download archives and R2](#download-archives-and-r2)
+- [Securing downloads against cost-abuse](#securing-downloads-against-cost-abuse)
+- [Building every site in containers](#building-every-site-in-containers)
+- [Reading a published archive from Claude Code](#reading-a-published-archive-from-claude-code)
+
+---
+
+## What gets published
+
+Three static artefacts, each pre-rendered to plain HTML and JSON — no database, no
+server-side code, servable by anything:
+
+| Artefact | What it is | Built into | Config |
+|---|---|---|---|
+| A **site** | The export app over the channels one site selects: search, transcripts, charts, downloads. One corpus can publish several. | `export/out` (docker mode: `export/.export-builds/<siteId>/out`) | `transcripts/sites/<id>/site.json` ([SITE.md](SITE.md)) |
+| The **hub** | The export app in hub mode: federated search across the family of sites. | `export/out` | `transcripts/sites/_homepage/homepage.json` |
+| The **homepage** | The project's own site (`homepage/`). | `homepage/out` | — |
+
+The hub and the homepage are two different Pages projects: the hub deploys to the
+project `homepage.json` names (e.g. `archilyzer-hub`) and is refused `archilyzer`,
+which is the homepage's.
+
+A site build has three steps, run in `export/`: the **data phase** (the LMDB index,
+the stats datasets and the chart templates — `archilyzer index`, `build stats`,
+`build templates`), **compose** (the site's slice of the shared index into
+`export/public`, plus its download archives — `archilyzer compose site <id>`) and
+`next build`. `--nodata` skips the data phase and reuses the last one's staging.
+
+## The three ways to drive it
+
+The editor's **/sites** page, `pnpm ops` (HTTP to a running editor, with its
+`WORKER_TOKEN`) and the `archilyzer` CLI (local, no editor needed) call the same
+entry points in `common/publish/build.ts`.
+
+| To | Editor | `pnpm ops` | `pnpm archilyzer …` |
+|---|---|---|---|
+| Build one site | a site's **Publish** tab → *Build static export* | `build-site` | `build site <id> [--nodata] [--skip-archives]` |
+| Deploy the built site | *Deploy to production* / *Deploy preview* | `deploy-site` | `deploy site <id> [--preview <branch>]` |
+| Build, then deploy | *Build & deploy* | `build-deploy` | `build site <id>` then `deploy site <id>` |
+| Build every site | /sites → **Build all sites** | — | `build all [--skip-archives]` |
+| The hub | /sites → Hub → **Build hub** / **Deploy hub** | `build-hub`, `deploy-hub` | `build hub`, `deploy hub [--preview <branch>]` |
+| The homepage | /sites → Homepage → **Build homepage** (tick *Deploy after build*) / **Deploy homepage**, with an optional preview branch | `build-homepage` (`{"deploy":true}` to deploy after), `deploy-homepage` (`{"preview":"<branch>"}`) | `build homepage`, `deploy homepage [--preview <branch>]` |
+
+`pnpm archilyzer <command>` is the short form of
+`pnpm --filter yt-dlp-transcript-common exec tsx bin/archilyzer.ts <command>`;
+`pnpm archilyzer --help` lists every command, and `pnpm archilyzer doctor` checks that
+this machine has what a build needs. From `export/`, `pnpm run build` is
+`archilyzer build site` and `pnpm run deploy` is `archilyzer deploy site`; both take
+the site from `SITE_ID` when no id is given.
+
+Deploy-only ships whatever is in `export/out`, which the basic build composes one
+site at a time into a single shared directory — so it **refuses, before starting a
+job, if `export/out` holds a build of another site** (or no build at all), naming the
+site to build first. A deploy queued behind another site's build re-checks when it
+starts. `build-deploy` cannot hit this: it builds.
+
+---
+
+## Cloudflare Pages
+
+Sites deploy to **Cloudflare Pages** with `wrangler pages deploy`; download archives
+too large for Pages' per-file limit overflow to **Cloudflare R2**. Everything here
+fits inside Cloudflare's **free tier**.
+
+- **A Pages project must exist before its first deploy.** wrangler offers to create a
+ missing project only on an interactive terminal, and the deploy's stdin is a pipe,
+ so a missing project fails at once with wrangler's own "does not exist" sentence.
+ Create it first: `pnpm dlx wrangler pages project create <name> --production-branch
+ main`. A site's project is `site.json`'s `cloudflareProject`.
+- **wrangler's own auth.** The deploy reuses whatever auth wrangler already has —
+ `wrangler login`, or `CLOUDFLARE_API_TOKEN` in the environment.
+- **A production deploy inherits the checkout's git branch.** `wrangler pages deploy`
+ with no `--branch` infers one from the repository it runs in, so a *production*
+ deploy from a feature-branch checkout silently produces a preview instead. Only the
+ preview path passes `--branch`; `deploy homepage` passes `--branch main`. If a
+ "production" deploy did not go live, check what branch the checkout is on.
+
+## Preview deployments
+
+A **preview** is the same built bundle deployed to a branch that is not the Pages
+project's production branch. Cloudflare publishes it at a **branch alias** —
+
+```
+https://<branch>.<project>.pages.dev
+```
+
+— and leaves the live site alone. Each deploy also gets an immutable
+per-deployment URL (`https://<hash>.<project>.pages.dev`), which wrangler prints as
+"Deployment complete! Take a peek over at …"; the editor repeats both on a
+`[preview]` line at the end of the job log, because the streamed log scrolls.
+
+The alias is a function of the project and the branch and nothing else, so it is
+known *before* the deploy runs — which is why the editor can link it while you are
+still typing the name.
+
+**Branch names** must be 1–28 lowercase letters, digits and dashes, starting and
+ending with a letter or digit. That is exactly what Cloudflare's alias sanitizer
+preserves verbatim, so the alias shown is the alias that resolves. `main`, `master`
+and `production` are refused: a deploy to the production branch is not a preview, it
+is the live site.
+
+| Surface | How |
+|---|---|
+| Editor | A site's **Publish** tab → *Individual steps* → **Deploy a preview**: type a branch, press **Deploy preview**. The production button beside it says **Deploy to production**. |
+| Ops API | `POST /api/ops/deploy-site` `{ "siteId": "...", "preview": "<branch>" }` — deploy-only, of the already-built `export/out`. `POST /api/ops/build-deploy` takes `preview` too (build *then* preview-deploy). Both answer with `previewUrl`. `pnpm ops deploy-site --json '{"siteId":"anilyzer","preview":"tags-exclude"}' --wait` prints the alias on its own line after the log. |
+| CLI | `pnpm archilyzer deploy site anilyzer --preview tags-exclude` |
+
+The high-value loop is **build once, preview, then promote**: build the site, deploy
+it with a preview branch, look at it, then deploy again with no preview — the same
+`export/out`, unrebuilt.
+
+**A preview shares the production R2 archive bucket.** R2 has no per-branch
+namespace, and the keys are `<siteId>/archives/<file>.zip` either way. In practice
+this is cheap and harmless — the upload skips any object R2 already holds at the same
+size, and an unchanged channel re-zips byte-stable — but a *changed* archive replaces
+the one production's manifest links to. The editor says so once at the top of every
+preview deploy.
+
+**Who can open a preview is a Cloudflare setting, not ours.** Pages projects have a
+*preview deployment access* setting (Settings → General): **public** by default, or
+restricted to Cloudflare Access. A default-configured project's preview URL is
+world-readable by anyone who has the link.
+
+---
+
+## Download archives and R2
+
+At compose time, `archilyzer compose site` generates one transcript zip and one
+live-chat zip **per channel** into `export/public/archives/`, and records them in
+`public/archives/manifest.json`. The `/downloads` page and the header **Downloads**
+link read that manifest. A site can turn its archives off (`site.json` `archives: false`), and a build
+can skip them (`--skip-archives`, `BUILD_ARCHIVES=0`, or the **Skip archive zips**
+checkbox).
+
+Cloudflare Pages rejects any single asset larger than **25 MB**, and real channels
+blow past that easily (a channel's live-chat zip can be hundreds of MB). So the
+pipeline splits archives by size:
+
+| Archive size | Where it's served from | Manifest entry |
+|---|---|---|
+| ≤ 25 MB | Cloudflare **Pages** (shipped in `out/`, free) | `filename`, no `url` |
+| > 25 MB, R2 configured | Cloudflare **R2**, uploaded on deploy | `url` → R2 |
+| > 25 MB, R2 **not** configured | not served | `oversize: true`, shown as "Too large to host" |
+
+(The cap is `MAX_ARCHIVE_BYTES`, or a site's own `archiveMaxBytes`; `0` = no cap.)
+
+Oversize archives are staged during compose into `export/.r2-staging/<siteId>/archives/`
+(gitignored, kept out of `public/`), then uploaded to
+`<bucket>/<siteId>/archives/<file>.zip` **before** the Pages deploy runs, so the
+manifest URLs resolve immediately. **Every deploy path uploads them** — the editor's
+**Deploy** / **Build & deploy**, `pnpm ops deploy-site`, and `archilyzer deploy site`
+(which `pnpm run deploy` in `export/` runs). A build alone only *stages* them. The
+bucket is read from `settings.json`; the R2 credentials come from the environment
+(step 3 below).
+
+Uploads go through R2's **S3 API** (the AWS SDK's multipart uploader), not
+`wrangler r2 object put` — wrangler caps a single upload at **300 MiB**, and real
+live-chat archives are larger (multipart has no such limit). That is why archive
+uploads need S3 credentials in addition to the wrangler auth the Pages deploy uses.
+
+### One-time R2 setup
+
+**1. Create the bucket.** One bucket serves **all** your sites — objects are
+namespaced by `siteId`, so you do **not** need one bucket per site. In the dashboard
+(**R2 → Create bucket**) or with the same wrangler the deploy already uses:
+
+```sh
+wrangler r2 bucket create my-archives-bucket
+```
+
+**2. Expose it publicly.** R2 buckets are private by default, and the Downloads page
+links to plain `https://…/<file>.zip` URLs, so the bucket needs public read access.
+There are three ways; the first needs **no domain purchase** and is the recommended
+one:
+
+- **A Worker on `*.workers.dev` (recommended — no domain, no WHOIS):** deploy the
+ bundled `r2-proxy/` Worker to a free `<name>.workers.dev` subdomain (just like Pages'
+ `*.pages.dev`). It streams objects from the bucket and gives you edge caching **and**
+ tunable rate limiting in code — the same cost defenses a custom domain would, without
+ owning a domain. One Worker serves **every** site. See
+ [Option A](#option-a--a-worker-on-workersdev-no-domain) below. Set the editor's
+ public URL to `https://<name>.workers.dev`.
+- **Custom domain:** bucket **Settings → Public access → Custom Domains → Connect
+ Domain**, e.g. `archives.example.com` (the domain must be on your Cloudflare account).
+ Routes downloads through Cloudflare's CDN, WAF, caching, and dashboard rate-limiting
+ rules. See [Option B](#option-b--a-custom-domain).
+- **`r2.dev` subdomain (quick test only):** bucket **Settings → Public access → Allow
+ Access**. You get `https://pub-abc123.r2.dev` — free and domain-free, but Cloudflare
+ throttles `r2.dev` and gives you **no** cache / rate-limit control of your own. Fine
+ for a smoke test; use the Worker for a real instance.
+
+**3. The R2 S3 credentials.** Archive objects are uploaded over R2's S3-compatible
+API, which needs an Access Key ID + Secret. In the dashboard: **R2 → Manage R2 API
+Tokens → Create API token**, permission **Object Read & Write**, scoped to your
+bucket. Then set three environment variables where the editor (or the CLI) runs:
+
+```sh
+export R2_ACCESS_KEY_ID=<access key id>
+export R2_SECRET_ACCESS_KEY=<secret access key>
+export CLOUDFLARE_ACCOUNT_ID=<your account id> # used to build the S3 endpoint
+```
+
+The account id is on the R2 overview page; the S3 endpoint is derived as
+`https://<CLOUDFLARE_ACCOUNT_ID>.r2.cloudflarestorage.com`. Keep these in the
+environment (a shell profile, a systemd unit, a `.env` the editor loads) — **not** in
+the settings JSON, which is not a place for secrets. If a deploy has oversize
+archives to upload but these are unset, it fails *before* the Pages deploy (so the
+site never links to a missing file) with a message pointing back here.
+
+**4. Point the editor at the bucket.** In **Settings**:
+
+| Field | Value |
+|---|---|
+| **Archive overflow storage (R2 bucket)** | the bucket name, e.g. `my-archives-bucket` |
+| **Archive overflow public URL** | the public base from step 2, e.g. `https://archives.example.com` (no trailing slash needed) |
+
+Leave both blank to keep oversize archives dropped and shown as "Too large to host".
+On the next build and deploy, oversize archives upload to R2 and the Downloads page
+(and the header link, which reappears once anything is hostable) point at them.
+
+---
+
+## Securing downloads against cost-abuse
+
+The threat: someone scripts repeated downloads of large archives to run up your bill.
+
+**The reassuring part — R2 egress is free.** Unlike S3, Cloudflare R2 charges **$0**
+for bandwidth/egress. An attacker looping downloads of a 480 MB zip cannot run up a
+bandwidth bill. The *only* metered cost from reads is **Class B operations** (10M free
+per month, then $0.36/M) — and the defenses below make even that hard to reach.
+
+You get these controls one of two ways — **a Worker on `workers.dev`** (no domain) or
+**a custom domain**. Pick one. The Worker is recommended if you do not want to own a
+domain.
+
+### Option A — a Worker on workers.dev (no domain)
+
+The bundled **`r2-proxy/`** Worker is a small, self-contained project that binds the R2
+bucket and serves archive objects on a free `<name>.workers.dev` subdomain. It is a
+**pure passthrough**: the request path `<siteId>/archives/<file>.zip` maps straight to
+the bucket key, so **one Worker serves every site** (deploy it once, not per site).
+What it gives you, all on the free tier:
+
+- **Edge caching** — full downloads are cached with the Cache API, so repeat pulls skip
+ R2 (no billable Class B op). It honors the `Cache-Control` set on each object.
+- **Rate limiting** — Cloudflare's native, free rate-limit binding caps requests per
+ client IP + file (dashboard rate-limit rules need a paid zone; this does not).
+- **Path allow-listing** — it only serves `*/archives/*.zip`, never arbitrary keys.
+- **Range / resumable downloads** — honors `Range` requests so big zips can resume.
+
+**Deploy it (once for the whole instance):**
+
+1. Edit `r2-proxy/wrangler.toml` and set `bucket_name` to the **same bucket** you use in
+ the editor's Settings. Optionally rename the Worker (`name`) and tune the rate limit
+ (`limit` / `period`).
+2. From the repo root:
+ ```sh
+ cd r2-proxy
+ pnpm install # first time only
+ pnpm run deploy # = wrangler deploy, reusing your host wrangler auth
+ ```
+ wrangler prints the deployed URL, e.g. `https://ytdlp-archive-proxy.<you>.workers.dev`.
+ (An "unsafe fields are experimental" warning for the rate-limit binding is expected.)
+3. In the editor's **Settings**, set **Archive overflow public URL** to that
+ `workers.dev` URL. Re-deploy a site and its Downloads links resolve through the Worker.
+
+The Worker code is `r2-proxy/src/index.ts`; the caching and rate-limit logic are small
+and commented.
+
+> **Free-tier limit:** Workers Free allows **100,000 requests/day** (resets daily). Far
+> more than a downloads endpoint needs; if you ever exceed it, requests get a `429`
+> (fail closed — no surprise bill) until the next day, or upgrade to Workers Paid ($5/mo).
+
+The archive `Cache-Control` is set at upload time (`Cache-Control: public,
+max-age=3600`, constant `ARCHIVE_CACHE_CONTROL` in `common/publish/build.ts`); the
+Worker reads it back when caching. Archive filenames are stable and overwritten in
+place on re-deploy, so this 1-hour bound is what keeps a re-uploaded archive from being
+served stale for long — raise it if your archives rarely change.
+
+### Option B — a custom domain
+
+A custom domain has its own upsides — dashboard WAF, managed bot rules, and
+rate-limiting rules without touching code. Connect it per step 2 above and add these,
+in order of impact:
+
+1. **Edge caching.** Through a custom domain, Cloudflare's CDN caches each archive at
+ the edge (it honors the `Cache-Control: public, max-age=3600` set at upload), so
+ repeated downloads are served from cache and **never hit R2**. To make caching
+ aggressive, add a **Cache Rule** (**Caching → Cache Rules → Create**): *when* `URI
+ Path` contains `/archives/`, *then* Eligible for cache, **Edge TTL → Override → 1
+ day** (or longer). If you raise the TTL a lot, **purge the cache on deploy**
+ (dashboard **Caching → Purge**, or `wrangler`/the API) so a re-uploaded archive is
+ not served stale.
+2. **Rate limiting — the hard backstop.** A Rate Limiting rule caps how fast any
+ single client can pull archives, stopping a flood that misses cache (**Security →
+ WAF → Rate limiting rules → Create**; the free plan includes one rule): *when*
+ `URI Path` contains `/archives/`, *rate* e.g. **20 requests per 1 minute** per
+ client IP, *then* Block for 10 minutes (or Managed Challenge). Tune the threshold
+ to real usage — legitimate users download a handful of files, not dozens per
+ minute.
+3. **Bot Fight Mode + WAF managed rules.** **Security → Bots → Bot Fight Mode**
+ (free) blocks the low-effort scripted abuse that makes up most of this traffic; the
+ free **WAF managed ruleset** adds a baseline at no cost.
+4. **Hotlink protection (optional).** Stops other sites embedding your archives and
+ spending your ops budget serving their audience. A WAF custom rule (**Security →
+ WAF → Custom rules**): *when* `URI
+ Path` contains `/archives/` **and** `Referer` does not contain your domain **and**
+ `Referer` is not empty, *then* Block. (Allow an empty `Referer` so direct clicks and
+ privacy-conscious browsers still work.)
+
+### Billing alerts, and what to skip
+
+R2 has no hard spend cap, but Cloudflare **Notifications** (**Notifications → Add**)
+can email you when R2 storage or Class A/B operations cross a threshold — cheap
+insurance so nothing surprises you.
+
+**Signed URLs / token-gated downloads** are the heavyweight option — they add key
+management and friction for legitimate users. Given egress is free and caching
+neutralizes the ops cost, they are overkill for *cost* defense (the `r2-proxy` Worker
+is a plain passthrough, not an access gate). Reach for them only if you want *access
+control* (private archives), not cost control.
+
+### Cost expectations
+
+Measured across all sites in this project, total compressed archives are **≈ 2–4.5 GB**.
+Against R2's free tier:
+
+| Resource | Free tier / month | This project's usage |
+|---|---|---|
+| **Egress / bandwidth** | unlimited, **$0** | irrelevant — no egress charge exists |
+| **Storage** | 10 GB-month | ~2–4.5 GB — comfortable |
+| **Class A ops** (writes) | 1,000,000 | ~one PUT per archive per deploy — negligible |
+| **Class B ops** (reads) | 10,000,000 | mostly absorbed by CDN cache |
+
+The realistic bill for hosting these archives is **$0**. The configuration above exists
+to keep it that way under adversarial traffic, not because normal usage is close to
+any limit.
+
+---
+
+## Building every site in containers
+
+> **Not about running the apps in containers** — that is
+> [RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md), built from the root `Dockerfile`. This
+> is about *building sites*: fanning per-site export builds out across containers,
+> using `Dockerfile.build`. The two share nothing but the word "docker".
+
+**Build all sites** builds **every site in parallel** in isolated containers, then
+deploys them serially — a large speedup when you host several sites, and stronger
+isolation than building one site at a time in `export/`. It does so **whenever a
+container engine answers** (`docker version`), and builds serially on the host when
+none does. The **Build pipeline → mode** (Basic / Docker, in Settings or on the
+/sites toggle) is persisted and shown, but no build reads it today: it is a label,
+not a switch. A single site's build always runs in `export/`, one at a time.
+
+**Prerequisites.**
+
+- A container engine: **Docker**, or **podman** (set `DOCKER_BIN=podman`). Rootless
+ podman is a good fit — it maps container files to your host user automatically.
+- The editor host still needs Node + pnpm (Phase A and the deploys run on the host)
+ and the Cloudflare/R2 credentials in the environment. Credentials are **never**
+ passed into a container — deploy runs on the host.
+- Inside the runtime container (`docker compose up`) there is no `docker` binary, and
+ the pipeline falls back to the serial host build.
+
+The build image is built (and cached) automatically from `Dockerfile.build` the first
+time you run; **Build image** / **Dockerfile** in Settings override the tag and path.
+
+**How it works.** Trigger it with **Build all sites** on /sites, or `pnpm archilyzer
+build all`; both use containers when `docker version` answers. One job runs three
+ordered phases:
+
+1. **Phase A — shared, on the host, once.** The data phase (the search index and the
+ `.export-index` staging), then `archilyzer build archives` (the shared archive-zip
+ cache for the union of all sites' channels). Only the host writes this shared
+ state, so containers never race it. This phase is serial and is the long pole on a
+ cold build; on a warm rebuild it is near-instant (unchanged channels are skipped).
+2. **Phase B — per-site, in parallel containers.** Each site's compose + `next build`
+ runs in its own container (`docker/build-site.sh`: `archilyzer build site <id>
+ --nodata`), capped by **Max parallel builds**. Each writes an isolated `out/` under
+ `export/.export-builds/<siteId>/`. Containers mount the corpus, index, staging and
+ archive cache **read-only** (`ARCHIVES_READONLY=1`). Network is left on: `next
+ build` fetches the site's fonts through `next/font/google`; isolation comes from the
+ read-only mounts, the per-site output dir and the non-root user.
+3. **Phase C — deploy, on the host, serially.** After every build finishes, each built
+ site is deployed in turn (the R2 upload, then `wrangler pages deploy`). A single
+ site failing to build or deploy is reported and skipped; the rest still ship.
+
+If no container engine is available, the action logs a notice and falls back to a
+serial host build+deploy (one site at a time).
+
+**Mounts, per Phase-B container.**
+
+| Host | Container | Mode |
+|---|---|---|
+| `transcripts/` (corpus + `index.mdb` + archive cache) | `/data/transcripts` | ro |
+| `export/.export-index` (shared + per-site staging) | `/data/export/.export-index` | ro |
+| `export/.export-builds/<siteId>` (public/out/.next/caches) | `/site` | rw |
+| `settings.json` (build config, mounted fresh — not baked) | `/data/settings.json` | ro |
+
+The per-site `/site` mount is persistent, so incremental `next build` (`.next`) and
+incremental compose (`.compose-cache`) stay warm across builds.
+
+**Tuning.**
+
+- **Max parallel builds** (setting) — how many site containers run at once. Each `next
+ build` can use up to ~8 GB; a safe starting point is `floor(RAM_GB / 9)`.
+- `DOCKER_BIN` — the container binary (default `docker`; e.g. `podman`).
+- `DOCKER_BUILD_MEMORY`, `DOCKER_BUILD_CPUS` — optional per-container `--memory` /
+ `--cpus` caps so a fan-out cannot OOM or peg the host.
+- `BUILD_ARCHIVES=0` (or **Skip archive zips**) — skip the archive warm and the
+ per-site archive materialize for a faster build with no download bundles.
+
+**Notes.** Containers run as your host uid/gid (`-u`), so files under
+`.export-builds/` are host-owned, not root-owned. The image bakes the repo source and
+deps; a code change rebuilds it, but layer caching keeps that cheap (deps re-install
+only when the lockfile moves). `.export-builds/` is gitignored and excluded from the
+image build context. The editor mounts the host's `docker/build-site.sh` over the
+baked one, so an image older than the checkout still runs today's script.
+
+---
+
+## Reading a published archive from Claude Code
+
+Every published archive serves a machine contract (`/corpus.json`, `/llms.txt`), and
+the MCP server reads it over HTTP. Register it against your public URL — as
+`archilyzer`, which the tracked `/ask` and `/sweep` commands expect:
+
+```sh
+claude mcp add archilyzer \
+ --env TRANSCRIPT_SITE_URL=https://<your-site>.pages.dev \
+ -- pnpm -C "$PWD" archilyzer mcp
+```
+
+`archilyzer mcp` starts the same server as `pnpm --filter yt-dlp-transcript-mcp exec
+tsx src/index.ts` (the form [mcp/README.md](mcp/README.md) uses), with the
+environment passed through and nothing on stdout but the protocol.
diff --git a/README.md b/README.md
@@ -329,6 +329,7 @@ claude # then try: /ask what has he said about magic tournaments?
The two editor lines are optional: they let `fetch_clip` ask a local editor for clip
media (`WORKER_TOKEN` is the editor's own). Leave them out for research alone.
+`-- pnpm -C "$PWD" archilyzer mcp` starts the same server through the repo's CLI.
> **Register the server as `archilyzer`.** The shipped commands call
> `mcp__archilyzer__ask_plan` / `mcp__archilyzer__sweep_plan`, and that tool name
@@ -449,34 +450,25 @@ Two more pieces round out the workspace: the **project site** (`homepage/`) and
Transcripts and per-channel state live at `<repo>/transcripts/` — **its own git repo**,
untouched by the workspace.
-Everything resolves through `getPaths()` (`common/lib/paths.ts`) and can be overridden
-by environment variables:
-
-| Variable | Default | Purpose |
-| --- | --- | --- |
-| `TRANSCRIPTS_DIR` | `<repo>/transcripts` | The corpus: channels, media, index, job logs. |
-| `SAVED_VIDEOS_DIR` | inside `TRANSCRIPTS_DIR` | Persisted source-video store; can live on another disk. |
-| `SITES_DIR` | inside `TRANSCRIPTS_DIR` | Per-site configuration. |
-| `EXPORT_PUBLIC_DIR` | `<repo>/export/public` | Where the index writes paginated JSON. |
-| `SETTINGS_FILE` | `<repo>/settings.json` | Operational settings — every key, default and meaning is in [SETTINGS.md](SETTINGS.md). |
-| `YTDLP_BIN` | `yt-dlp` on PATH | The downloader. |
-| `WHISPER_BIN` / `WHISPER_MODEL` | `whisper-cli` on PATH | Default transcription backend and its model. |
-| `FFMPEG_BIN` / `FFPROBE_BIN` | on PATH | Transcode and duration checks. |
+Every path and binary resolves through `getPaths()` (`common/lib/paths.ts`), and each
+can be overridden by an environment variable — `TRANSCRIPTS_DIR` moves the whole corpus,
+`YTDLP_BIN` / `FFMPEG_BIN` / `WHISPER_BIN` name the tools. The full list, with every
+other variable the code reads, is **[ENVIRONMENT.md](ENVIRONMENT.md)** (generated from
+a list the tests hold to the code). `pnpm archilyzer doctor` prints which overrides are set
+and whether every tool this machine is configured to use is there.
Everything else lives in the editor's **Settings** page and is optional — a missing or
-partial settings file falls back to defaults. Full list in
-[SETUP.md](SETUP.md#configuration--environment-variables).
+partial settings file falls back to defaults. Every key is in [SETTINGS.md](SETTINGS.md).
## Publishing
The static export can be served by anything. The path with the most support is
Cloudflare Pages, with download archives too large for Pages' 25 MB per-file limit
-overflowing to R2 — see **[DEPLOY_CLOUDFLARE.md](DEPLOY_CLOUDFLARE.md)**, which also
-covers the configuration that keeps public archive downloads from being abused to run
-up costs.
-
-If you host several sites from one corpus, the opt-in Docker build pipeline builds them
-all in parallel in isolated containers — see **[DEPLOY_DOCKER.md](DEPLOY_DOCKER.md)**.
+overflowing to R2. **[PUBLISH.md](PUBLISH.md)** covers building and deploying a site,
+the hub and the homepage (from the editor, `pnpm ops` or `pnpm archilyzer`), previews,
+the R2 setup, the configuration that keeps public archive downloads from being abused
+to run up costs, and the opt-in pipeline that builds several sites in parallel in
+isolated containers.
Channels can sync automatically on a per-channel cadence via a cron heartbeat — see
**[SCHEDULED_SYNC.md](SCHEDULED_SYNC.md)**.
@@ -499,11 +491,11 @@ See [CONTRIBUTING.md](CONTRIBUTING.md) to work on the code.
| Document | Covers |
|---|---|
-| [SETUP.md](SETUP.md) | Full per-OS install, every environment variable, transcription backends. |
-| [CONTRIBUTING.md](CONTRIBUTING.md) | Workspace layout, tests, CLI shims, internals. |
+| [SETUP.md](SETUP.md) | Full per-OS install, transcription backends. |
+| [ENVIRONMENT.md](ENVIRONMENT.md) | Every environment variable, by audience (generated). |
+| [CONTRIBUTING.md](CONTRIBUTING.md) | Workspace layout, tests, the `archilyzer` CLI, internals. |
| [RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md) | `docker compose up` for the whole stack: exposure model, auth, GPU. |
-| [DEPLOY_CLOUDFLARE.md](DEPLOY_CLOUDFLARE.md) | Pages + R2, and cost-abuse protection. |
-| [DEPLOY_DOCKER.md](DEPLOY_DOCKER.md) | Parallel multi-site export builds in containers. |
+| [PUBLISH.md](PUBLISH.md) | Building and deploying sites: Pages + R2, cost-abuse protection, parallel builds in containers. |
| [SCHEDULED_SYNC.md](SCHEDULED_SYNC.md) | Unattended per-channel syncing. |
| [WORKTREES.md](WORKTREES.md) | Parallel checkouts and the port scheme. |
| [mcp/README.md](mcp/README.md) | The MCP server: tools, links, client setup. |
diff --git a/RUNNING_IN_DOCKER.md b/RUNNING_IN_DOCKER.md
@@ -1,13 +1,14 @@
# Running an archive in Docker
One command stands up a working archive server: the editor, the tools it drives
-(yt-dlp, ffmpeg, whisper.cpp), and a reverse proxy that is the only thing on the
-box with an open port.
+(yt-dlp, ffmpeg, and a transcription engine — whisper.cpp, or parakeet.cpp in the
+Vulkan image), and a reverse proxy that is the only thing on the box with an open
+port.
-> This document is about **running the apps**. [DEPLOY_DOCKER.md](DEPLOY_DOCKER.md)
-> is a different thing entirely — it is about fanning per-site *export builds* out
-> across containers, and it uses `Dockerfile.build`. Neither file affects the
-> other.
+> This document is about **running the apps**. Fanning per-site *export builds* out
+> across containers is a different thing entirely — it uses `Dockerfile.build`, and
+> it is in [PUBLISH.md](PUBLISH.md#building-every-site-in-containers). Neither
+> affects the other.
---
@@ -28,9 +29,12 @@ apps. Afterwards, `docker compose up -d` is seconds.
On first boot the container also:
- creates the corpus, config, models and builds volumes;
-- writes a `settings.json` **with one enabled whisper.cpp worker** (a settings
- file with zero workers means zero transcription slots, and auto-transcribe
- would report `no-workers` and quietly do nothing);
+- writes a `settings.json` **with one enabled worker for the image's engine**
+ (whisper.cpp here, parakeet.cpp in the Vulkan image). With no `workers` key, or
+ no file, `getSettings()` would synthesize `parallelTranscriptions` (default 2)
+ enabled whisper.cpp workers — two CPU slots, and never parakeet; a file that
+ says `"workers": []` has zero slots, and auto-transcribe would report
+ `no-workers` and quietly do nothing;
- downloads the `base.en` whisper model (~142 MB) into the models volume.
Every key that `settings.json` can carry, with its default, is in [SETTINGS.md](SETTINGS.md).
@@ -538,7 +542,7 @@ silently loses formats), `ffmpeg`/`ffprobe`, `whisper-cli` (statically linked),
### The multi-site build pipeline falls back inside a container
The editor can fan per-site export builds out across containers
-(`buildPipeline.mode = "docker"`, see DEPLOY_DOCKER.md). Inside a container there
+(`buildPipeline.mode = "docker"`, see [PUBLISH.md](PUBLISH.md#building-every-site-in-containers)). Inside a container there
is no `docker` binary, so that path is unavailable. It already handles this — the
build logs
diff --git a/SETTINGS.md b/SETTINGS.md
@@ -556,14 +556,14 @@ Default:
## `buildPipeline`
-How the static export is built: "basic" reuses the single export/ tree and serializes builds on one queue (the long-standing behavior); "docker" runs each site's build in an isolated container for safe parallelism. The Docker pipeline itself is a follow-up; this block persists the chosen mode plus the container/concurrency knobs the deploy page and the future orchestrator read.
+The build pipeline's settings. A single site builds in the shared export/ tree, serialized on one queue. Build all sites (and Build & deploy all) builds every site at once, each in its own container (Dockerfile.build, the image and maxParallelBuilds below), then deploys them serially, whenever a container engine answers, and serially on the host when none does — see PUBLISH.md. `mode` is persisted and shown on /sites, but no build path reads it today: it is a label, not a switch.
#### `buildPipeline`
| Key | Default | Description |
|---|---|---|
-| `mode` | `"basic"` | "basic" — `pnpm run build` in export/, serialized on the build queue (shared output tree, no safe parallelism). "docker" — isolated per-site container builds, parallel up to `maxParallelBuilds`. |
-| `maxParallelBuilds` | `2` | Cap on concurrent per-site container builds in docker mode. Ignored in basic mode (which is always serial). Clamped to [1, BUILD_MAX_PARALLEL_MAX]. |
+| `mode` | `"basic"` | "basic" or "docker". Persisted and shown on /sites, but no build path reads it today — a label, not a switch: Build all sites uses containers whenever a container engine answers, and builds serially on the host when none does, in either mode. |
+| `maxParallelBuilds` | `2` | Cap on concurrent per-site container builds when Build all sites runs in containers (whenever a container engine answers, whatever `mode` says). Clamped to [1, BUILD_MAX_PARALLEL_MAX]. |
| `dockerImage` | `"yt-dlp-transcript-browser-build"` | Tag of the reusable build image (built once, reused for every site). |
| `dockerfile` | `"Dockerfile.build"` | Dockerfile path relative to the monorepo root, used to (re)build the image. |
diff --git a/SETUP.md b/SETUP.md
@@ -46,7 +46,7 @@ homepage):
| **ffmpeg** + **ffprobe** | Audio transcode + duration checks for `transcribe` channels. | `ffmpeg` / `ffprobe` on `PATH` |
| A **transcription backend** | `handling: "transcribe"` channels only. Default is **whisper.cpp** (`whisper-cli`); `chough` and `parakeet.cpp` are alternatives. | `whisper-cli` on `PATH` |
| **rsync** | Backing up the saved-video store. | `rsync` on `PATH` |
-| **Docker** | Running the whole stack in containers ([RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md)), the parallel multi-site build ([DEPLOY_DOCKER.md](DEPLOY_DOCKER.md)), and the sharded e2e run (`pnpm e2e:sharded`). | — |
+| **Docker** | Running the whole stack in containers ([RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md)), the parallel multi-site build ([PUBLISH.md](PUBLISH.md#building-every-site-in-containers)), and the sharded e2e run (`pnpm e2e:sharded`). | — |
Every binary above is overridable by an environment variable (e.g. `YTDLP_BIN`) —
see [Configuration & environment variables](#configuration--environment-variables).
@@ -260,28 +260,27 @@ starting template. Both are generated from the settings schema
[CHANNEL.md](CHANNEL.md). Settings are optional — a missing/partial `settings.json` falls
back to built-in defaults, so the app runs out of the box.
-Paths and binaries resolve through `getPaths()` in `common/lib/paths.ts`. Override
-any of them via environment variables before launching:
+Paths and binaries resolve through `getPaths()` in `common/lib/paths.ts`, and each is
+overridden by an environment variable before launching — `TRANSCRIPTS_DIR` (the corpus,
+default `<repo>/transcripts`), `SETTINGS_FILE`, `YTDLP_BIN`, `FFMPEG_BIN` /
+`FFPROBE_BIN`, `WHISPER_BIN` / `WHISPER_MODEL`, `PARAKEET_CLI` / `PARAKEET_MODEL` and
+the rest. **Every variable the code reads is in [ENVIRONMENT.md](ENVIRONMENT.md)**, by
+audience: the path overrides, the runtime tokens and knobs (`WORKER_TOKEN`,
+`SYNC_TICK_URL`, the R2 credentials, …), the ports, the docker `ARCHILYZER_*` set and
+the test-only ones. It is generated from `common/lib/envVars.ts`, and a test fails when
+the code reads a variable that list does not declare.
-| Variable | Default | Purpose |
-| --- | --- | --- |
-| `TRANSCRIPTS_DIR` | `<repo>/transcripts` | Channels, archives, LMDB index, job logs. |
-| `SAVED_VIDEOS_DIR` | `<TRANSCRIPTS_DIR>/saved-videos` | Persisted source-video store (can live on a separate disk). |
-| `SITES_DIR` | `<TRANSCRIPTS_DIR>/sites` | Per-site config (`sites/<id>/site.json` — every key in [SITE.md](SITE.md)). |
-| `EXPORT_PUBLIC_DIR` | `<repo>/export/public` | Where the index writes paginated JSON. |
-| `SETTINGS_FILE` | `<repo>/settings.json` | Site-settings file. |
-| `YTDLP_BIN` | `yt-dlp` (PATH) | Pipeline downloader. |
-| `WHISPER_BIN` | `whisper-cli` (PATH) | whisper.cpp binary. |
-| `WHISPER_MODEL` | `~/whispercpp/whisper.cpp/models/ggml-base.en.bin` | whisper.cpp model file. |
-| `CHOUGH_BIN` / `CHOUGH_URL` / `CHOUGH_MODEL` | `chough` / — / — | chough backend binary, remote server, model. |
-| `PARAKEET_CLI` / `PARAKEET_MODEL` / `PARAKEET_STITCH_BIN` | `parakeet-cli` / — / `scripts/parakeet-stitch.mjs` | parakeet.cpp CLI, model, and wrapper. |
-| `FFMPEG_BIN` / `FFPROBE_BIN` | `ffmpeg` / `ffprobe` (PATH) | Audio transcode + duration checks. |
-| `RSYNC_BIN` | `rsync` (PATH) | Saved-video backup. |
-| `WORKER_TOKEN` | — | Bearer token for the remote-worker transcription API (set on both ends when used), and for the `/api/ops/*` HTTP layer over the editor's actions — see [RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md#driving-the-editor-without-a-browser) and `pnpm ops`. Unset means both surfaces are off. |
-
-Feature-area docs cover their own env vars: [SCHEDULED_SYNC.md](SCHEDULED_SYNC.md)
-(`SYNC_HEARTBEAT_SECONDS`, `SYNC_TICK_URL`, `SYNC_TICK_TOKEN`) and
-[DEPLOY_CLOUDFLARE.md](DEPLOY_CLOUDFLARE.md) (R2 credentials).
+To see what this machine has, run:
+
+```sh
+pnpm archilyzer doctor
+```
+
+It is read-only: the checkout, the corpus and each channel's media, `settings.json`,
+every binary the paths name plus each enabled worker's engine and model, umtool's
+report-pipeline tools, and this checkout's port block. It exits 1 only for something
+the machine is configured to do and cannot (an enabled worker's engine missing beside
+a corpus, a settings file that does not parse, an override naming a missing binary).
---
@@ -293,7 +292,7 @@ but you do need the Playwright browser:
```sh
npx playwright install chromium # one-time: download the test browser
-pnpm e2e # sequential run on the host (next dev, port 3001)
+pnpm e2e # sequential run on the host (next dev, port 3011)
```
For a faster parallel run, `pnpm e2e:sharded` splits the suite across N Docker
@@ -308,11 +307,12 @@ collisions, see [WORKTREES.md](WORKTREES.md).
## Where to go next
-- [README.md](README.md) — project overview, pipeline modes, CLI shims.
+- [README.md](README.md) — project overview, pipeline modes.
- [SCHEDULED_SYNC.md](SCHEDULED_SYNC.md) — automatic per-channel sync (internal
heartbeat or external cron).
-- [DEPLOY_CLOUDFLARE.md](DEPLOY_CLOUDFLARE.md) — deploying to Cloudflare Pages + R2
- archive overflow.
+- [PUBLISH.md](PUBLISH.md) — building and deploying sites: Cloudflare Pages, R2
+ archive overflow, parallel builds in containers.
+- [ENVIRONMENT.md](ENVIRONMENT.md) — every environment variable, by audience.
- [WORKTREES.md](WORKTREES.md) — parallel development with per-worktree ports.
- [mcp/README.md](mcp/README.md) — MCP server exposing the archive to Claude Code /
Desktop / Cursor.
diff --git a/WORKTREES.md b/WORKTREES.md
@@ -15,21 +15,10 @@ Each worktree gets an offset of `index * 100`, where `index` is the worktree's p
`git worktree list`. The **main** worktree is always first, so it keeps the original
defaults — nothing changes for the primary checkout.
-| Env var | Base (main) | Used by |
-|---|---|---|
-| `EDITOR_PORT` | 3001 | editor real dev/start (`pnpm dev:editor`) |
-| `PORT` | 3011 | editor test server + Playwright editor baseURL |
-| `EXPORT_PORT` | 3010 | export server launched by the editor e2e suite |
-| `EXPORT_DEV_PORT` | 3000 | export real dev (`pnpm dev:export`) |
-| `EXPORT_E2E_PORT` | 3020 | export's own Playwright suite |
-| `OLLAMA_STUB_PORT` | 11435 | digest-lane stub server in the editor e2e suite |
-| `HOMEPAGE_DEV_PORT` | 3030 | homepage (hub) real dev (`pnpm dev:homepage`) |
-| `HOMEPAGE_PORT` | 3031 | homepage static `serve out` (`pnpm start:homepage`) |
-| `HOMEPAGE_E2E_PORT` | 3040 | homepage's own Playwright suite |
-| `UMTOOL_PORT` | 3050 | um-clip triage tool real dev (`pnpm dev:umtool`) |
-| `UMTOOL_E2E_PORT` | 3051 | umtool's own Playwright suite |
-| `EDITOR_STUB_PORT` | 3052 | stub editor the umtool e2e suite fetches clips from |
-| `PLAYWRIGHT_BASE_URL` | `http://localhost:3011` | node-side fetches in specs |
+The port table itself is `common/lib/ports.mjs` — one copy, which `scripts/worktree.mjs`
+offsets and `pnpm archilyzer doctor` reports. Every port, its base and what uses it is in
+**[ENVIRONMENT.md → Ports](ENVIRONMENT.md#ports)** (generated from it). The injector also
+sets `PLAYWRIGHT_BASE_URL` (`http://localhost:<PORT>`) for node-side fetches in specs.
So worktree #1 runs editor on **3101**, test server on **3111**, export on **3110**, etc.
The hundreds digit is the worktree index. The allocator never overrides a variable already
diff --git a/common/bin/_cli.test.ts b/common/bin/_cli.test.ts
@@ -1,9 +1,13 @@
import { test } from "node:test";
import assert from "node:assert/strict";
+import { existsSync, readdirSync, readFileSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
import { parseArgv } from "./_parseFlags";
import {
argumentProblem,
booleanFlags,
+ passthroughCommand,
resolveCommand,
runCli,
usage,
@@ -195,7 +199,7 @@ test("parseCutArgs checks the changelog, the version and the date before anythin
test("a refused release command exits 2 and prints why, touching nothing", async () => {
const errors: string[] = [];
const out = { log: () => {}, error: (s: string) => errors.push(s) };
- const ctx = (positionals: string[]) => ({ positionals, flags: {}, env: {} });
+ const ctx = (positionals: string[]) => ({ positionals, flags: {}, env: {}, argv: [] });
assert.equal(await cutMain(ctx(["site", "next"]), out), 2);
assert.equal(await showMain(ctx(["hub"]), out), 2);
assert.equal(errors.length, 2);
@@ -309,3 +313,60 @@ test("release show says the latest release, its date and what is pending, and wh
assert.equal(formatAllLine([editor, exp]), "all: next 0.9.3, next-minor 0.10.0");
assert.equal(formatAllLine([editor]), null);
});
+
+// ── passthrough commands (one-core Phase 4 slice 3) ─────────────────────────
+
+test("a passthrough command gets every word after its path, verbatim and unchecked", async () => {
+ let seen: { argv: string[]; positionals: string[] } | null = null;
+ const table = [
+ cmd(["posts", "fetch"], {
+ passthrough: true,
+ run: async (ctx) => {
+ seen = { argv: ctx.argv, positionals: ctx.positionals };
+ return 7;
+ },
+ }),
+ cmd(["posts"], { maxPositionals: 1 }),
+ ];
+ const quiet = { log: () => {}, error: () => {} };
+ assert.equal(
+ await runCli(table, ["posts", "fetch", "--slug", "x", "--full", "--", "--weird=1"], {}, quiet),
+ 7,
+ );
+ assert.deepEqual(seen, { argv: ["--slug", "x", "--full", "--", "--weird=1"], positionals: [] });
+ // `--help` as the first word after the path is the usage line, not the bin's.
+ const logged: string[] = [];
+ assert.equal(
+ await runCli(table, ["posts", "fetch", "--help"], {}, { log: (s) => logged.push(s), error: () => {} }),
+ 0,
+ );
+ assert.match(logged.join("\n"), /archilyzer posts fetch/);
+});
+
+test("a passthrough path matches only the LEADING words, never a flag's value", () => {
+ const table = [cmd(["mcp"], { passthrough: true }), cmd(["run"], { maxPositionals: 9 })];
+ assert.equal(passthroughCommand(table, ["run", "digest", "--lane", "mcp"]), null);
+ assert.deepEqual(passthroughCommand(table, ["mcp", "--local", "x"])?.path, ["mcp"]);
+ assert.equal(passthroughCommand(table, ["--local", "mcp"]), null);
+});
+
+test("every bin in common/bin is reachable as a subcommand", () => {
+ const here = path.dirname(fileURLToPath(import.meta.url));
+ const table = readFileSync(path.join(here, "archilyzer.ts"), "utf8");
+ const reached = new Set([
+ ...[...table.matchAll(/import\("\.\/([\w-]+)"\)/g)].map((m) => m[1]),
+ ...[...table.matchAll(/script\(\[[^\]]*\],\s*"([\w-]+)\.ts"/g)].map((m) => m[1]),
+ ]);
+ const bins = readdirSync(here)
+ .filter((n) => n.endsWith(".ts") && !n.endsWith(".test.ts") && !n.startsWith("_"))
+ .map((n) => n.slice(0, -3))
+ .filter((n) => n !== "archilyzer");
+ assert.deepEqual(bins.filter((b) => !reached.has(b)), []);
+ // And every script row names a file that exists.
+ for (const m of table.matchAll(/script\(\[[^\]]*\],\s*"([\w-]+\.ts)"/g)) {
+ assert.ok(existsSync(path.join(here, m[1])), m[1]);
+ }
+ // No two rows share a path.
+ const paths = COMMANDS.map((c) => c.path.join(" "));
+ assert.equal(new Set(paths).size, paths.length);
+});
diff --git a/common/bin/_cli.ts b/common/bin/_cli.ts
@@ -16,6 +16,9 @@ export type CommandContext = {
positionals: string[];
flags: Record<string, FlagValue>;
env: NodeJS.ProcessEnv;
+ // A passthrough command's words after its path, verbatim (every other
+ // command gets []).
+ argv: string[];
};
export type Command = {
@@ -28,6 +31,12 @@ export type Command = {
flags?: Record<string, FlagKind>;
// At most this many positionals after the path (default 0).
maxPositionals?: number;
+ // The command parses its own arguments (a bin with its own flags, the MCP
+ // server): runCli hands it everything after its path, untouched, as `argv`,
+ // and checks nothing. Matched on the LEADING words of argv only, so a flag
+ // value can never be mistaken for its path. `--help` as the first word after
+ // the path still prints the usage line.
+ passthrough?: boolean;
// The exit code.
run: (ctx: CommandContext) => Promise<number>;
};
@@ -115,6 +124,15 @@ export async function runCli(
env: NodeJS.ProcessEnv = process.env,
out: { log: (s: string) => void; error: (s: string) => void } = console,
): Promise<number> {
+ const through = passthroughCommand(table, argv);
+ if (through) {
+ const rest = argv.slice(through.path.length);
+ if (rest[0] === "--help" || rest[0] === "-h") {
+ out.log(usage([through]));
+ return 0;
+ }
+ return through.run({ positionals: [], flags: {}, env, argv: rest });
+ }
const { flags, positionals } = parseArgv(argv, booleanFlags(table));
if (positionals.length === 0) {
(flags.help ? out.log : out.error)(usage(table));
@@ -134,7 +152,25 @@ export async function runCli(
out.error(`${problem}\n\n${usage([hit.command])}`);
return 2;
}
- return hit.command.run({ positionals: hit.rest, flags, env });
+ return hit.command.run({ positionals: hit.rest, flags, env, argv: [] });
+}
+
+/**
+ * The passthrough command whose path is the longest run of LEADING words of
+ * argv — or null. Only the leading words: `run digest --lane x` must never be
+ * read as some command named by a flag's value.
+ */
+export function passthroughCommand(
+ table: readonly Command[],
+ argv: readonly string[],
+): Command | null {
+ let best: Command | null = null;
+ for (const c of table) {
+ if (!c.passthrough || c.path.length > argv.length) continue;
+ if (!c.path.every((w, i) => argv[i] === w)) continue;
+ if (!best || c.path.length > best.path.length) best = c;
+ }
+ return best;
}
/**
diff --git a/common/bin/_spawnBin.ts b/common/bin/_spawnBin.ts
@@ -0,0 +1,81 @@
+// Run a TypeScript entry point as a CHILD process under tsx, stdio inherited,
+// and resolve with its exit code. For the CLI rows whose target parses its own
+// argv at import time (the `_parseFlags` bins) or is another package's entry
+// (the MCP server): importing either into the CLI's process would hand it the
+// CLI's argv, or make common depend on a package that depends on common.
+//
+// Signals: a terminal's Ctrl-C reaches the child directly (same process
+// group), so SIGINT is only held here — forwarding it too would deliver it
+// twice, which a bin that treats a second Ctrl-C as "force" would honour.
+// SIGTERM and SIGHUP come from a supervisor (an MCP client stopping its
+// server, say) and are forwarded.
+
+import { spawn } from "node:child_process";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+
+const COMMON = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
+export const REPO_ROOT = path.resolve(COMMON, "..");
+
+/** The tsx that a package's own scripts run (its devDependency). */
+export function tsxFor(packageDir: string): string {
+ return path.join(packageDir, "node_modules", ".bin", "tsx");
+}
+
+export function runChild(opts: {
+ command: string;
+ args: string[];
+ cwd?: string;
+ env?: NodeJS.ProcessEnv;
+}): Promise<number> {
+ return new Promise((resolve) => {
+ const child = spawn(opts.command, opts.args, {
+ cwd: opts.cwd,
+ env: opts.env ?? process.env,
+ stdio: "inherit",
+ });
+ const hold = () => {};
+ const forward = (sig: NodeJS.Signals) => () => {
+ if (!child.killed) child.kill(sig);
+ };
+ const onTerm = forward("SIGTERM");
+ const onHup = forward("SIGHUP");
+ process.on("SIGINT", hold);
+ process.on("SIGTERM", onTerm);
+ process.on("SIGHUP", onHup);
+ const done = (code: number) => {
+ process.off("SIGINT", hold);
+ process.off("SIGTERM", onTerm);
+ process.off("SIGHUP", onHup);
+ resolve(code);
+ };
+ child.on("error", (err) => {
+ console.error(`${path.basename(opts.command)}: ${err.message}`);
+ done(1);
+ });
+ child.on("exit", (code, signal) => {
+ done(code ?? (signal ? 128 + (signalNumber(signal) ?? 0) : 1));
+ });
+ });
+}
+
+function signalNumber(sig: NodeJS.Signals): number | undefined {
+ return ({ SIGHUP: 1, SIGINT: 2, SIGKILL: 9, SIGTERM: 15 } as Record<string, number>)[sig];
+}
+
+/**
+ * A common/bin script that reads process.argv itself, run with `argv` exactly
+ * as given. `heapMb` is the NODE_OPTIONS heap the old package.json script gave
+ * it (the corpus-wide passes need more than node's default).
+ */
+export function runBinScript(file: string, argv: string[], heapMb?: number): Promise<number> {
+ const env = { ...process.env };
+ if (heapMb) {
+ env.NODE_OPTIONS = `${env.NODE_OPTIONS ?? ""} --max-old-space-size=${heapMb}`.trim();
+ }
+ return runChild({
+ command: tsxFor(COMMON),
+ args: [path.join(COMMON, "bin", file), ...argv],
+ env,
+ });
+}
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -7,8 +7,6 @@
// Every row imports its implementation LAZILY, so `archilyzer sync tick` never
// loads the AWS SDK and `archilyzer settings example` never opens LMDB. The
// machinery (parser, lookup, usage) is `_cli.ts`; this file is only the table.
-//
-// Not here yet (one-core Phase 4 slice 3): `doctor`, `run <operation>`, `mcp`.
import type { Command } from "./_cli";
import { runCli, runIfEntryPoint } from "./_cli";
@@ -23,6 +21,30 @@ export const COMMANDS: Command[] = [
},
},
{
+ path: ["build", "stats"],
+ usage: "rebuild the stats datasets (reads the index; the data phase's second step)",
+ run: async () => {
+ await (await import("./build-stats")).main();
+ return 0;
+ },
+ },
+ {
+ path: ["build", "templates"],
+ usage: "bake each site's chart templates into its export staging dir (the data phase's third step)",
+ run: async () => {
+ await (await import("./build-chart-templates")).main();
+ return 0;
+ },
+ },
+ {
+ path: ["build", "archives"],
+ usage: "warm the shared archive-zip cache for every enabled site's channels, once",
+ run: async () => {
+ await (await import("./build-archives")).main();
+ return 0;
+ },
+ },
+ {
path: ["compose", "site"],
usage: "<id> compose one site's export/public (default: SITE_ID)",
maxPositionals: 1,
@@ -172,17 +194,95 @@ export const COMMANDS: Command[] = [
},
},
{
+ path: ["run"],
+ usage:
+ "<operation> <channel> [ids…] [--lane local|remote] run one catalogued operation over a channel offline, as the editor's job does (sync, downloads and transcription are refused: they run in the editor; it does not see the editor's lanes, so not beside one on the same channel)",
+ flags: { lane: "string" },
+ maxPositionals: Number.MAX_SAFE_INTEGER,
+ run: async ({ positionals, flags }) => {
+ const [operation, channel, ...ids] = positionals;
+ if (!operation || !channel) {
+ console.error("run: which operation, over which channel? `archilyzer run <operation> <channel> [ids…]`");
+ return 2;
+ }
+ const lane = flags.lane;
+ if (lane !== undefined && lane !== "local" && lane !== "remote") {
+ console.error(`run: --lane is local or remote, not "${String(lane)}"`);
+ return 2;
+ }
+ const { runOperation } = await import("./run-operation");
+ return runOperation(
+ { operation, channel, ids, lane: lane as "local" | "remote" | undefined },
+ { signal: interrupted() },
+ );
+ },
+ },
+ // The bins that parse their own flags, run as children with their argv
+ // verbatim (_spawnBin.ts says why). Each row is the whole integration.
+ script(["duplicates"], "duplicate-shorts.ts",
+ "[--threshold N] [--all-durations] [--blocking title|duration|both] [--near F] [--tolerance N] … on-demand duplicate detection (after index + stats)", 8192),
+ script(["posts", "fetch"], "fetch-posts.ts",
+ "--slug <channel> [--full] [--limit N] fetch a social channel's posts into its posts corpus"),
+ script(["posts", "check"], "check-post-availability.ts",
+ "--slug <channel> [--mode stale|unchecked|all] [--limit N] which archived posts were deleted at the source"),
+ script(["diarize", "backfill"], "diarize-backfill.ts",
+ "[--dry-run] [--scope transcribed|channel:<slug>|video:<slug>/<id>] [--limit N] [--force] … diarize videos whose audio is still on disk"),
+ script(["digest", "plan"], "digest-plan.ts",
+ "[--lane local|remote] [--channels a,b] [--top N] [--json] [--census] … price the digest backfill; writes nothing"),
+ script(["digest", "validate"], "digest-validate.ts",
+ "<channel> [<channel> …] score digests already on disk"),
+ script(["reconcile", "video-dirs"], "reconcile-video-dirs.ts",
+ "[--channel <slug>] [--dry-run] [--verbose] rename video dirs to the canonical id layout"),
+ script(["verify", "transcripts"], "verify-transcripts.ts",
+ "--channel <slug> list duplicate and missing transcripts"),
+ script(["migrate", "channel-priority"], "migrate-channel-priority.ts",
+ "[--dry-run] the one-shot channel-priority migration (plans/channel-priority.md, S5)"),
+ {
+ path: ["brand", "media"],
+ usage: "[--out <dir>] [--video-kit] render the Archilyzer Media channel's assets",
+ passthrough: true,
+ run: async ({ argv }) => (await import("./brand-media")).main(argv),
+ },
+ {
+ path: ["mcp"],
+ usage:
+ "[--local <dir>|--remote <url>|--hub <url>] start the MCP server on stdio (as `pnpm --filter yt-dlp-transcript-mcp exec tsx src/index.ts`)",
+ passthrough: true,
+ run: async ({ argv }) => (await import("./mcp")).main(argv),
+ },
+ {
path: ["sync", "tick"],
usage: "POST one scheduler tick to the editor (SYNC_TICK_URL, SYNC_TICK_TOKEN)",
run: async () => (await import("./sync-tick")).tick(),
},
{
+ path: ["docs", "env"],
+ usage: "[--check] write ENVIRONMENT.md from the declared env-var list (lib/envVars.ts)",
+ flags: { check: "boolean" },
+ run: async ({ flags }) =>
+ (await import("./env-docs")).main({ check: flags.check === true }),
+ },
+ {
+ path: ["docs", "files"],
+ usage: "[--check] write SITE.md + CHANNEL.md from the file schemas",
+ flags: { check: "boolean" },
+ run: async ({ flags }) =>
+ (await import("./file-schemas-docs")).main({ check: flags.check === true }),
+ },
+ {
path: ["settings", "example"],
usage: "[--check] write settings.json.example + SETTINGS.md from the schema",
flags: { check: "boolean" },
run: async ({ flags }) =>
(await import("./settings-example")).main({ check: flags.check === true }),
},
+ {
+ path: ["doctor"],
+ usage:
+ "[--json] read-only report: node, the checkout, the corpus, settings, every tool, the port block; exit 1 on a failure",
+ flags: { json: "boolean" },
+ run: async ({ flags }) => (await import("./doctor")).main({ json: flags.json === true }),
+ },
// Release notes, cut locally: the same writer as the /sites and /changelog
// form and POST /api/ops/cut-release, with no editor running (release.ts).
{
@@ -202,6 +302,17 @@ export const COMMANDS: Command[] = [
},
];
+// A row for a bin that parses its own argv: a passthrough command that runs
+// `common/bin/<file>` as a child with the words after its path.
+function script(path: string[], file: string, usage: string, heapMb?: number): Command {
+ return {
+ path,
+ usage,
+ passthrough: true,
+ run: async ({ argv }) => (await import("./_spawnBin")).runBinScript(file, argv, heapMb),
+ };
+}
+
// The site a site command names: its argument, else SITE_ID (which is how
// export's `build` / `deploy` scripts are called). Prints and returns null when
// there is neither.
diff --git a/common/bin/doctor.test.ts b/common/bin/doctor.test.ts
@@ -0,0 +1,231 @@
+// `archilyzer doctor` over temp checkouts and a stubbed PATH.
+//
+// Run with:
+// pnpm --filter yt-dlp-transcript-common test
+//
+// Every scenario builds its own tree under the OS temp dir, points a Paths at it
+// and hands the doctor a PATH holding only the fake binaries the scenario
+// wants. The last assertion of each is the one that matters most: the tree is
+// byte-for-byte and mtime-for-mtime what it was — the doctor wrote nothing.
+
+import { test, after } from "node:test";
+import assert from "node:assert/strict";
+import {
+ chmodSync,
+ mkdirSync,
+ mkdtempSync,
+ readdirSync,
+ rmSync,
+ statSync,
+ symlinkSync,
+ writeFileSync,
+} from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import { collectDoctorReport, renderDoctorReport, type DoctorReport } from "./doctor";
+
+const TMP = mkdtempSync(path.join(os.tmpdir(), "doctor-"));
+after(() => rmSync(TMP, { recursive: true, force: true }));
+
+let n = 0;
+function checkout(): { root: string; bin: string; paths: Paths } {
+ const root = path.join(TMP, `c${n++}`);
+ mkdirSync(path.join(root, "node_modules", ".pnpm"), { recursive: true });
+ writeFileSync(path.join(root, "pnpm-workspace.yaml"), "packages: []\n");
+ mkdirSync(path.join(root, "scripts"), { recursive: true });
+ writeFileSync(path.join(root, "scripts", "diarize.mjs"), "");
+ const bin = path.join(root, ".bin");
+ mkdirSync(bin);
+ const transcriptsDir = path.join(root, "transcripts");
+ const paths = {
+ monorepoRoot: root,
+ transcriptsDir,
+ channelsDir: path.join(transcriptsDir, "channels"),
+ lmdbPath: path.join(transcriptsDir, "index.mdb"),
+ settingsFile: path.join(root, "settings.json"),
+ ytdlpBin: "yt-dlp",
+ ffmpegBin: "ffmpeg",
+ ffprobeBin: "ffprobe",
+ galleryDlBin: "gallery-dl",
+ rsyncBin: "rsync",
+ findmntBin: "findmnt",
+ udisksctlBin: "udisksctl",
+ claudeBin: "claude",
+ diarizeBin: path.join(root, "scripts", "diarize.mjs"),
+ whisperModel: path.join(root, "models", "ggml-base.en.bin"),
+ parakeetModel: "",
+ parakeetCliBin: "parakeet-cli",
+ } as unknown as Paths;
+ return { root, bin, paths };
+}
+
+function fake(binDir: string, name: string, version = "1.2.3"): void {
+ const p = path.join(binDir, name);
+ writeFileSync(p, `#!/bin/sh\necho "${name} ${version}"\n`);
+ chmodSync(p, 0o755);
+}
+
+// Every file and dir under root, with its size and mtime: what "wrote nothing"
+// is checked against.
+function tree(root: string): string[] {
+ const out: string[] = [];
+ const walk = (d: string) => {
+ for (const e of readdirSync(d)) {
+ const p = path.join(d, e);
+ const st = statSync(p, { throwIfNoEntry: false });
+ out.push(`${path.relative(root, p)} ${st ? `${st.size} ${st.mtimeMs}` : "dangling"}`);
+ if (st?.isDirectory()) walk(p);
+ }
+ };
+ walk(root);
+ return out.sort();
+}
+
+async function run(c: ReturnType<typeof checkout>, env: NodeJS.ProcessEnv = {}): Promise<DoctorReport> {
+ return collectDoctorReport({
+ env: { PATH: c.bin, ...env },
+ paths: c.paths,
+ nodeVersion: "22.0.0",
+ portBlock: async () => null,
+ portInUse: async () => false,
+ umtoolTools: async () => null,
+ });
+}
+
+const status = (r: DoctorReport, id: string) => r.checks.find((c) => c.id === id)?.status;
+
+test("a clone with no corpus, no settings and no tools is not broken", async () => {
+ const c = checkout();
+ const before = tree(c.root);
+ const r = await run(c);
+ assert.equal(r.ok, true, renderDoctorReport(r));
+ assert.equal(status(r, "transcripts"), "info");
+ assert.equal(status(r, "settings.json"), "info");
+ assert.equal(status(r, "yt-dlp"), "info"); // absent, and nothing needs it
+ assert.deepEqual(tree(c.root), before);
+});
+
+test("a corpus without ffmpeg fails, and names what needs it", async () => {
+ const c = checkout();
+ mkdirSync(path.join(c.paths.channelsDir, "chan"), { recursive: true });
+ writeFileSync(path.join(c.paths.channelsDir, "chan", "config.json"), "{}");
+ writeFileSync(c.paths.settingsFile, "{}");
+ fake(c.bin, "yt-dlp", "2026.01.01");
+ const before = tree(c.root);
+ const r = await run(c);
+ assert.equal(r.ok, false);
+ assert.equal(status(r, "yt-dlp"), "ok");
+ assert.equal(status(r, "ffmpeg"), "fail");
+ assert.equal(status(r, "ffprobe"), "fail");
+ assert.match(r.checks.find((x) => x.id === "ffmpeg")!.detail, /a corpus is here/);
+ // No index yet: a warning, and the doctor did not build one.
+ assert.equal(status(r, "index"), "warn");
+ assert.deepEqual(tree(c.root), before);
+});
+
+test("a settings file that does not parse fails", async () => {
+ const c = checkout();
+ writeFileSync(c.paths.settingsFile, "{ not json");
+ const r = await run(c);
+ assert.equal(status(r, "settings.json"), "fail");
+ assert.equal(r.ok, false);
+ assert.match(renderDoctorReport(r), /FAIL settings\.json/);
+});
+
+test("an enabled whisper worker needs its engine and its model — beside a corpus", async () => {
+ const settings = {
+ workers: [
+ { id: "w1", name: "GPU", kind: "local", enabled: true, priority: 0, appId: "whisper-cpp", config: { bin: "whisper-cli" } },
+ ],
+ };
+ // No corpus yet: nothing to transcribe, so a warning, not a failure.
+ const bare = checkout();
+ writeFileSync(bare.paths.settingsFile, JSON.stringify(settings));
+ let r = await run(bare);
+ assert.equal(status(r, "engine:w1"), "warn");
+ assert.equal(r.ok, true, renderDoctorReport(r));
+
+ const c = checkout();
+ mkdirSync(path.join(c.paths.channelsDir, "chan"), { recursive: true });
+ writeFileSync(path.join(c.paths.channelsDir, "chan", "config.json"), "{}");
+ writeFileSync(c.paths.lmdbPath, "");
+ for (const b of ["yt-dlp", "ffmpeg", "ffprobe"]) fake(c.bin, b);
+ writeFileSync(c.paths.settingsFile, JSON.stringify(settings));
+ r = await run(c);
+ assert.equal(status(r, "engine:w1"), "fail");
+ assert.equal(status(r, "model:w1"), "fail");
+ assert.equal(r.ok, false);
+ fake(c.bin, "whisper-cli");
+ mkdirSync(path.dirname(c.paths.whisperModel), { recursive: true });
+ writeFileSync(c.paths.whisperModel, "model");
+ const before = tree(c.root);
+ r = await run(c);
+ assert.equal(status(r, "engine:w1"), "ok", renderDoctorReport(r));
+ assert.equal(status(r, "model:w1"), "ok");
+ assert.equal(r.ok, true, renderDoctorReport(r));
+ assert.deepEqual(tree(c.root), before);
+});
+
+test("an override naming a binary that is not there fails even when nothing needs it", async () => {
+ const c = checkout();
+ const r = await run(c, { YTDLP_BIN: "/nonexistent/yt-dlp" });
+ // The report shows the override…
+ assert.match(r.checks.find((x) => x.id === "overrides")!.detail, /YTDLP_BIN=\/nonexistent\/yt-dlp/);
+ // …but the probe here ran the Paths' binary name, so pass the override's
+ // path the way getPaths() would have.
+ const c2 = checkout();
+ (c2.paths as { ytdlpBin: string }).ytdlpBin = "/nonexistent/yt-dlp";
+ const r2 = await run(c2, { YTDLP_BIN: "/nonexistent/yt-dlp" });
+ assert.equal(status(r2, "yt-dlp"), "fail");
+ assert.equal(r2.ok, false);
+});
+
+test("an unmounted drive is a warning, not an empty channel and not a failure", async () => {
+ const c = checkout();
+ const chan = path.join(c.paths.channelsDir, "moved");
+ mkdirSync(chan, { recursive: true });
+ const target = path.join(c.root, "elsewhere", "moved", "data");
+ writeFileSync(path.join(chan, "config.json"), JSON.stringify({ dataDir: target }));
+ symlinkSync(target, path.join(chan, "data")); // the target does not exist
+ // No workers: a `{}` file would synthesize whisper workers, whose engine
+ // this machine does not have — a real failure, and not this test's.
+ writeFileSync(c.paths.settingsFile, JSON.stringify({ workers: [] }));
+ for (const b of ["yt-dlp", "ffmpeg", "ffprobe"]) fake(c.bin, b);
+ writeFileSync(c.paths.lmdbPath, "");
+ const before = tree(c.root);
+ const r = await run(c);
+ const media = r.checks.find((x) => x.id === "media")!;
+ assert.equal(media.status, "warn");
+ assert.match(media.detail, /moved: unreachable/);
+ assert.equal(r.ok, true, renderDoctorReport(r));
+ assert.deepEqual(tree(c.root), before);
+});
+
+test("node older than next's floor fails", async () => {
+ const c = checkout();
+ const r = await collectDoctorReport({
+ env: { PATH: c.bin },
+ paths: c.paths,
+ nodeVersion: "20.8.1",
+ portBlock: async () => null,
+ portInUse: async () => false,
+ umtoolTools: async () => null,
+ });
+ assert.equal(status(r, "node"), "fail");
+});
+
+test("umtool's table and the port block are reported, never failed", async () => {
+ const c = checkout();
+ const r = await collectDoctorReport({
+ env: { PATH: c.bin },
+ paths: c.paths,
+ nodeVersion: "22.0.0",
+ portBlock: async () => ({ label: "worktree #3 (offset 300)", ports: { EDITOR_PORT: "3301" } }),
+ portInUse: async (p) => p === 3301,
+ umtoolTools: async () => [{ id: "qrencode", bin: "qrencode", neededBy: ["report-video QR"], required: true }],
+ });
+ assert.equal(status(r, "qrencode"), "warn");
+ assert.match(r.checks.find((x) => x.id === "EDITOR_PORT")!.detail, /^3301 in use/);
+ assert.equal(r.ok, true);
+});
diff --git a/common/bin/doctor.ts b/common/bin/doctor.ts
@@ -0,0 +1,411 @@
+// `archilyzer doctor` — can this checkout do what it is configured to do?
+//
+// One read-only report over what used to be three places: the path and binary
+// overrides (lib/paths.ts), umtool's tool probe (`umtool doctor`, whose table
+// this reads and whose probe it shares — lib/toolProbe.mjs) and the port block
+// (scripts/worktree.mjs over lib/ports.mjs).
+//
+// STRICTLY READ-ONLY. It stats, reads and runs version flags. It never opens
+// LMDB (the index is stat'd, not opened), never mkdirs, never writes settings,
+// and never binds a port (a port is "in use" when a TCP connect succeeds). The
+// one process-state change is a chdir around umtool's table, which resolves a
+// path from the cwd; it is put back before anything else runs.
+//
+// A FAIL is something this machine is CONFIGURED to do and cannot: a settings
+// file that does not parse, an enabled worker whose engine is missing, a binary
+// an env override names that is not there, yt-dlp missing beside a corpus. It
+// exits 1. Everything else is a warning or a note — a clone with no corpus and
+// no media tools is a complete research environment over a published archive,
+// and must not be told it is broken.
+
+import { execFile } from "node:child_process";
+import { accessSync, constants, existsSync, readFileSync, statSync } from "node:fs";
+import { readdir } from "node:fs/promises";
+import net from "node:net";
+import path from "node:path";
+import { pathToFileURL } from "node:url";
+import { promisify } from "node:util";
+import type { Paths } from "../lib/paths";
+import { ENV_VARS } from "../lib/envVars";
+import { PORTS, portsForOffset } from "../lib/ports.mjs";
+import { probeTool } from "../lib/toolProbe.mjs";
+
+const execFileP = promisify(execFile);
+
+export type CheckStatus = "ok" | "info" | "warn" | "fail";
+
+export type DoctorCheck = {
+ section: string;
+ id: string;
+ status: CheckStatus;
+ detail: string;
+};
+
+export type DoctorReport = {
+ root: string;
+ checks: DoctorCheck[];
+ // No check failed.
+ ok: boolean;
+};
+
+type ToolSpec = {
+ id: string;
+ bin?: string;
+ file?: string;
+ args?: string[];
+ fallback?: string;
+ neededBy: string[];
+ required?: boolean;
+};
+type ToolReport = Awaited<ReturnType<typeof probeTool>>;
+
+export type DoctorDeps = {
+ env: NodeJS.ProcessEnv;
+ paths: Paths;
+ nodeVersion?: string;
+ probe?: (t: ToolSpec) => Promise<ToolReport>;
+ portInUse?: (port: number) => Promise<boolean>;
+ // The worktree's port env, or null when this is not a git checkout. Default:
+ // `node scripts/worktree.mjs ports`, the one implementation of "which block".
+ portBlock?: () => Promise<{ label: string; ports: Record<string, string> } | null>;
+ // umtool's report-pipeline table, or null when there is no umtool here.
+ umtoolTools?: () => Promise<ToolSpec[] | null>;
+};
+
+const MIN_NODE = [20, 9, 0] as const; // next 16's engines field
+
+export async function collectDoctorReport(deps: DoctorDeps): Promise<DoctorReport> {
+ const { env, paths } = deps;
+ const probe = deps.probe ?? ((t: ToolSpec) => probeTool(t, { env }));
+ const checks: DoctorCheck[] = [];
+ const add = (section: string, id: string, status: CheckStatus, detail: string) =>
+ checks.push({ section, id, status, detail });
+
+ // ── workspace ────────────────────────────────────────────────────────────
+ const W = "workspace";
+ const nodeV = deps.nodeVersion ?? process.versions.node;
+ add(W, "node", versionAtLeast(nodeV, MIN_NODE) ? "ok" : "fail",
+ `v${nodeV} (needs >= ${MIN_NODE.join(".")})`);
+ const root = paths.monorepoRoot;
+ if (existsSync(path.join(root, "pnpm-workspace.yaml"))) {
+ add(W, "checkout", "ok", root);
+ } else {
+ add(W, "checkout", "fail",
+ `no pnpm-workspace.yaml above ${root} — run this from inside the repo, or every path below is wrong`);
+ }
+ add(W, "dependencies", existsSync(path.join(root, "node_modules", ".pnpm")) ? "ok" : "fail",
+ existsSync(path.join(root, "node_modules", ".pnpm")) ? "node_modules installed" : "node_modules missing — run `pnpm install`");
+ const overrides = ENV_VARS.filter((v) => v.audience === "paths" && env[v.name] != null && env[v.name] !== "");
+ add(W, "overrides", "info",
+ overrides.length === 0
+ ? "no path or binary overrides set (ENVIRONMENT.md lists them)"
+ : overrides.map((v) => `${v.name}=${env[v.name]}`).join(" "));
+
+ // ── corpus ───────────────────────────────────────────────────────────────
+ const C = "corpus";
+ let channelSlugs: string[] = [];
+ if (!existsSync(paths.transcriptsDir)) {
+ add(C, "transcripts", "info",
+ `no local corpus at ${paths.transcriptsDir} — fine for research over a published archive (README, "Use an archive"); archiving needs one`);
+ } else {
+ channelSlugs = await listChannelSlugs(paths.channelsDir);
+ add(C, "transcripts", "ok", `${paths.transcriptsDir} (${channelSlugs.length} channel${channelSlugs.length === 1 ? "" : "s"})`);
+ const unreachable: string[] = [];
+ const { inspectChannelMedia } = await import("../lib/channelMedia");
+ for (const slug of channelSlugs) {
+ const loc = await inspectChannelMedia(paths, slug);
+ if (loc.status !== "ok" && loc.status !== "in-place") {
+ unreachable.push(`${slug}: ${loc.status} — ${loc.detail ?? ""}`.trim());
+ }
+ }
+ if (unreachable.length > 0) {
+ for (const u of unreachable) add(C, "media", "warn", u);
+ } else if (channelSlugs.length > 0) {
+ add(C, "media", "ok", "every channel's data/ is reachable");
+ }
+ const index = statOrNull(paths.lmdbPath);
+ if (index) {
+ add(C, "index", "ok", `${paths.lmdbPath} (stat only; built ${index.mtime.toISOString().slice(0, 16).replace("T", " ")})`);
+ } else if (channelSlugs.length > 0) {
+ add(C, "index", "warn", `no LMDB index at ${paths.lmdbPath} — run \`archilyzer index\``);
+ }
+ }
+
+ // ── settings ─────────────────────────────────────────────────────────────
+ const S = "settings";
+ const settingsText = readOrNull(paths.settingsFile);
+ if (settingsText === null) {
+ add(S, "settings.json", channelSlugs.length > 0 ? "warn" : "info",
+ `${paths.settingsFile} does not exist: every process runs on the defaults (SETTINGS.md)`);
+ } else {
+ let parsed: unknown;
+ try {
+ parsed = JSON.parse(settingsText);
+ } catch (err) {
+ parsed = err;
+ }
+ if (parsed && typeof parsed === "object" && !Array.isArray(parsed) && !(parsed instanceof Error)) {
+ add(S, "settings.json", "ok", paths.settingsFile);
+ } else {
+ add(S, "settings.json", "fail",
+ `${paths.settingsFile} is not a JSON object${parsed instanceof Error ? ` (${parsed.message})` : ""} — every process silently reads it as the defaults`);
+ }
+ }
+ // The effective settings, read the way every process reads them (defaults
+ // when the file is absent). Read-only: the reader never writes.
+ const { settingsFromFile } = await import("../lib/settings");
+ let settings: ReturnType<typeof settingsFromFile> | null = null;
+ try {
+ settings = settingsFromFile(paths.settingsFile);
+ } catch (err) {
+ add(S, "schema", "fail", `settings do not load: ${(err as Error).message}`);
+ }
+
+ // ── tools ────────────────────────────────────────────────────────────────
+ const T = "tools";
+ const hasCorpus = channelSlugs.length > 0;
+ const needs = new Map<string, string[]>(); // binary id -> why it is required
+ const need = (id: string, why: string) => needs.set(id, [...(needs.get(id) ?? []), why]);
+ // Needed by what is switched on, but with no corpus there is nothing for it
+ // to run on yet: a warning ("this will fail once you archive"), not a failure.
+ const later = new Map<string, string[]>();
+ const needLater = (id: string, why: string) =>
+ (hasCorpus ? need : (i: string, w: string) => later.set(i, [...(later.get(i) ?? []), w]))(id, why);
+ if (hasCorpus) {
+ need("yt-dlp", "a corpus is here (every fetch)");
+ need("ffmpeg", "a corpus is here (audio extraction)");
+ need("ffprobe", "a corpus is here (duration checks)");
+ }
+ const specs: ToolSpec[] = [
+ { id: "yt-dlp", bin: paths.ytdlpBin, neededBy: ["downloads, sync, metadata scans"] },
+ { id: "ffmpeg", bin: paths.ffmpegBin, args: ["-version"], neededBy: ["audio extraction", "diarization", "parakeet"] },
+ { id: "ffprobe", bin: paths.ffprobeBin, args: ["-version"], neededBy: ["duration checks"] },
+ { id: "gallery-dl", bin: paths.galleryDlBin, neededBy: ["X/Twitter post fetches"] },
+ { id: "rsync", bin: paths.rsyncBin, neededBy: ["saved-video backup"] },
+ { id: "findmnt", bin: paths.findmntBin, neededBy: ["storage-location identity (optional)"] },
+ // Looked up, not run: udisksctl has no version flag.
+ { id: "udisksctl", file: onPath(paths.udisksctlBin, env.PATH) ?? paths.udisksctlBin, neededBy: ["mounting a volume from /storage (optional)"] },
+ { id: "claude", bin: paths.claudeBin, neededBy: ["the metered digest lane"] },
+ { id: "diarize", file: paths.diarizeBin, neededBy: ["speaker diarization (the wrapper script)"] },
+ ];
+ if (settings) {
+ if (settings.savedVideoBackup.dest.trim() !== "") need("rsync", "a backup destination is set");
+ if (settings.digest.remoteEnabled) need("claude", "the metered digest lane is on");
+ if (settings.diarization.enabled) need("diarize", "diarization is on");
+ for (const w of settings.workers) {
+ if (!w.enabled || w.kind !== "local" || !w.appId) continue;
+ const label = `worker "${w.name || w.id}" (${w.appId}) is enabled`;
+ const cfg = w.config ?? {};
+ const { getTranscriptionApp } = await import("../lib/transcriptionApps");
+ const app = getTranscriptionApp(w.appId);
+ const bin = cfg.bin?.trim() || app.defaultBin();
+ // An engine is LOOKED UP, not run: a transcription engine's CLI has no
+ // cheap version flag, and running one to find out is not a doctor's call.
+ const id = `engine:${w.id}`;
+ specs.push({ id, file: onPath(bin, env.PATH) ?? bin, neededBy: [label] });
+ needLater(id, label);
+ const model = cfg.model?.trim() || (w.appId === "whisper-cpp" ? paths.whisperModel : w.appId === "parakeet" ? paths.parakeetModel : "");
+ if (w.appId === "parakeet") {
+ const cli = paths.parakeetCliBin;
+ specs.push({ id: `parakeet-cli:${w.id}`, file: onPath(cli, env.PATH) ?? cli, neededBy: [label] });
+ needLater(`parakeet-cli:${w.id}`, label);
+ }
+ if (w.appId !== "chough" || model) {
+ const mid = `model:${w.id}`;
+ if (model) specs.push({ id: mid, file: model, neededBy: [label] });
+ else add(T, mid, hasCorpus ? "fail" : "warn", `${label} but names no model, and ${w.appId === "parakeet" ? "PARAKEET_MODEL" : "WHISPER_MODEL"} is unset`);
+ if (model) needLater(mid, label);
+ }
+ }
+ if (!settings.workers.some((w) => w.enabled && w.kind !== "llm")) {
+ add(T, "workers", hasCorpus ? "warn" : "info", "no transcription worker is enabled — auto-transcribe does nothing (configure one on /workers)");
+ }
+ }
+ const overridden = new Set(
+ ENV_VARS.filter((v) => v.audience === "paths" && /_(BIN|CLI)$/.test(v.name) && env[v.name]).map((v) => v.name),
+ );
+ const envNameFor: Record<string, string> = {
+ "yt-dlp": "YTDLP_BIN", ffmpeg: "FFMPEG_BIN", ffprobe: "FFPROBE_BIN", "gallery-dl": "GALLERY_DL_BIN",
+ rsync: "RSYNC_BIN", findmnt: "FINDMNT_BIN", udisksctl: "UDISKSCTL_BIN", claude: "CLAUDE_BIN", diarize: "DIARIZE_BIN",
+ };
+ const reports = await Promise.all(specs.map((s) => probe(s)));
+ for (const r of reports) {
+ const why = needs.get(r.id);
+ const envName = envNameFor[r.id];
+ const explicit = envName !== undefined && overridden.has(envName);
+ const what = `${r.bin}${r.version ? ` ${r.version}` : ""} — ${r.neededBy.join(", ")}`;
+ if (r.present && !r.error) add(T, r.id, "ok", what);
+ else if (r.present) add(T, r.id, "warn", `${what}\n${r.error}`);
+ else if (why) add(T, r.id, "fail", `${r.error ?? "absent"} — required: ${why.join("; ")}`);
+ else if (later.has(r.id)) add(T, r.id, "warn", `${r.error ?? "absent"} — ${later.get(r.id)!.join("; ")}; it will fail once there is audio to transcribe`);
+ else if (explicit) add(T, r.id, "fail", `${r.error ?? "absent"} — ${envName} names it explicitly`);
+ else add(T, r.id, "info", `${r.error ?? "absent"} — needed only for ${r.neededBy.join(", ")}`);
+ }
+
+ // ── report pipeline (umtool) ─────────────────────────────────────────────
+ const U = "report pipeline (umtool)";
+ const umtoolSpecs = await (deps.umtoolTools ?? (() => loadUmtoolTools(root)))();
+ if (umtoolSpecs) {
+ const ur = await Promise.all(umtoolSpecs.map((s) => probe(s)));
+ for (const r of ur) {
+ const what = `${r.bin}${r.version ? ` ${r.version}` : ""} — ${r.neededBy.join(", ")}`;
+ add(U, r.id, r.present && !r.error ? "ok" : r.required ? "warn" : "info",
+ r.present && !r.error ? what : `${r.error ?? "absent"} — ${r.neededBy.join(", ")}${r.required ? " (`umtool doctor` exits 1 for this)" : ""}`);
+ }
+ }
+
+ // ── ports ────────────────────────────────────────────────────────────────
+ const P = "ports";
+ const block = await (deps.portBlock ?? (() => worktreePortBlock(root)))();
+ const inUse = deps.portInUse ?? tcpPortInUse;
+ const ports = block?.ports ?? portsForOffset(0);
+ add(P, "block", "info", block ? block.label : "not a git checkout: the base ports (common/lib/ports.mjs)");
+ for (const name of Object.keys(PORTS)) {
+ const port = Number(env[name] || ports[name]);
+ const busy = await inUse(port);
+ add(P, name, "info", `${port} ${busy ? "in use" : "free"}${env[name] ? " (from the environment)" : ""} — ${PORTS[name].what}`);
+ }
+
+ return { root, checks, ok: !checks.some((c) => c.status === "fail") };
+}
+
+export function renderDoctorReport(report: DoctorReport): string {
+ const mark: Record<CheckStatus, string> = { ok: "ok", info: "--", warn: "WARN", fail: "FAIL" };
+ const out: string[] = [`archilyzer doctor — ${report.root} (read-only)`];
+ let section = "";
+ for (const c of report.checks) {
+ if (c.section !== section) {
+ section = c.section;
+ out.push("", section);
+ }
+ const [first, ...rest] = c.detail.split("\n");
+ out.push(` ${mark[c.status].padEnd(5)}${c.id.padEnd(20)} ${first}`);
+ for (const r of rest) out.push(` ${"".padEnd(25)} ${r}`);
+ }
+ const fails = report.checks.filter((c) => c.status === "fail").length;
+ const warns = report.checks.filter((c) => c.status === "warn").length;
+ out.push(
+ "",
+ fails === 0
+ ? `no failures${warns ? `, ${warns} warning${warns === 1 ? "" : "s"}` : ""}`
+ : `${fails} failure${fails === 1 ? "" : "s"}${warns ? `, ${warns} warning${warns === 1 ? "" : "s"}` : ""} — exit 1`,
+ );
+ return out.join("\n");
+}
+
+export async function main(opts: { json?: boolean; env?: NodeJS.ProcessEnv } = {}): Promise<number> {
+ const { getPaths } = await import("../lib/paths");
+ const report = await collectDoctorReport({ env: opts.env ?? process.env, paths: getPaths() });
+ console.log(opts.json ? JSON.stringify(report, null, 2) : renderDoctorReport(report));
+ return report.ok ? 0 : 1;
+}
+
+// ── helpers ────────────────────────────────────────────────────────────────
+
+function versionAtLeast(v: string, min: readonly [number, number, number]): boolean {
+ const parts = v.split(".").map((n) => Number.parseInt(n, 10) || 0);
+ for (let i = 0; i < 3; i++) {
+ if ((parts[i] ?? 0) !== min[i]) return (parts[i] ?? 0) > min[i];
+ }
+ return true;
+}
+
+// A bare name resolved against PATH the way a spawn would, without running it.
+// A name with a slash is a path and is returned as it is (or null if absent).
+function onPath(bin: string, envPath: string | undefined): string | null {
+ if (bin.includes("/")) return existsSync(bin) ? bin : null;
+ for (const dir of (envPath ?? "").split(path.delimiter)) {
+ if (!dir) continue;
+ const p = path.join(dir, bin);
+ try {
+ accessSync(p, constants.X_OK);
+ if (statSync(p).isFile()) return p;
+ } catch {
+ /* not here */
+ }
+ }
+ return null;
+}
+
+function statOrNull(p: string) {
+ try {
+ return statSync(p);
+ } catch {
+ return null;
+ }
+}
+
+function readOrNull(p: string): string | null {
+ try {
+ return readFileSync(p, "utf8");
+ } catch {
+ return null;
+ }
+}
+
+// The channel list the editor sees: DIRECTORIES under channels/ (a symlinked
+// channel dir is not a channel — AGENTS.md), dot-dirs excluded.
+async function listChannelSlugs(channelsDir: string): Promise<string[]> {
+ try {
+ const ents = await readdir(channelsDir, { withFileTypes: true });
+ return ents.filter((d) => d.isDirectory() && !d.name.startsWith(".")).map((d) => d.name).sort();
+ } catch {
+ return [];
+ }
+}
+
+// umtool's table, from its own module (plain ESM, no app imports) by a runtime
+// import of the file — common does not depend on umtool, and must not. Its
+// facecrop path is resolved from the cwd, which is umtool/ when umtool runs it,
+// so that is the cwd it is evaluated under here.
+async function loadUmtoolTools(root: string): Promise<ToolSpec[] | null> {
+ const dir = path.join(root, "umtool");
+ const file = path.join(dir, "lib", "tools.mjs");
+ if (!existsSync(file)) return null;
+ const mod = (await import(pathToFileURL(file).href)) as { TOOLS: () => ToolSpec[] };
+ const prev = process.cwd();
+ process.chdir(dir);
+ try {
+ return mod.TOOLS();
+ } finally {
+ process.chdir(prev);
+ }
+}
+
+// Which port block this checkout has: scripts/worktree.mjs is the one
+// implementation (it asks git), so it is asked rather than re-derived.
+async function worktreePortBlock(
+ root: string,
+): Promise<{ label: string; ports: Record<string, string> } | null> {
+ const script = path.join(root, "scripts", "worktree.mjs");
+ if (!existsSync(script)) return null;
+ try {
+ const { stdout, stderr } = await execFileP(process.execPath, [script, "ports"], {
+ cwd: root,
+ timeout: 10_000,
+ });
+ const ports: Record<string, string> = {};
+ for (const line of stdout.split("\n")) {
+ const eq = line.indexOf("=");
+ if (eq > 0) ports[line.slice(0, eq)] = line.slice(eq + 1).trim();
+ }
+ const label = (stderr.split("\n")[0] ?? "").trim() || "worktree block";
+ return { label, ports };
+ } catch {
+ return null;
+ }
+}
+
+// In use = something accepts a TCP connection on 127.0.0.1. Never binds.
+function tcpPortInUse(port: number): Promise<boolean> {
+ return new Promise((resolve) => {
+ const sock = net.connect({ port, host: "127.0.0.1" });
+ const done = (v: boolean) => {
+ sock.destroy();
+ resolve(v);
+ };
+ sock.setTimeout(500, () => done(false));
+ sock.once("connect", () => done(true));
+ sock.once("error", () => done(false));
+ });
+}
diff --git a/common/bin/env-docs.ts b/common/bin/env-docs.ts
@@ -0,0 +1,34 @@
+// WRITE ENVIRONMENT.MD FROM THE DECLARED LIST (lib/envVars.ts).
+//
+// archilyzer docs env [--check]
+//
+// `--check` writes nothing and returns 1 when the committed file differs from
+// what the list generates (the same claim lib/envVars.test.ts makes). The
+// sibling of file-schemas-docs.ts (SITE.md, CHANNEL.md) and settings-example.ts
+// (SETTINGS.md).
+
+import { readFile, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { renderEnvironmentMarkdown } from "../lib/envVars";
+import { runIfEntryPoint } from "./_cli";
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+export async function main(opts: { check?: boolean } = {}): Promise<number> {
+ const file = path.join(REPO, "ENVIRONMENT.md");
+ const want = renderEnvironmentMarkdown();
+ if (opts.check) {
+ const have = await readFile(file, "utf8").catch(() => "");
+ if (have !== want) {
+ console.error("ENVIRONMENT.md is stale — regenerate it with `archilyzer docs env`");
+ return 1;
+ }
+ return 0;
+ }
+ await writeFile(file, want);
+ console.log("wrote ENVIRONMENT.md");
+ return 0;
+}
+
+runIfEntryPoint(import.meta.url, () => main({ check: process.argv.includes("--check") }));
diff --git a/common/bin/mcp.test.ts b/common/bin/mcp.test.ts
@@ -0,0 +1,59 @@
+// `archilyzer mcp` starts the real MCP server, and stdout stays JSON-RPC.
+//
+// Run with:
+// pnpm --filter yt-dlp-transcript-common test
+//
+// Spawns the CLI the way `claude mcp add … -- <command>` would, sends the 2025
+// initialize request, and reads the answer. The arguments after `mcp` reach the
+// server untouched (the `--local` dir comes back in its stderr banner), and the
+// FIRST line on stdout is the server's JSON-RPC answer: anything the CLI printed
+// before it would break every MCP client.
+
+import { test, after } from "node:test";
+import assert from "node:assert/strict";
+import { spawn } from "node:child_process";
+import { mkdtempSync, rmSync } from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const TSX = path.join(HERE, "..", "node_modules", ".bin", "tsx");
+const DIR = mkdtempSync(path.join(os.tmpdir(), "archilyzer-mcp-"));
+after(() => rmSync(DIR, { recursive: true, force: true }));
+
+test("archilyzer mcp answers initialize on stdout, with its argv passed through", { timeout: 60_000 }, async () => {
+ const child = spawn(TSX, [path.join(HERE, "archilyzer.ts"), "mcp", "--local", DIR], {
+ stdio: ["pipe", "pipe", "pipe"],
+ });
+ // Listened for from the start: a CLI that refuses at once exits before any
+ // later listener could see it.
+ const exited = new Promise<number | null>((r) => child.on("exit", (c) => r(c)));
+ let stdout = "";
+ let stderr = "";
+ child.stdout.on("data", (b) => (stdout += b));
+ child.stderr.on("data", (b) => (stderr += b));
+ child.stdin.write(
+ `${JSON.stringify({
+ jsonrpc: "2.0",
+ id: 1,
+ method: "initialize",
+ params: { protocolVersion: "2025-06-18", capabilities: {}, clientInfo: { name: "test", version: "0" } },
+ })}\n`,
+ );
+ const deadline = Date.now() + 30_000;
+ let ended = false;
+ void exited.then(() => (ended = true));
+ while (!stdout.includes("\n") && !ended && Date.now() < deadline) {
+ await new Promise((r) => setTimeout(r, 50));
+ }
+ child.stdin.end();
+ const code = await exited;
+ const first = stdout.split("\n")[0];
+ const msg = JSON.parse(first) as { id: number; result?: { serverInfo?: { name?: string } } };
+ assert.equal(msg.id, 1);
+ assert.equal(msg.result?.serverInfo?.name, "yt-dlp-transcript-mcp");
+ assert.match(stderr, new RegExp(`default corpus: local:${DIR.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}`));
+ // stdin closed: the server ends, and the CLI ends with it.
+ assert.equal(code, 0);
+});
diff --git a/common/bin/mcp.ts b/common/bin/mcp.ts
@@ -0,0 +1,23 @@
+// `archilyzer mcp [args…]` — start the MCP server, exactly as
+// `pnpm --filter yt-dlp-transcript-mcp exec tsx src/index.ts [args…]` does:
+// mcp's own tsx, cwd mcp/, the environment passed through untouched (the
+// server reads TRANSCRIPT_SITE_URL / TRANSCRIPT_HUB_URL / TRANSCRIPT_LOCAL_DIR,
+// ARCHILYZER_EDITOR_URL and WORKER_TOKEN from it), stdio inherited.
+//
+// A CHILD, never an import: the mcp package depends on common, so common
+// importing mcp would make the dependency a cycle. And NOTHING may be printed
+// to stdout on the way in — stdout is the JSON-RPC channel.
+
+import { existsSync } from "node:fs";
+import path from "node:path";
+import { REPO_ROOT, runChild, tsxFor } from "./_spawnBin";
+
+export async function main(argv: string[], root = REPO_ROOT): Promise<number> {
+ const dir = path.join(root, "mcp");
+ const entry = path.join(dir, "src", "index.ts");
+ if (!existsSync(entry)) {
+ console.error(`mcp: no MCP server at ${entry}`);
+ return 1;
+ }
+ return runChild({ command: tsxFor(dir), args: ["src/index.ts", ...argv], cwd: dir });
+}
diff --git a/common/bin/run-operation.test.ts b/common/bin/run-operation.test.ts
@@ -0,0 +1,203 @@
+// `archilyzer run <operation> <channel> [ids…]` over a temp corpus.
+//
+// Run with:
+// pnpm --filter yt-dlp-transcript-common test
+//
+// Its own file for the SETTINGS SEAM, like operationBatchRelocation.test.ts:
+// getPaths() memoizes its first answer, so TRANSCRIPTS_DIR and SETTINGS_FILE are
+// set before anything imports it. Settings are re-read on every call, so each
+// test writes the file it needs.
+//
+// Every test that could start a job carries a timeout. A lane whose gate is
+// shut HOLDS rather than failing: with only unref'd poll timers left, node
+// cancels the remaining tests (exit 1), and the timeout bounds the case where
+// something else keeps the loop alive.
+//
+// The operation that RUNS here is diarization over videos that have a
+// transcript and no audio: each classifies `missing-input` (with re-download
+// off), which is counted and never dispatched — no engine, no model, no
+// network — and still goes through the whole job: the record, the log, the
+// batch, the summary line and the snapshot.
+
+import { mkdtempSync, readdirSync, readFileSync, writeFileSync } from "node:fs";
+import { mkdir, rm, symlink, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { test, after } from "node:test";
+import assert from "node:assert/strict";
+
+const ROOT = mkdtempSync(path.join(os.tmpdir(), "run-operation-"));
+process.env.TRANSCRIPTS_DIR = ROOT;
+const SETTINGS_FILE = path.join(ROOT, "settings.json");
+process.env.SETTINGS_FILE = SETTINGS_FILE;
+
+const { runOperation, offlineRefusal } = await import("./run-operation");
+const { getPaths } = await import("../lib/paths");
+const { operationCatalog } = await import("../lib/operations");
+
+after(() => rm(ROOT, { recursive: true, force: true }));
+
+function settings(over: Record<string, unknown> = {}): void {
+ writeFileSync(
+ SETTINGS_FILE,
+ JSON.stringify({
+ autoQueue: { backfill: { enabled: true, held: false } },
+ backfill: { allowRedownload: false, concurrency: 1 },
+ diarization: { enabled: true, segModel: "/models/seg.onnx", embModel: "/models/emb.onnx" },
+ attribution: { enabled: false, diarizedEnabled: false, textOnlyEnabled: false },
+ ...over,
+ }),
+ );
+}
+
+async function seed(slug: string, ids: string[]): Promise<void> {
+ const channelDir = path.join(getPaths().channelsDir, slug);
+ await mkdir(channelDir, { recursive: true });
+ await writeFile(
+ path.join(channelDir, "config.json"),
+ JSON.stringify({ handling: "transcribe", url: "https://example.com/c" }),
+ );
+ for (const id of ids) {
+ const videoDir = path.join(channelDir, "data", id);
+ await mkdir(videoDir, { recursive: true });
+ await writeFile(
+ path.join(videoDir, "transcript.json"),
+ JSON.stringify({ transcription: [{ text: "hello", offsets: {} }] }),
+ );
+ }
+}
+
+function capture() {
+ const o = { stdout: "", stderr: "" };
+ return {
+ o,
+ out: {
+ log: (s: string) => (o.stdout += `${s}\n`),
+ error: (s: string) => (o.stderr += `${s}\n`),
+ write: (s: string) => (o.stdout += s),
+ },
+ };
+}
+
+const jobRecords = () => {
+ try {
+ return readdirSync(getPaths().jobsDir).filter((n) => n.endsWith(".json"));
+ } catch {
+ return [];
+ }
+};
+
+test("an unknown operation is refused with the list", async () => {
+ const { o, out } = capture();
+ const code = await runOperation({ operation: "diarisation", channel: "x", ids: [] }, { out });
+ assert.equal(code, 2);
+ assert.match(o.stderr, /no operation "diarisation"/);
+ assert.match(o.stderr, /Runs here: .*diarization.*digest/);
+ assert.match(o.stderr, /run only by the editor: .*sync.*transcription/);
+});
+
+test("sync, the scan, downloads and transcription are refused with a sentence, not half-run", async () => {
+ for (const id of ["sync", "metadata-scan", "download"]) {
+ const { o, out } = capture();
+ assert.equal(await runOperation({ operation: id, channel: "x", ids: [] }, { out }), 1, id);
+ assert.match(o.stderr, /download queue inside the editor/, id);
+ }
+ const { o, out } = capture();
+ assert.equal(await runOperation({ operation: "transcription", channel: "x", ids: [] }, { out }), 1);
+ assert.match(o.stderr, /worker pool/);
+ // Every catalogued operation has an answer, and exactly the registry's run.
+ const runs = operationCatalog().filter((op) => offlineRefusal(op) === null).map((op) => op.id);
+ assert.deepEqual(runs.sort(), ["attribution-diarized", "attribution-text", "diarization", "digest"]);
+});
+
+test("an unknown channel, a switched-off operation, a paused lane and a stray --lane are refused", { timeout: 60_000 }, async () => {
+ settings();
+ await seed("known", ["v1"]);
+ let c = capture();
+ assert.equal(await runOperation({ operation: "diarization", channel: "nope", ids: [] }, { out: c.out }), 2);
+ assert.match(c.o.stderr, /no channel "nope"/);
+
+ c = capture();
+ assert.equal(await runOperation({ operation: "attribution-text", channel: "known", ids: [] }, { out: c.out }), 1);
+ assert.match(c.o.stderr, /switched off in settings\.json/);
+
+ settings({ autoQueue: { backfill: { enabled: true, held: true } } });
+ c = capture();
+ assert.equal(await runOperation({ operation: "diarization", channel: "known", ids: [] }, { out: c.out }), 1);
+ assert.match(c.o.stderr, /backfill lane is paused/);
+
+ settings();
+ c = capture();
+ assert.equal(
+ await runOperation({ operation: "diarization", channel: "known", ids: [], lane: "local" }, { out: c.out }),
+ 2,
+ );
+ assert.match(c.o.stderr, /--lane is the digest engine lane/);
+});
+
+test("the media guard refuses an unmounted channel before any job exists", { timeout: 60_000 }, async () => {
+ settings();
+ const slug = "moved";
+ const channelDir = path.join(getPaths().channelsDir, slug);
+ await mkdir(channelDir, { recursive: true });
+ const target = path.join(ROOT, "platter", slug, "data"); // never created
+ await writeFile(
+ path.join(channelDir, "config.json"),
+ JSON.stringify({ handling: "transcribe", url: "https://example.com/m", dataDir: target }),
+ );
+ await symlink(target, path.join(channelDir, "data"));
+ const before = jobRecords();
+ const { o, out } = capture();
+ const code = await runOperation({ operation: "diarization", channel: slug, ids: [] }, { out });
+ assert.equal(code, 1);
+ assert.match(o.stderr, /moved/);
+ assert.match(o.stderr, /not mounted|unreachable|does not exist/);
+ assert.deepEqual(jobRecords(), before, "no job record for a refused run");
+});
+
+test("diarization runs through the editor's job body: record, log, summary, snapshot", { timeout: 60_000 }, async () => {
+ settings();
+ await seed("chan", ["v1", "v2"]);
+ const before = new Set(jobRecords());
+ const { o, out } = capture();
+ const code = await runOperation({ operation: "diarization", channel: "chan", ids: [] }, { out });
+ assert.equal(code, 0, o.stderr + o.stdout);
+ assert.match(o.stdout, /Backfill chan: 0 done, 0 already current, 0 failed; 2 still need their media re-acquired/);
+ const added = jobRecords().filter((n) => !before.has(n));
+ assert.equal(added.length >= 1, true, "a job record was written");
+ const records = added.map((n) => JSON.parse(readFileSync(path.join(getPaths().jobsDir, n), "utf8")));
+ const job = records.find((r) => r.kind === "backfill-channel");
+ assert.ok(job, JSON.stringify(records));
+ assert.equal(job.channelSlug, "chan");
+ assert.equal(job.status, "done");
+ assert.deepEqual(job.spec?.params?.kindIds, ["diarization"]);
+ // The report the editor would have refreshed at job end.
+ assert.ok(
+ readdirSync(path.join(getPaths().channelsDir, "chan")).some((n) => n.startsWith("snapshot")),
+ "the channel snapshot was regenerated",
+ );
+});
+
+test("ids scope the run, and ids with no data dir are named", { timeout: 60_000 }, async () => {
+ settings();
+ await seed("scoped", ["v1", "v2", "v3"]);
+ const { o, out } = capture();
+ const code = await runOperation(
+ { operation: "diarization", channel: "scoped", ids: ["v2", "gone", "v2"] },
+ { out },
+ );
+ assert.equal(code, 0, o.stderr);
+ assert.match(o.stderr, /1 of 2 id\(s\) have no data\/<id>\/ on disk and are skipped: gone/);
+ assert.match(o.stdout, /Backfill scoped: 0 done, 0 already current, 0 failed; 1 still need their media re-acquired/);
+});
+
+test("ids none of which is on disk are refused, and no job is started", { timeout: 60_000 }, async () => {
+ settings();
+ await seed("empty-ids", ["v1"]);
+ const before = jobRecords();
+ const { o, out } = capture();
+ const code = await runOperation({ operation: "diarization", channel: "empty-ids", ids: ["gone", "also-gone"] }, { out });
+ assert.equal(code, 1);
+ assert.match(o.stderr, /none of the 2 id\(s\) has a data\/<id>\/ on disk/);
+ assert.deepEqual(jobRecords(), before);
+});
diff --git a/common/bin/run-operation.ts b/common/bin/run-operation.ts
@@ -0,0 +1,188 @@
+// `archilyzer run <operation> <channel> [ids…]` — one catalogued operation over
+// one channel, offline, through the body the editor's job runs.
+//
+// THE SAME BODY: `runOperationChannelJob` (controller/operationJobs.ts), which
+// the /channels group buttons, the stage cards and the lane runners call. So a
+// run here is a real job — kind `backfill-channel` or `digest-channel-*`, a
+// record and a log under `transcripts/.jobs/`, the same summary line, and the
+// media guard: `runManagedFunction` refuses a `needsMedia` kind for a channel
+// whose media is unreachable, before any record is made.
+//
+// WHAT RUNS HERE AND WHAT DOES NOT, decided from the descriptor, not the id:
+// - the registry operations (dispatch "backfill": diarization, both
+// attribution passes, digest) run — their engines are subprocesses and
+// HTTP endpoints this process can reach as well as the editor can;
+// - an `external` operation is refused with a sentence. Sync, the metadata
+// scan and downloads run on the per-platform download queue, which paces
+// every request to one source INSIDE the editor; a second process would
+// fetch beside it. Transcription is dispatched across the editor's worker
+// pool (leases, tiers, drain), which does not exist out here.
+//
+// WHAT IS DIFFERENT FROM THE EDITOR'S RUN, deliberately and in the open:
+// - it does not see the editor's jobs: the backfill lane's yield-to-
+// transcription reads THIS process's transcription activity, which is none;
+// - a paused lane is REFUSED up front rather than held: the batch would
+// idle-wait for a resume that only the editor can give.
+//
+// - it does not see the editor's lanes either: a `run` over a channel the
+// editor's lane is working on at the same moment does the same videos
+// twice (the writes are atomic, so it is wasted work, not damage). Run it
+// when that lane is paused or elsewhere.
+//
+// Ctrl-C cancels the job the way the editor's Cancel does. Exit: 0 done;
+// 1 failed or refused (an operation this process will not run, one switched
+// off, a paused lane, unreachable media, no given id on disk); 2 usage (an
+// unknown operation or channel, a stray flag); 130 cancelled.
+
+import { existsSync } from "node:fs";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import type { OperationDescriptor } from "../lib/operations";
+
+export type RunOut = { log: (s: string) => void; error: (s: string) => void; write: (s: string) => void };
+
+const defaultOut: RunOut = {
+ log: (s) => console.log(s),
+ error: (s) => console.error(s),
+ write: (s) => {
+ process.stdout.write(s);
+ },
+};
+
+// Why an operation cannot run outside the editor, as one sentence — or null
+// when it can. Read off the descriptor, so a new external entry gets an honest
+// refusal without an edit here.
+export function offlineRefusal(op: OperationDescriptor): string | null {
+ if (op.dispatch !== "external") return null;
+ if (op.lane.contendsFor === "network") {
+ return (
+ `${op.label} runs on the per-platform download queue inside the editor, which paces ` +
+ `every request to one source; a second process would fetch beside it and defeat that ` +
+ `pacing. Run it from the editor, or through \`pnpm ops\`.`
+ );
+ }
+ if (op.runner === "transcription") {
+ return (
+ `${op.label} is dispatched across the editor's worker pool (leases, tiers, drain), ` +
+ `which does not exist outside the editor. Run it from the editor.`
+ );
+ }
+ return `${op.label} is dispatched by the editor, not through the operation registry.`;
+}
+
+export type RunOperationArgs = {
+ operation: string;
+ channel: string;
+ ids: string[];
+ // Digest only: which engine lane. Absent = the configured one.
+ lane?: "local" | "remote";
+};
+
+export async function runOperation(
+ args: RunOperationArgs,
+ deps: { paths?: Paths; out?: RunOut; signal?: AbortSignal } = {},
+): Promise<number> {
+ const out = deps.out ?? defaultOut;
+ const { operationCatalog, getOperation, DIGEST_OPERATION_ID } = await import("../lib/operations");
+ const catalog = operationCatalog();
+ const op = catalog.find((o) => o.id === args.operation);
+ if (!op) {
+ const runs = catalog.filter((o) => !offlineRefusal(o)).map((o) => o.id);
+ const refused = catalog.filter((o) => offlineRefusal(o)).map((o) => o.id);
+ out.error(
+ `run: no operation "${args.operation}". Runs here: ${runs.join(", ")}. ` +
+ `Catalogued but run only by the editor: ${refused.join(", ")}.`,
+ );
+ return 2;
+ }
+ const refusal = offlineRefusal(op);
+ if (refusal) {
+ out.error(`run: ${refusal}`);
+ return 1;
+ }
+ if (args.lane && op.id !== DIGEST_OPERATION_ID) {
+ out.error(`run: --lane is the digest engine lane; ${op.label} has none`);
+ return 2;
+ }
+
+ const paths = deps.paths ?? (await import("../lib/paths")).getPaths();
+ const channelDir = path.join(paths.channelsDir, args.channel);
+ if (!existsSync(path.join(channelDir, "config.json"))) {
+ out.error(`run: no channel "${args.channel}" (no ${path.join(channelDir, "config.json")})`);
+ return 2;
+ }
+
+ const { getSettings } = await import("../lib/settings");
+ const settings = getSettings();
+ const registered = getOperation(op.id);
+ if (!registered || !registered.enabled(settings)) {
+ out.error(
+ `run: ${op.label} is switched off in settings.json` +
+ (op.settingsBlock ? ` (its \`${op.settingsBlock}\` block)` : "") +
+ ` — the editor's lane would not run it either.`,
+ );
+ return 1;
+ }
+ const { pauseLaneFor, isGateHeld } = await import("../lib/pauseGates");
+ const gate = pauseLaneFor(op.id);
+ if (gate && isGateHeld(settings, gate)) {
+ out.error(
+ `run: the ${gate} lane is paused (autoQueue.${gate}.held) — the editor's run would hold ` +
+ `until it is resumed. Resume it on /operations/${op.id} first.`,
+ );
+ return 1;
+ }
+
+ const ids = [...new Set(args.ids.map((s) => s.trim()).filter(Boolean))];
+ if (ids.length > 0) {
+ const absent = ids.filter((id) => !existsSync(path.join(channelDir, "data", id)));
+ if (absent.length === ids.length) {
+ out.error(
+ `run: none of the ${ids.length} id(s) has a data/<id>/ on disk (${absent.join(", ")}) — nothing to run`,
+ );
+ return 1;
+ }
+ if (absent.length > 0) {
+ out.error(
+ `run: ${absent.length} of ${ids.length} id(s) have no data/<id>/ on disk and are skipped: ${absent.join(", ")}`,
+ );
+ }
+ }
+
+ const { runOperationChannelJob } = await import("../controller/operationJobs");
+ const result = await runOperationChannelJob({
+ paths,
+ channelSlug: args.channel,
+ operation: op.id,
+ ids: ids.length > 0 ? ids : undefined,
+ ...(args.lane ? { digest: { lane: args.lane } } : {}),
+ });
+ if (!result.ok) {
+ out.error(`run: ${result.error}`);
+ return 1;
+ }
+ out.error(`run: ${op.label} over ${args.channel}${ids.length ? ` (${ids.length} id(s))` : ""} — job ${result.jobId}`);
+
+ const { getRegistry } = await import("../jobs/registry");
+ const cancel = () => {
+ getRegistry().cancel(result.jobId);
+ };
+ deps.signal?.addEventListener("abort", cancel, { once: true });
+ const reader = result.stream.getReader();
+ for (;;) {
+ const { value, done } = await reader.read();
+ if (done) break;
+ out.write(value);
+ }
+ const { status } = await result.done;
+ deps.signal?.removeEventListener("abort", cancel);
+
+ // The editor refreshes the channel's report after a job; a process that is
+ // about to exit has to ask for it now, or the debounce would never fire.
+ const { flushChannelSnapshotsNow } = await import("../jobs/snapshotScheduler");
+ await flushChannelSnapshotsNow();
+
+ if (status === "done") return 0;
+ if (status === "cancelled") return 130;
+ return 1;
+}
diff --git a/common/bin/sync-tick.ts b/common/bin/sync-tick.ts
@@ -7,7 +7,7 @@
//
// Env:
// SYNC_TICK_URL full URL of the tick endpoint. Default targets the editor's
-// real dev/start port (3001):
+// real dev/start port (EDITOR_PORT in lib/ports.mjs, 3001):
// http://127.0.0.1:3001/api/scheduler/tick
// SYNC_TICK_TOKEN optional bearer token. When set, it must match the token in
// the editor server's environment or the request is rejected.
@@ -16,8 +16,10 @@
// the output); a normal tick prints a one-line summary for the cron log.
import { runIfEntryPoint } from "./_cli";
+import { PORT_BASES } from "../lib/ports.mjs";
-const DEFAULT_URL = "http://127.0.0.1:3001/api/scheduler/tick";
+// The primary's editor, never a worktree's: cron runs this from the primary.
+const DEFAULT_URL = `http://127.0.0.1:${PORT_BASES.EDITOR_PORT}/api/scheduler/tick`;
// The exit code: 1 on an HTTP error (cron mails it), 0 otherwise. A network
// failure throws.
diff --git a/common/bin/transform.ts b/common/bin/transform.ts
@@ -1,25 +0,0 @@
-#!/usr/bin/env tsx
-import { getPaths } from "../lib/paths";
-import { runWhisperBatch } from "../controller/whisperBatch";
-import { parseFlags } from "./_parseFlags";
-
-const flags = parseFlags(process.argv.slice(2));
-const channelSlug = flags.channel;
-if (!channelSlug) {
- console.error("Usage: transform.ts --channel <slug>");
- process.exit(2);
-}
-
-runWhisperBatch({
- channelSlug,
- paths: getPaths(),
-})
- .then((result) => {
- console.log(
- `Done: ${result.succeeded} succeeded, ${result.failed} failed, ${result.skipped} skipped, ${result.attempted} attempted.`,
- );
- })
- .catch((err) => {
- console.error(err);
- process.exit(1);
- });
diff --git a/common/controller/operationJobs.ts b/common/controller/operationJobs.ts
@@ -169,6 +169,9 @@ export type DigestChannelJobOptions = {
order?: string;
limitCount?: number;
force?: boolean;
+ // Only these videos (intersected with disk by the batch), as on the backfill
+ // job. In `spec.params` so a replay stays scoped. Absent = the channel.
+ ids?: string[];
background?: boolean;
onStarted?: (jobId: string) => void;
onDone?: () => void;
@@ -220,6 +223,7 @@ export async function runDigestChannelJob(
order: opts.order,
limitCount: opts.limitCount,
force: opts.force,
+ ids: opts.ids,
},
},
fn: async (onLog, signal, setProgress, ctx) => {
@@ -229,6 +233,7 @@ export async function runDigestChannelJob(
const missing = (
await countOperationWork("digest", paths, channelSlug, {
digestLane: lane,
+ ids: opts.ids,
})
).reachable;
// The bar measures THIS RUN, from zero, and the batch reports its own
@@ -241,6 +246,7 @@ export async function runDigestChannelJob(
channelSlug,
paths,
digestLane: lane,
+ ids: opts.ids,
digestOrder: isDigestOrder(opts.order) ? opts.order : undefined,
// The local lane takes everything up to the long-tail cutoff; the
// metered lane exists for the tail above it. Passing no window (the
@@ -291,6 +297,8 @@ export type OperationChannelJobOptions = {
channelSlug: string;
// A catalog operation id — "digest", "diarization", "attribution-text", …
operation: string;
+ // Only these videos; absent = the whole channel.
+ ids?: string[];
queueKey?: string;
background?: boolean;
onStarted?: (jobId: string) => void;
@@ -330,6 +338,7 @@ export function runOperationChannelJob(
order: opts.digest?.order,
limitCount: opts.digest?.limitCount,
force: opts.digest?.force,
+ ids: opts.ids,
queueKey: opts.queueKey,
background: opts.background,
onStarted: opts.onStarted,
@@ -340,6 +349,7 @@ export function runOperationChannelJob(
paths,
channelSlug,
kindIds: [operation],
+ ids: opts.ids,
// RESERVE BY THE LANE'S OWN KEY, never invent one. This is what keeps digest
// and diarization overlapping instead of taking turns, which backfill.spec
// pins. laneForOperation is live (it asks the kind's laneFor), so a caller
diff --git a/common/lib/envVars.test.ts b/common/lib/envVars.test.ts
@@ -0,0 +1,146 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { readdirSync, readFileSync, statSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { ENV_AUDIENCES, ENV_VARS, renderEnvironmentMarkdown } from "./envVars";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common test
+//
+// envVars.ts is the one declared list of environment variables, and
+// ENVIRONMENT.md is generated from it. These tests are what keep the list true:
+// the code is read as text, in both directions.
+
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+// Where the apps' code lives. umtool is out of scope on purpose (envVars.ts
+// says why); tests are skipped because a test SETS variables for itself.
+const CODE_ROOTS = ["common", "editor", "export", "homepage", "mcp/src", "scripts"];
+const SKIP_DIRS = new Set(["node_modules", ".next", "out", "public", "test-results", "blob-report", "test-transcripts"]);
+
+// The platform's own variables: read here, documented by node, Next, a shell.
+const PLATFORM = new Set(["CI", "NODE_ENV", "NEXT_RUNTIME", "LD_LIBRARY_PATH", "PATH", "HOME", "NODE_OPTIONS"]);
+
+function codeFiles(): string[] {
+ const out: string[] = [];
+ const walk = (dir: string) => {
+ for (const n of readdirSync(dir)) {
+ if (SKIP_DIRS.has(n) || n.startsWith(".")) continue;
+ const p = path.join(dir, n);
+ if (statSync(p).isDirectory()) walk(p);
+ else if (/\.(ts|tsx|mts|mjs|js)$/.test(n) && !/\.test\.(ts|mjs)$/.test(n)) out.push(p);
+ }
+ };
+ for (const r of CODE_ROOTS) walk(path.join(REPO, r));
+ return out;
+}
+
+// Comment lines are dropped first: prose that names `process.env.NAME` as an
+// example is not a read.
+function codeText(file: string): string {
+ return readFileSync(file, "utf8")
+ .split("\n")
+ .filter((l) => !/^\s*(\/\/|\*|\/\*)/.test(l))
+ .join("\n");
+}
+
+const READ_PATTERNS = [
+ /process\.env\??\.([A-Z][A-Z0-9_]+)\b/g,
+ /process\.env\[["']([A-Z][A-Z0-9_]+)["']\]/g,
+ // `env.X` on an env object handed in (a spawn's env, a testable `env =
+ // process.env` parameter).
+ /\b[eE]nv\??\.([A-Z][A-Z0-9_]{2,})\b/g,
+ // audioCheckedDownload.ts reads its test overrides through a helper.
+ /envIntOverride\(["']([A-Z][A-Z0-9_]+)["']\)/g,
+];
+
+function reads(): Map<string, Set<string>> {
+ const byName = new Map<string, Set<string>>();
+ for (const file of codeFiles()) {
+ const text = codeText(file);
+ for (const re of READ_PATTERNS) {
+ for (const m of text.matchAll(re)) {
+ const name = m[1];
+ if (PLATFORM.has(name)) continue;
+ if (!byName.has(name)) byName.set(name, new Set());
+ byName.get(name)!.add(path.relative(REPO, file));
+ }
+ }
+ }
+ return byName;
+}
+
+test("ENVIRONMENT.md is what envVars.ts generates", () => {
+ assert.equal(
+ readFileSync(path.join(REPO, "ENVIRONMENT.md"), "utf8"),
+ renderEnvironmentMarkdown(),
+ "ENVIRONMENT.md is stale: run `archilyzer docs env`",
+ );
+});
+
+test("every variable the code reads is declared", () => {
+ const declared = new Set(ENV_VARS.map((v) => v.name));
+ const missing = [...reads()]
+ .filter(([name]) => !declared.has(name))
+ .map(([name, files]) => `${name} (read by ${[...files].join(", ")})`);
+ assert.deepEqual(missing, [], "declare these in common/lib/envVars.ts");
+});
+
+// docker/'s files, the Dockerfiles and the compose files: where the
+// container's own ARCHILYZER_* set is read.
+function dockerTexts(): string[] {
+ return [
+ ...readdirSync(path.join(REPO, "docker"))
+ .filter((n) => statSync(path.join(REPO, "docker", n)).isFile())
+ .map((n) => readFileSync(path.join(REPO, "docker", n), "utf8")),
+ ...readdirSync(REPO)
+ .filter((n) => /^docker-compose.*\.yml$|^Dockerfile/.test(n))
+ .map((n) => readFileSync(path.join(REPO, n), "utf8")),
+ ];
+}
+
+test("every ARCHILYZER_* name in docker/, the Dockerfiles and the compose files is declared", () => {
+ const declared = new Set(ENV_VARS.map((v) => v.name));
+ const names = new Set(dockerTexts().flatMap((t) => [...t.matchAll(/\bARCHILYZER_[A-Z0-9_]+\b/g)].map((m) => m[0])));
+ assert.deepEqual([...names].filter((n) => !declared.has(n)).sort(), [], "declare these in common/lib/envVars.ts");
+});
+
+test("every declared variable is still named outside the list: the code, docker/, a Dockerfile, a compose file or a script", () => {
+ // envVars.ts itself is left out: every declared name is spelled there, so
+ // with it in the corpus this test could never fail.
+ const corpus = [
+ ...codeFiles()
+ .filter((f) => !f.endsWith(path.join("common", "lib", "envVars.ts")))
+ .map((f) => readFileSync(f, "utf8")),
+ ...dockerTexts(),
+ ...["", "editor", "export", "homepage"].map((d) =>
+ readFileSync(path.join(REPO, d, "package.json"), "utf8"),
+ ),
+ ].join("\n");
+ const stale = ENV_VARS.filter((v) => !new RegExp(`\\b${v.name}\\b`).test(corpus)).map((v) => v.name);
+ assert.deepEqual(stale, [], "nothing mentions these any more: delete their entries");
+});
+
+test("the paths audience is exactly what getPaths() reads", () => {
+ const pathsTs = codeText(path.join(REPO, "common/lib/paths.ts"));
+ const inPaths = new Set([...pathsTs.matchAll(/process\.env\.([A-Z][A-Z0-9_]+)/g)].map((m) => m[1]));
+ const declared = new Set(ENV_VARS.filter((v) => v.audience === "paths").map((v) => v.name));
+ assert.deepEqual([...declared].filter((n) => !inPaths.has(n)), [], "declared paths but not read by getPaths()");
+ assert.deepEqual([...inPaths].filter((n) => !declared.has(n)), [], "read by getPaths() but not declared as paths");
+});
+
+test("names are unique and every audience has a section", () => {
+ const names = ENV_VARS.map((v) => v.name);
+ assert.equal(new Set(names).size, names.length);
+ const audiences = new Set(ENV_AUDIENCES.map((a) => a.id));
+ for (const v of ENV_VARS) assert.ok(audiences.has(v.audience), v.name);
+ const md = renderEnvironmentMarkdown();
+ for (const v of ENV_VARS) assert.ok(md.includes(`| \`${v.name}\` |`), v.name);
+});
+
+test("the docker audience is the ARCHILYZER_ set", () => {
+ for (const v of ENV_VARS.filter((x) => x.audience === "docker")) {
+ assert.match(v.name, /^ARCHILYZER_/);
+ }
+});
diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts
@@ -0,0 +1,229 @@
+// EVERY ENVIRONMENT VARIABLE THE REPO READS — the one declared list.
+//
+// ENVIRONMENT.md is generated from this (common/bin/env-docs.ts, `--check` in
+// the common tests), and `envVars.test.ts` holds the list to the code in both
+// directions: every variable read by common/, editor/, export/, homepage/,
+// mcp/src and scripts/ (and every `ARCHILYZER_*` name in docker/, the
+// Dockerfiles and the compose files) is declared here, and every entry here is
+// still NAMED somewhere outside this file — the code, docker/, a Dockerfile, a
+// compose file or a package.json script. A variable added without an entry, or
+// whose last mention is deleted while its entry stays, fails the build. (The
+// second check is a mention, not a proven read: a name left only in a comment
+// passes it.) umtool's own knobs are NOT here — its song and report scripts read
+// dozens, documented in umtool/docs, and fold into the core in one-core Phase 5.
+//
+// THE AUDIENCES, which are the point of the table:
+// paths — the ONE override surface for where things live and which binary
+// runs: every one is read by getPaths() (lib/paths.ts) and nowhere
+// else. The test checks both directions.
+// runtime — the other knobs, secrets and tokens a running process reads.
+// port — generated from lib/ports.mjs, never listed twice.
+// internal — set BY the pipeline for a process it spawns. Listed so a reader
+// knows what it is; nobody sets it by hand.
+// docker — the container's ARCHILYZER_* set, read by docker/*.sh, the
+// compose files and Caddy — a different process from the apps,
+// documented in RUNNING_IN_DOCKER.md.
+// test — read only by a test harness, a fake binary or a test-mode branch.
+//
+// Pure data (no imports but the port table), so a doc generator and a doctor can
+// both read it without loading anything else.
+
+import { PORTS } from "./ports.mjs";
+
+export type EnvAudience = "paths" | "runtime" | "port" | "internal" | "docker" | "test";
+
+export type EnvVarDecl = {
+ name: string;
+ audience: EnvAudience;
+ // What it does, one or two sentences.
+ doc: string;
+ // What an unset variable means, in words (a value, "off", "—").
+ default: string;
+ // Where it is read — a file, or a short list of them.
+ readBy: string;
+};
+
+const paths = (name: string, def: string, doc: string): EnvVarDecl => ({
+ name,
+ audience: "paths",
+ doc,
+ default: def,
+ readBy: "common/lib/paths.ts (getPaths)",
+});
+
+const DECLARED: EnvVarDecl[] = [
+ // ── paths: getPaths() ──────────────────────────────────────────────────
+ paths("TRANSCRIPTS_DIR", "`<repo>/transcripts`", "The corpus: channels, sites, the LMDB index, job logs, the saved-video store."),
+ paths("SAVED_VIDEOS_DIR", "`<TRANSCRIPTS_DIR>/saved-videos`", "The persisted source-video store, when it should live on another disk."),
+ paths("SITES_DIR", "`<TRANSCRIPTS_DIR>/sites`", "Per-site config (`<id>/site.json`, every key in [SITE.md](SITE.md)) and the homepage's `_homepage/`."),
+ paths("SETTINGS_FILE", "`<repo>/settings.json`", "The settings file (every key in [SETTINGS.md](SETTINGS.md))."),
+ paths("EXPORT_PUBLIC_DIR", "`<repo>/export/public`", "The dir the export site serves at `/`, composed one site at a time."),
+ paths("EXPORT_INDEX_DIR", "`.export-index` beside `EXPORT_PUBLIC_DIR`", "The build's staging area (not served): the shared index and per-site aggregates."),
+ paths("EXPORT_BUILDS_DIR", "`.export-builds` beside `EXPORT_PUBLIC_DIR`", "Per-site `out/` bundles from a docker-mode build."),
+ paths("EDITOR_CHANGELOG_FILE", "`<repo>/editor/CHANGELOG.md`", "The editor changelog the release cutter reads and rewrites. The e2e server points it at a gitignored copy."),
+ paths("EXPORT_CHANGELOG_FILE", "`<repo>/export/CHANGELOG.md`", "The export changelog, likewise."),
+ paths("CHARTS_CONFIG_FILE", "`<repo>/chart-templates.json`", "The legacy chart-templates file, read only by a migration."),
+ paths("SEARCH_ALIASES_FILE", "`<TRANSCRIPTS_DIR>/search-aliases.json`", "The corpus-wide search-alias dictionary."),
+ paths("CURATED_TAGS_FILE", "`<TRANSCRIPTS_DIR>/tags.json`", "Curated per-video tags. Written only through `applyTagAssignments`."),
+ paths("YTDLP_BIN", "`yt-dlp` on PATH", "The downloader. Every fetch goes through it."),
+ paths("FFMPEG_BIN", "`ffmpeg` on PATH", "Audio extraction for transcription and diarization."),
+ paths("FFPROBE_BIN", "`ffprobe` on PATH", "Duration checks (the short-audio guard, windowing)."),
+ paths("WHISPER_BIN", "`whisper-cli` on PATH", "whisper.cpp, one of the three transcription engines (with chough and parakeet.cpp)."),
+ paths("WHISPER_MODEL", "`~/whispercpp/whisper.cpp/models/ggml-base.en.bin`", "whisper.cpp's model, when a worker names none."),
+ paths("PARAKEET_STITCH_BIN", "`<repo>/scripts/parakeet-stitch.mjs`", "The parakeet.cpp engine's wrapper (overlapping windows, stitched)."),
+ paths("PARAKEET_CLI", "`parakeet-cli` on PATH", "The parakeet.cpp binary the wrapper drives (the wrapper reads it too)."),
+ paths("PARAKEET_MODEL", "none", "parakeet.cpp's `.gguf`, when a worker names none (the wrapper reads it too)."),
+ paths("DIARIZE_BIN", "`<repo>/scripts/diarize.mjs`", "The speaker-diarization wrapper. The e2e suite swaps in a fake here."),
+ paths("RSYNC_BIN", "`rsync` on PATH", "Mirrors the saved-video store to a backup destination."),
+ paths("FINDMNT_BIN", "`findmnt` on PATH", "The read-only volume-identity probe behind storage locations. Optional."),
+ paths("UDISKSCTL_BIN", "`udisksctl` on PATH", "Mounts an attached volume from `/storage`. Optional."),
+ paths("GALLERY_DL_BIN", "`gallery-dl` on PATH", "The X/Twitter post fetcher, for social channels."),
+ paths("OLLAMA_URL", "`http://127.0.0.1:11434`", "The local ollama server, the local digest and attribution engine."),
+ paths("CLAUDE_BIN", "`claude` on PATH", "The `claude` CLI, driving the opt-in metered digest lane."),
+
+ // ── runtime ────────────────────────────────────────────────────────────
+ { name: "WORKER_TOKEN", audience: "runtime", default: "unset (both surfaces off)", readBy: "common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts", doc: "Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`, the MCP's `fetch_clip`). Set the same value on both ends." },
+ { name: "SYNC_HEARTBEAT_SECONDS", audience: "runtime", default: "`settings.syncScheduler.heartbeatSeconds`", readBy: "editor/app/scheduler/heartbeat.ts", doc: "Overrides the editor's in-process sync heartbeat. `0` = no internal timer (tick from cron instead)." },
+ { name: "SYNC_TICK_URL", audience: "runtime", default: "`http://127.0.0.1:3001/api/scheduler/tick`", readBy: "common/bin/sync-tick.ts", doc: "Where `archilyzer sync tick` (cron's heartbeat) posts." },
+ { name: "SYNC_TICK_TOKEN", audience: "runtime", default: "unset (no auth)", readBy: "common/bin/sync-tick.ts, editor/app/scheduler/auth.ts", doc: "Bearer token for the tick endpoint; set on both the editor and the cron job." },
+ { name: "R2_ACCESS_KEY_ID", audience: "runtime", default: "—", readBy: "common/publish/build.ts", doc: "R2 S3 credentials for uploading oversize archives at deploy time (with `R2_SECRET_ACCESS_KEY` and `CLOUDFLARE_ACCOUNT_ID`). See [PUBLISH.md](PUBLISH.md)." },
+ { name: "R2_SECRET_ACCESS_KEY", audience: "runtime", default: "—", readBy: "common/publish/build.ts", doc: "See `R2_ACCESS_KEY_ID`." },
+ { name: "CLOUDFLARE_ACCOUNT_ID", audience: "runtime", default: "—", readBy: "common/publish/build.ts", doc: "The account the R2 endpoint belongs to. wrangler reads its own credentials." },
+ { name: "DOCKER_BIN", audience: "runtime", default: "`docker`", readBy: "common/publish/build.ts", doc: "The container engine for docker-mode builds (e.g. `podman`)." },
+ { name: "DOCKER_BUILD_MEMORY", audience: "runtime", default: "no cap", readBy: "common/publish/build.ts", doc: "Per-container memory cap for a docker-mode build (`--memory`)." },
+ { name: "DOCKER_BUILD_CPUS", audience: "runtime", default: "no cap", readBy: "common/publish/build.ts", doc: "Per-container CPU cap for a docker-mode build (`--cpus`)." },
+ { name: "ARCHIVE_CHANNEL_CONCURRENCY", audience: "runtime", default: "`4`", readBy: "common/bin/build-archives.ts", doc: "How many channels' archive zips `build archives` builds at once." },
+ { name: "MAX_ARCHIVE_BYTES", audience: "runtime", default: "the Cloudflare-safe cap", readBy: "common/bin/compose-site.ts", doc: "The served-file size cap for archives, in bytes; `0` = no cap. A site's own `archiveMaxBytes` wins." },
+ { name: "CHOUGH_BIN", audience: "runtime", default: "`chough` on PATH", readBy: "common/lib/transcriptionApps.ts", doc: "The chough transcription engine, when a worker names no binary." },
+ { name: "CHOUGH_MODEL", audience: "runtime", default: "chough's own", readBy: "chough (set by common/lib/transcriptionApps.ts)", doc: "Passed to chough from a worker's model field; chough auto-downloads one when unset." },
+ { name: "CHOUGH_URL", audience: "runtime", default: "local", readBy: "chough (set by common/lib/transcriptionApps.ts)", doc: "Passed to chough from a worker's remote-server field." },
+ { name: "OLLAMA_DIGEST_MODEL", audience: "runtime", default: "`qwen2.5:7b`", readBy: "common/lib/digestApps.ts", doc: "The ollama model the local digest lane asks for when settings name none." },
+ { name: "CLAUDE_DIGEST_MODEL", audience: "runtime", default: "the CLI's default", readBy: "common/lib/digestApps.ts", doc: "The model the metered digest lane asks `claude` for when settings name none." },
+ { name: "NITTER_INSTANCES", audience: "runtime", default: "a built-in list", readBy: "common/social/xNitterFetcher.ts", doc: "Comma-separated Nitter instances for the X fallback fetcher, in order of preference." },
+ { name: "UMTOOL_URL", audience: "runtime", default: "unset (no link)", readBy: "editor/app/channels/[slug]/videos/[id]/page.tsx", doc: "umtool's front door; when set, the video page links to it." },
+ { name: "TRANSCRIPT_SITE_URL", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: one published archive to read over HTTP." },
+ { name: "TRANSCRIPT_HUB_URL", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: a hub, federating every archive it lists." },
+ { name: "TRANSCRIPT_LOCAL_DIR", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: a composed public dir on disk." },
+ { name: "TRANSCRIPT_PLATFORM_LINKS", audience: "runtime", default: "off", readBy: "common/lib/archive/reader-fs.ts", doc: "`1` cites platform watch pages instead of the archive's own pages." },
+ { name: "MCP_IO_STATS", audience: "runtime", default: "off", readBy: "common/lib/archive/io-stats.ts", doc: "`1` turns on per-call I/O accounting, for `mcp/bench`." },
+ { name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool", doc: "Which editor `pnpm ops` and the MCP's `fetch_clip` talk to." },
+ { name: "ARCHILYZER_AGENT", audience: "runtime", default: "`cli`", readBy: "scripts/archilyzer-ops.mjs", doc: "Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`." },
+ { name: "DIARIZE_ENGINE_KIND", audience: "runtime", default: "`sherpa-onnx`", readBy: "scripts/diarize.mjs", doc: "The diarization engine: `sherpa-onnx` or `sortformer`." },
+ { name: "DIARIZE_ENGINE_CMD", audience: "runtime", default: "the bundled sherpa script", readBy: "scripts/diarize.mjs", doc: "The engine command the wrapper runs." },
+ { name: "DIARIZE_PYTHON", audience: "runtime", default: "`python3`", readBy: "scripts/diarize.mjs", doc: "The python for the default engine." },
+ { name: "DIARIZE_SEG_MODEL", audience: "runtime", default: "— (required)", readBy: "scripts/diarize.mjs", doc: "Segmentation model. The editor passes the settings' value as a flag." },
+ { name: "DIARIZE_EMB_MODEL", audience: "runtime", default: "— (required)", readBy: "scripts/diarize.mjs", doc: "Speaker-embedding model. The editor passes the settings' value as a flag." },
+ { name: "DIARIZE_THRESHOLD", audience: "runtime", default: "`0.5`", readBy: "scripts/diarize.mjs", doc: "Clustering threshold." },
+ { name: "DIARIZE_THREADS", audience: "runtime", default: "`4`", readBy: "scripts/diarize.mjs", doc: "Engine threads." },
+ { name: "DIARIZE_WINDOW_MINUTES", audience: "runtime", default: "`45`", readBy: "scripts/diarize.mjs", doc: "Window length for long files; `0` never windows." },
+ { name: "DIARIZE_WINDOW_AFTER_MINUTES", audience: "runtime", default: "`90`", readBy: "scripts/diarize.mjs", doc: "Only files longer than this are windowed." },
+ { name: "SORTFORMER_BIN", audience: "runtime", default: "— (required for sortformer)", readBy: "scripts/diarize.mjs, scripts/diarize-sortformer.mjs", doc: "The sortformer engine binary." },
+ { name: "SORTFORMER_MODEL", audience: "runtime", default: "— (required for sortformer)", readBy: "scripts/diarize.mjs, scripts/diarize-sortformer.mjs", doc: "The sortformer `.gguf`." },
+ { name: "PARAKEET_SEGMENT_SEC", audience: "runtime", default: "`480`", readBy: "scripts/parakeet-stitch.mjs", doc: "parakeet window length, seconds (a worker's chunk size wins)." },
+ { name: "PARAKEET_OVERLAP_SEC", audience: "runtime", default: "`6`", readBy: "scripts/parakeet-stitch.mjs", doc: "parakeet window overlap, seconds." },
+ { name: "PARAKEET_DECODER", audience: "runtime", default: "parakeet-cli's", readBy: "scripts/parakeet-stitch.mjs", doc: "`ctc` or `tdt`, passed through to parakeet-cli." },
+ { name: "PARAKEET_LANG", audience: "runtime", default: "parakeet-cli's", readBy: "scripts/parakeet-stitch.mjs", doc: "A locale, passed through to parakeet-cli." },
+ { name: "PARAKEET_DEVICE", audience: "runtime", default: "parakeet-cli's", readBy: "scripts/parakeet-stitch.mjs", doc: "Compute device (`cpu`, `CUDA0`, `Vulkan1`, …), exported to parakeet-cli." },
+
+ // ── internal: the pipeline sets these for a process it spawns ──────────
+ { name: "SITE_ID", audience: "internal", default: "—", readBy: "common/bin/compose-site.ts, export/app/lib/site.ts", doc: "Which site a compose or an export build is for. `archilyzer build site <id>` sets it; `compose site` and `build site` fall back to it when no id is given." },
+ { name: "INSTANCE_MODE", audience: "internal", default: "a site", readBy: "export/app/lib/mode.ts, common/lib/archive/contract.ts", doc: "`hub` makes the export build the hub. Set by `archilyzer build hub`." },
+ { name: "BUILD_ARCHIVES", audience: "internal", default: "on", readBy: "common/bin/compose-site.ts, common/bin/build-archives.ts", doc: "`0` skips archive-zip generation for one build (`--skip-archives`)." },
+ { name: "ARCHIVES_READONLY", audience: "internal", default: "off", readBy: "common/bin/compose-site.ts", doc: "`1` inside a docker-mode build container: materialize archives, never write the shared cache." },
+ { name: "HOMEPAGE_PUBLIC_DIR", audience: "internal", default: "`<repo>/homepage/public`", readBy: "common/bin/compose-homepage.ts", doc: "Where `compose homepage` writes." },
+ { name: "HOMEPAGE_SUMMARY_FILE", audience: "internal", default: "`homepage/public/homepage-summary.json`", readBy: "homepage/app/lib/summary.ts", doc: "A dev-only summary file for the homepage; ignored by a production build. The homepage e2e sets it." },
+
+ // ── docker: the container's set ────────────────────────────────────────
+ { name: "ARCHILYZER_TRANSCRIBER", audience: "docker", default: "baked per image target (`whisper-cpp` in `runtime`)", readBy: "docker/entrypoint.sh", doc: "`whisper-cpp` or `parakeet`: which worker the first boot seeds and which model it fetches." },
+ { name: "ARCHILYZER_FETCH_MODEL", audience: "docker", default: "per transcriber", readBy: "docker/entrypoint.sh", doc: "Which model the first boot downloads; `none` skips it." },
+ { name: "ARCHILYZER_MODELS_DIR", audience: "docker", default: "`/data/models`", readBy: "docker/entrypoint.sh", doc: "Where models live in the container." },
+ { name: "ARCHILYZER_BUILDS_DIR", audience: "docker", default: "`/data/builds`", readBy: "docker/entrypoint.sh", doc: "Where the container keeps built sites." },
+ { name: "ARCHILYZER_SITE_OUT", audience: "docker", default: "`/data/builds/site`", readBy: "docker/entrypoint.sh, docker/publish-site.sh", doc: "The built export site the `site` service serves." },
+ { name: "ARCHILYZER_IDLE_BOOT", audience: "docker", default: "off", readBy: "common/lib/idleBoot.ts (the editor)", doc: "`1` boots the editor without arming the heartbeat or any auto-queue runner." },
+ { name: "ARCHILYZER_AUTH_MODE", audience: "docker", default: "`basic`", readBy: "docker/guard-exposure.sh, docker/caddy-start.sh", doc: "`basic`, `forward` or `none` — the only escape hatch from the exposure guard." },
+ { name: "ARCHILYZER_AUTH_USER", audience: "docker", default: "`archilyzer`", readBy: "docker/Caddyfile", doc: "Basic-auth user." },
+ { name: "ARCHILYZER_AUTH_HASH", audience: "docker", default: "—", readBy: "docker/Caddyfile, docker/guard-exposure.sh", doc: "Basic-auth bcrypt hash (`caddy hash-password`)." },
+ { name: "ARCHILYZER_AUTH_IMPORT", audience: "docker", default: "derived from the mode", readBy: "docker/Caddyfile", doc: "Set by docker/caddy-start.sh from the mode: which auth snippet the private sites import." },
+ { name: "ARCHILYZER_FORWARD_AUTH_UPSTREAM", audience: "docker", default: "—", readBy: "docker/Caddyfile", doc: "Forward-auth server (Authelia, tinyauth, …), `host:port`." },
+ { name: "ARCHILYZER_FORWARD_AUTH_URI", audience: "docker", default: "`/api/auth/caddy`", readBy: "docker/Caddyfile", doc: "The forward-auth server's verify path." },
+ { name: "ARCHILYZER_TAG", audience: "docker", default: "`local`", readBy: "docker-compose*.yml", doc: "The image tag the compose files build and run." },
+
+ // ── test: harnesses, fakes and test-mode branches ──────────────────────
+ { name: "EDITOR_TEST_ROUTES", audience: "test", default: "off", readBy: "editor/app/api/test/_guard.ts, editor/instrumentation.ts", doc: "`1` opens the editor's `/api/test/*` routes. The e2e server sets it." },
+ { name: "E2E_MODE", audience: "test", default: "dev", readBy: "editor/playwright.config.ts", doc: "`start` runs the editor suite against `next start` instead of `next dev`." },
+ { name: "E2E_QUEUE", audience: "test", default: "on", readBy: "scripts/queue-lock.mjs", doc: "`0` skips the machine-global e2e queue (the port check still runs)." },
+ { name: "E2E_PORT_CHECK", audience: "test", default: "on", readBy: "scripts/queue-lock.mjs", doc: "`0` skips the pre-run check that the suite's ports are free." },
+ { name: "E2E_QUEUE_TIMEOUT", audience: "test", default: "wait forever", readBy: "scripts/queue-lock.mjs", doc: "Seconds to wait for the queue before giving up." },
+ { name: "E2E_PORT_GRACE_MS", audience: "test", default: "`3000`", readBy: "scripts/queue-lock.mjs", doc: "How long the port check waits for a just-freed port." },
+ { name: "E2E_QUEUE_LOCK_FILE", audience: "test", default: "one per machine", readBy: "scripts/queue-lock.mjs", doc: "The queue's lock file; the queue's own tests point it elsewhere." },
+ { name: "QUEUE_LOCK_HELD", audience: "test", default: "—", readBy: "scripts/queue-lock.mjs", doc: "Set by the queue for the command it runs, so a nested wrapper passes through." },
+ { name: "PLAYWRIGHT_BASE_URL", audience: "test", default: "`http://localhost:<PORT>`", readBy: "editor/playwright.config.ts, editor/e2e/baseUrl.ts", doc: "The editor test server's URL; the worktree injector sets it." },
+ { name: "AUDIO_CHECK_INTERVAL_MS_OVERRIDE", audience: "test", default: "the real cadence", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Shrinks the mid-download audio check so the e2e suite sees it fire." },
+ { name: "AUDIO_CHECK_SIZE_GATE_OVERRIDE", audience: "test", default: "the real gate", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Likewise, the size gate." },
+ { name: "AUDIO_CHECK_INTERVAL_FLOOR_MS_OVERRIDE", audience: "test", default: "the real floor", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Likewise, the interval floor." },
+ { name: "AUDIO_CHECK_RECOVER_STEP_MS_OVERRIDE", audience: "test", default: "the real step", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Likewise, the recovery step." },
+ { name: "AUDIO_CHECK_RECOVER_AFTER_OVERRIDE", audience: "test", default: "the real count", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "Likewise, the recovery count." },
+ { name: "AUDIO_CHECK_DEBUG_PAUSE_MS", audience: "test", default: "off", readBy: "common/ytdlp/audioCheckedDownload.ts", doc: "A debugging pause inside the audio check." },
+ { name: "FAKE_YTDLP_AUDIO_CHECK_MODE", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: which audio-check scenario to act out." },
+ { name: "FAKE_YTDLP_CHUNK_DELAY_MS", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: delay between written chunks." },
+ { name: "FAKE_YTDLP_CORRUPT_AFTER_CHUNK", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: start corrupting after this chunk." },
+ { name: "FAKE_YTDLP_CORRUPT_RUNS", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: how many runs corrupt." },
+ { name: "FAKE_YTDLP_DETERMINISTIC_CORRUPT", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: corrupt deterministically." },
+ { name: "FAKE_YTDLP_RECOVER_ON_RESUME", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: a resumed run recovers." },
+ { name: "FAKE_YTDLP_TOTAL_CHUNKS", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-ytdlp.mjs", doc: "Fake yt-dlp: how many chunks a download has." },
+ { name: "FAKE_GALLERY_DL_AUTH_FAIL", audience: "test", default: "—", readBy: "editor/e2e/fixtures/bin/fake-gallery-dl.mjs", doc: "Fake gallery-dl: fail as an auth error." },
+ { name: "FIXTURE_MAX_LIFETIME_MS", audience: "test", default: "the watchdog's", readBy: "editor/e2e/fixtures/bin/_watchdog.mjs", doc: "How long a fake binary may live before its watchdog kills it." },
+ { name: "OLLAMA_STUB_MODEL", audience: "test", default: "`qwen2.5:7b`", readBy: "editor/e2e/fixtures/ollama-stub.mjs", doc: "The model the ollama stub claims to serve." },
+ { name: "RACK_SHOTS", audience: "test", default: "off (spec skipped)", readBy: "editor/e2e/channels-rack-audit.spec.ts", doc: "Runs the `/channels` rack screenshot audit." },
+ { name: "TWO_ORIGIN_REBUILD", audience: "test", default: "off", readBy: "export/e2e-2origin/globalSetup.ts", doc: "`1` rebuilds the two-origin suite's cached hub bundle." },
+ { name: "IMAGE", audience: "test", default: "`yt-dlp-transcript-browser-e2e`", readBy: "scripts/run-sharded-e2e.mjs", doc: "The sharded e2e run's image tag." },
+ { name: "SKIP_BUILD", audience: "test", default: "off", readBy: "scripts/run-sharded-e2e.mjs", doc: "`1` reuses the sharded e2e image instead of rebuilding it (`--no-build`)." },
+];
+
+// The port rows come from the port table, so a port is declared once.
+const PORT_ROWS: EnvVarDecl[] = Object.entries(PORTS).map(([name, decl]) => ({
+ name,
+ audience: "port",
+ doc: `${decl.what[0].toUpperCase()}${decl.what.slice(1)}. A worktree adds its offset (\`pnpm wt list\`).`,
+ default: `\`${decl.base}\``,
+ readBy: "common/lib/ports.mjs",
+}));
+
+export const ENV_VARS: readonly EnvVarDecl[] = [...DECLARED, ...PORT_ROWS];
+
+export const ENV_AUDIENCES: ReadonlyArray<{ id: EnvAudience; title: string; intro: string }> = [
+ { id: "paths", title: "Paths and binaries", intro: "The one override surface for where things live and which binary runs. Every one is read by `getPaths()` (`common/lib/paths.ts`) and nowhere else; nothing hardcodes a location." },
+ { id: "runtime", title: "Runtime", intro: "Tokens, credentials and knobs a running process reads. Most configuration is not here but in `settings.json` ([SETTINGS.md](SETTINGS.md))." },
+ { id: "port", title: "Ports", intro: "Every local server's default port, from `common/lib/ports.mjs`. The primary checkout uses these; worktree N adds N × 100 (`pnpm wt list`)." },
+ { id: "internal", title: "Set by the pipeline", intro: "The publish pipeline sets these for a process it spawns. Listed so a reader knows what they are; nobody sets them by hand." },
+ { id: "docker", title: "Docker", intro: "The container's own set, read by `docker/*.sh`, the compose files and Caddy — not by the apps' code (except `ARCHILYZER_IDLE_BOOT`). See [RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md)." },
+ { id: "test", title: "Tests only", intro: "Read only by a test harness, a fake binary or a test-mode branch. Never set one on a real instance." },
+];
+
+export function envVar(name: string): EnvVarDecl | undefined {
+ return ENV_VARS.find((v) => v.name === name);
+}
+
+// ENVIRONMENT.md — one table per audience.
+export function renderEnvironmentMarkdown(): string {
+ const cell = (s: string) => s.replace(/\|/g, "\\|");
+ const out: string[] = [
+ "# Environment variables",
+ "",
+ "<!-- GENERATED by common/bin/env-docs.ts from common/lib/envVars.ts — do not edit by hand. -->",
+ "",
+ "Every environment variable the repo's code reads, by who it is for. The list is code (`common/lib/envVars.ts`), and a test fails when the code reads a variable the list does not declare, or the list declares one that nothing outside the list names any more. umtool's own knobs are documented in [umtool/docs](umtool/docs/README.md).",
+ "",
+ "Regenerate this file with `pnpm archilyzer docs env`. `pnpm archilyzer doctor` prints which of the paths overrides are set on this machine.",
+ "",
+ ];
+ for (const a of ENV_AUDIENCES) {
+ const rows = ENV_VARS.filter((v) => v.audience === a.id);
+ out.push(`## ${a.title}`, "", a.intro, "", "| Variable | Default | What it does | Read by |", "|---|---|---|---|");
+ for (const v of rows) {
+ out.push(`| \`${v.name}\` | ${cell(v.default)} | ${cell(v.doc)} | ${cell(v.readBy)} |`);
+ }
+ out.push("");
+ }
+ return out.join("\n");
+}
diff --git a/common/lib/ports.mjs b/common/lib/ports.mjs
@@ -0,0 +1,96 @@
+// THE PORT DEFAULTS — the one copy.
+//
+// Every local server this repo starts has a default port, and a worktree adds
+// `index * OFFSET_STEP` to all of them (scripts/worktree.mjs, WORKTREES.md), so
+// the primary checkout keeps the numbers below and worktree #3 gets 33xx.
+//
+// Plain JS with JSDoc, like ytdlp/platformArgs.mjs, because the first reader is
+// scripts/worktree.mjs, which runs under bare `node`. TS callers (the CLI's
+// doctor, the playwright configs) import it as it is.
+//
+// WHO READS THE NUMBER, AND WHO CANNOT. `scripts/worktree.mjs` (the injector)
+// and `archilyzer doctor` import this table. Two kinds of place cannot, and
+// both are held to it by `ports.test.ts`, which reads them as text and fails
+// when one disagrees:
+// - a package.json script's shell default, `next dev --port ${EDITOR_PORT:-3001}`.
+// A script line is a shell command; it has no import. It keeps the default so
+// `pnpm --filter editor start` works with no wrapper at all, which is what a
+// primary checkout's operator types.
+// - the `--ports NAME:N` list a package's `e2e` script hands queue-lock.mjs.
+// Same reason, and the machine-global queue's argument syntax is shared with
+// checkouts on older code, so it does not change shape here.
+// Everything else — the playwright configs and the e2e helpers — reads it from
+// here or from the env the injector set.
+//
+// NOT HERE: the container's ports (docker-compose.yml, docker/entrypoint.sh,
+// docker/Caddyfile). They are the ports inside a container, where there is one
+// checkout and no worktree offset, and Caddy is the only thing that publishes.
+
+/**
+ * @typedef {{ base: number, what: string }} PortDecl
+ */
+
+/** @type {Readonly<Record<string, PortDecl>>} */
+export const PORTS = Object.freeze({
+ EDITOR_PORT: { base: 3001, what: "editor real dev/start (`pnpm dev:editor`)" },
+ PORT: { base: 3011, what: "editor test server + Playwright editor baseURL" },
+ EXPORT_PORT: { base: 3010, what: "export server launched by the editor e2e" },
+ EXPORT_DEV_PORT: { base: 3000, what: "export real dev (`pnpm dev:export`)" },
+ EXPORT_E2E_PORT: { base: 3020, what: "export's own Playwright suite" },
+ OLLAMA_STUB_PORT: { base: 11435, what: "digest-lane stub server in the editor e2e suite" },
+ HOMEPAGE_DEV_PORT: { base: 3030, what: "homepage real dev (`pnpm dev:homepage`)" },
+ HOMEPAGE_PORT: { base: 3031, what: "homepage static `serve out` (start:homepage)" },
+ HOMEPAGE_E2E_PORT: { base: 3040, what: "homepage's own Playwright suite" },
+ HUB_PORT: { base: 3041, what: "export's hub Playwright suite (e2e:hub)" },
+ UMTOOL_PORT: { base: 3050, what: "umtool real dev/start (`pnpm dev:umtool`)" },
+ UMTOOL_E2E_PORT: { base: 3051, what: "umtool's own Playwright suite" },
+ EDITOR_STUB_PORT: { base: 3052, what: "stub editor the umtool e2e suite fetches clips from" },
+ ORIGIN_B_PORT: { base: 4610, what: "export's two-origin suite: the member site (e2e:2origin)" },
+ HUB_A_PORT: { base: 4611, what: "export's two-origin suite: the hub (e2e:2origin)" },
+});
+
+/** Base ports (offset 0 == the primary checkout), name -> number. */
+/** @type {Readonly<Record<string, number>>} */
+export const PORT_BASES = Object.freeze(
+ Object.fromEntries(Object.entries(PORTS).map(([k, v]) => [k, v.base])),
+);
+
+/** A worktree's block is `index * OFFSET_STEP` above the bases. */
+export const OFFSET_STEP = 100;
+
+/**
+ * The port env map for a given offset, plus PLAYWRIGHT_BASE_URL (the editor
+ * test server's URL). Does not consult process.env.
+ * @param {number} offset
+ * @returns {Record<string, string>}
+ */
+export function portsForOffset(offset) {
+ /** @type {Record<string, string>} */
+ const env = {};
+ for (const [key, base] of Object.entries(PORT_BASES)) {
+ env[key] = String(base + offset);
+ }
+ env.PLAYWRIGHT_BASE_URL = `http://localhost:${PORT_BASES.PORT + offset}`;
+ return env;
+}
+
+/**
+ * The port a process should use: the env's value when it is set (the injector
+ * or the operator put it there), else the base. Throws for a name this table
+ * does not declare and for a value that is not a port, so a typo cannot
+ * quietly mean "undefined" or NaN.
+ * @param {string} name
+ * @param {Record<string, string | undefined>} [env]
+ * @returns {number}
+ */
+export function portFor(name, env = process.env) {
+ const decl = PORTS[name];
+ if (!decl) throw new Error(`ports.mjs: no port named ${name}`);
+ const raw = env[name];
+ if (raw != null && raw.trim() !== "") {
+ const n = Number(raw);
+ if (Number.isInteger(n) && n > 0 && n < 65536) return n;
+ throw new Error(`ports.mjs: ${name}=${raw} is not a port`);
+ }
+ return decl.base;
+}
diff --git a/common/lib/ports.test.ts b/common/lib/ports.test.ts
@@ -0,0 +1,133 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { readdirSync, readFileSync, statSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import {
+ OFFSET_STEP,
+ PORTS,
+ PORT_BASES,
+ portFor,
+ portsForOffset,
+} from "./ports.mjs";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common test
+//
+// ports.mjs is the one copy of the port defaults. The places that CANNOT import
+// it (a package.json script's `${NAME:-N}`, the `--ports NAME:N` list handed to
+// queue-lock) and the ones that have not been pointed at it yet still spell a
+// number; this reads them as text and fails the moment one disagrees with the
+// table, or names a port the table does not declare.
+
+const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..");
+const PACKAGES = ["", "common", "editor", "export", "homepage", "mcp", "umtool", "umtool/report-to-video"];
+
+function scripts(dir: string): Record<string, string> {
+ const file = path.join(ROOT, dir, "package.json");
+ return (JSON.parse(readFileSync(file, "utf8")).scripts ?? {}) as Record<string, string>;
+}
+
+// The playwright configs and every e2e helper: where a `process.env.X ?? N`
+// fallback can live. Walked, not listed, so a new helper is covered.
+function e2eSources(): string[] {
+ const out: string[] = [];
+ const walk = (dir: string) => {
+ let names: string[];
+ try {
+ names = readdirSync(dir);
+ } catch {
+ return;
+ }
+ for (const n of names) {
+ if (n === "node_modules" || n.startsWith(".")) continue;
+ const p = path.join(dir, n);
+ if (statSync(p).isDirectory()) walk(p);
+ else if (/\.(ts|mts|mjs|js)$/.test(n)) out.push(p);
+ }
+ };
+ for (const pkg of ["editor", "export", "homepage", "umtool"]) {
+ for (const n of readdirSync(path.join(ROOT, pkg))) {
+ if (/^playwright.*\.config\.ts$/.test(n)) out.push(path.join(ROOT, pkg, n));
+ }
+ for (const d of ["e2e", "e2e-2origin"]) walk(path.join(ROOT, pkg, d));
+ }
+ return out;
+}
+
+type Found = { where: string; name: string; port: number };
+
+function found(): Found[] {
+ const hits: Found[] = [];
+ for (const pkg of PACKAGES) {
+ for (const [script, line] of Object.entries(scripts(pkg))) {
+ const where = `${pkg || "."}/package.json "${script}"`;
+ for (const m of line.matchAll(/\$\{([A-Z0-9_]+):-(\d+)\}/g)) {
+ hits.push({ where, name: m[1], port: Number(m[2]) });
+ }
+ for (const m of line.matchAll(/--ports\s+(\S+)/g)) {
+ for (const entry of m[1].split(",")) {
+ const [name, port] = entry.split(":");
+ hits.push({ where: `${where} --ports`, name, port: Number(port) });
+ }
+ }
+ }
+ }
+ for (const file of e2eSources()) {
+ const text = readFileSync(file, "utf8");
+ const where = path.relative(ROOT, file);
+ // A chain (`process.env.EXPORT_E2E_PORT ?? process.env.PORT ?? 3020`) is the
+ // FIRST name's default; the later names are fallbacks for a bare run.
+ for (const m of text.matchAll(
+ /process\.env\.([A-Z0-9_]*PORT)(?:\s*\?\?\s*process\.env\.[A-Z0-9_]+)*\s*\?\?\s*(\d+)/g,
+ )) {
+ hits.push({ where, name: m[1], port: Number(m[2]) });
+ }
+ for (const m of text.matchAll(/PLAYWRIGHT_BASE_URL\s*\?\?\s*["'`]http:\/\/localhost:(\d+)/g)) {
+ hits.push({ where, name: "PORT", port: Number(m[1]) });
+ }
+ }
+ return hits;
+}
+
+test("every spelled port default agrees with ports.mjs", () => {
+ const hits = found();
+ // The scan is not vacuous: the editor's dev script and its queue-lock list
+ // are both spelled today, and must be found.
+ assert.ok(hits.some((h) => h.where.startsWith("editor/package.json") && h.name === "EDITOR_PORT"));
+ assert.ok(hits.some((h) => h.where.endsWith("--ports") && h.name === "PORT"));
+ const wrong = hits
+ .filter((h) => PORT_BASES[h.name] !== h.port)
+ .map((h) =>
+ h.name in PORT_BASES
+ ? `${h.where}: ${h.name} defaults to ${h.port}, ports.mjs says ${PORT_BASES[h.name]}`
+ : `${h.where}: ${h.name} is not declared in common/lib/ports.mjs`,
+ );
+ assert.deepEqual(wrong, []);
+});
+
+test("no two ports share a base, and a worktree block never overlaps the next", () => {
+ const bases = Object.values(PORT_BASES);
+ assert.equal(new Set(bases).size, bases.length);
+ // The 30xx family must fit inside one OFFSET_STEP, or worktree #1's block
+ // would reach into worktree #2's.
+ const family = bases.filter((b) => b >= 3000 && b < 4000);
+ assert.ok(Math.max(...family) - Math.min(...family) < OFFSET_STEP);
+});
+
+test("portsForOffset adds the offset to every port and names the editor test URL", () => {
+ const env = portsForOffset(300);
+ for (const [name, base] of Object.entries(PORT_BASES)) {
+ assert.equal(env[name], String(base + 300));
+ }
+ assert.equal(env.PLAYWRIGHT_BASE_URL, `http://localhost:${PORT_BASES.PORT + 300}`);
+ assert.equal(Object.keys(env).length, Object.keys(PORTS).length + 1);
+});
+
+test("portFor: the env wins, else the base; an unknown name or a non-port throws", () => {
+ assert.equal(portFor("PORT", {}), 3011);
+ assert.equal(portFor("PORT", { PORT: "3611" }), 3611);
+ assert.equal(portFor("PORT", { PORT: " " }), 3011);
+ assert.throws(() => portFor("PROT", {}), /no port named PROT/);
+ assert.throws(() => portFor("PORT", { PORT: "30x1" }), /is not a port/);
+});
diff --git a/common/lib/settings.ts b/common/lib/settings.ts
@@ -119,7 +119,13 @@ function finishRawMigrations(
}
export function getSettings(): SiteSettings {
- const raw = rawObject(readRawSettings(getPaths().settingsFile));
+ return settingsFromFile(getPaths().settingsFile);
+}
+
+// getSettings for a named file: the same read, the same migrations, no write.
+// `archilyzer doctor` reads the file its (injectable) paths name through this.
+export function settingsFromFile(file: string): SiteSettings {
+ const raw = rawObject(readRawSettings(file));
return finishRawMigrations(siteSettingsSchema.parse(premigrateRaw(raw)), raw);
}
diff --git a/common/lib/settingsSchema.ts b/common/lib/settingsSchema.ts
@@ -414,10 +414,12 @@ export const DIGEST_SETTINGS_FIELD_DOCS: FieldDocs<DigestSettings> = {
"fresh. Empty = default.",
};
-// "basic" — `pnpm run build` in export/, serialized on the build queue (shared
-// output tree → no safe parallelism).
-// "docker" — isolated per-site container builds (follow-up); enables real
-// parallel multi-site builds capped by maxParallelBuilds.
+// "basic" | "docker" — persisted and shown on /sites, but A LABEL TODAY: no
+// build path reads it. Build all / Build & deploy all (editor buildAction.ts)
+// fan out in containers (publish/build.ts, runDockerBuildAllPhase, capped by
+// maxParallelBuilds) whenever `docker version` answers, and build serially on
+// the host otherwise, whichever mode is set. Whether they should honour it is
+// an open question for the operator (plans/release-11.md, slice O6).
export type BuildMode = "basic" | "docker";
// Each field is documented in BUILD_PIPELINE_SETTINGS_FIELD_DOCS below (rendered into SETTINGS.md).
@@ -430,10 +432,11 @@ export type BuildPipelineSettings = {
export const BUILD_PIPELINE_SETTINGS_FIELD_DOCS: FieldDocs<BuildPipelineSettings> = {
mode:
- "\"basic\" — `pnpm run build` in export/, serialized on the build queue (shared output tree, no safe parallelism). \"docker\" — isolated per-site container builds, parallel up to `maxParallelBuilds`.",
+ "\"basic\" or \"docker\". Persisted and shown on /sites, but no build path reads it today — a label, not a switch: Build all sites uses containers whenever a container engine answers, and builds serially on the host when none does, in either mode.",
maxParallelBuilds:
- "Cap on concurrent per-site container builds in docker mode. Ignored in" +
- " basic mode (which is always serial). Clamped to [1, " +
+ "Cap on concurrent per-site container builds when Build all sites runs in" +
+ " containers (whenever a container engine answers, whatever `mode` says)." +
+ " Clamped to [1, " +
"BUILD_MAX_PARALLEL_MAX].",
dockerImage:
"Tag of the reusable build image (built once, reused for every site).",
@@ -1524,7 +1527,7 @@ export const siteSettingsSchema = z.object({
"Where a channel's downloaded media goes when it is relocated off the corpus disk. A DEFAULT ONLY: the relocate controller never reads it and always takes an explicit root, so this is the value the per-channel Storage panel prefills and the /channels bulk move falls back to. Blank = no default. See StorageSettings.",
),
buildPipeline: settingsField((v): BuildPipelineSettings => sanitizeBuildPipeline(v)).describe(
- "How the static export is built: \"basic\" reuses the single export/ tree and serializes builds on one queue (the long-standing behavior); \"docker\" runs each site's build in an isolated container for safe parallelism. The Docker pipeline itself is a follow-up; this block persists the chosen mode plus the container/concurrency knobs the deploy page and the future orchestrator read.",
+ "The build pipeline's settings. A single site builds in the shared export/ tree, serialized on one queue. Build all sites (and Build & deploy all) builds every site at once, each in its own container (Dockerfile.build, the image and maxParallelBuilds below), then deploys them serially, whenever a container engine answers, and serially on the host when none does — see PUBLISH.md. `mode` is persisted and shown on /sites, but no build path reads it today: it is a label, not a switch.",
),
digest: settingsField((v): DigestSettings => sanitizeDigest(v)).describe(
"AI digest generation (chapters + topic tags over the existing transcripts). Local-first: the metered lane is off by default. See DigestSettings.",
diff --git a/common/lib/toolProbe.mjs b/common/lib/toolProbe.mjs
@@ -0,0 +1,94 @@
+// Is this external tool on this machine, and which version? — the one probe.
+//
+// Plain ESM (no app imports) so umtool's `lib/tools.mjs`, which `umtool doctor`
+// runs under bare node, and `archilyzer doctor` probe a binary the SAME way: a
+// doctor that asked a different question than its twin would be worse than no
+// doctor. Each caller keeps its own TABLE of tools (what it needs, and why);
+// only the asking lives here.
+//
+// READ-ONLY by construction: a probe runs the tool's version flag (or stats a
+// file) and nothing else. Never call it from a page render — it forks.
+import { execFile } from "node:child_process";
+import { stat } from "node:fs/promises";
+import { promisify } from "node:util";
+
+const execFileP = promisify(execFile);
+
+/**
+ * @typedef {{ id: string, bin?: string, file?: string, args?: string[], fallback?: string, neededBy: string[], required?: boolean }} ToolSpec
+ * @typedef {{ id: string, bin: string, present: boolean, version: string | null, error: string | null, neededBy: string[], required: boolean }} ToolReport
+ */
+
+const firstLine = (/** @type {unknown} */ s) =>
+ String(s ?? "").split("\n").find((l) => l.trim()) ?? "";
+
+/**
+ * `ffmpeg version 7.1.1 …` -> `7.1.1`; `2025.08.11` -> itself.
+ * @param {unknown} text
+ * @returns {string | null}
+ */
+export function versionOf(text) {
+ const line = firstLine(text);
+ const m = line.match(/(\d+\.\d+(?:\.\d+)*(?:[-_.][A-Za-z0-9]+)*)/);
+ return m ? m[1] : line.slice(0, 60) || null;
+}
+
+/**
+ * Probe one tool: run `bin args` (5 s cap), or stat `file` when the spec names a
+ * file instead of a binary. ENOENT is "not on this machine"; any other exit
+ * means the binary RAN — an unusual version flag, say — which is presence,
+ * honestly reported. A `fallback` (ImageMagick 6's `convert` for 7's `magick`)
+ * is reported as absent-with-a-reason, because the pipeline calls `bin`.
+ * `env` is the environment the tool runs under (its PATH decides which binary a
+ * bare name finds); default: this process's.
+ * @param {ToolSpec} t
+ * @param {{ env?: NodeJS.ProcessEnv }} [opts]
+ * @returns {Promise<ToolReport>}
+ */
+export async function probeTool(t, opts = {}) {
+ const base = {
+ id: t.id,
+ bin: t.file ?? t.bin ?? "",
+ neededBy: t.neededBy,
+ required: !!t.required,
+ };
+ if (t.file) {
+ const ok = await stat(t.file).then(
+ (s) => s.isFile(),
+ () => false,
+ );
+ return { ...base, present: ok, version: null, error: ok ? null : `${t.file} is missing` };
+ }
+ const args = t.args ?? ["--version"];
+ /** @param {string} bin */
+ const run = async (bin) => {
+ try {
+ const { stdout, stderr } = await execFileP(bin, args, {
+ timeout: 5000,
+ maxBuffer: 1 << 20,
+ ...(opts.env ? { env: opts.env } : {}),
+ });
+ return { present: true, version: versionOf(stdout || stderr), error: null };
+ } catch (/** @type {any} */ err) {
+ if (err?.code === "ENOENT") return { present: false, version: null, error: `${bin}: not found` };
+ const said = versionOf(err?.stdout || err?.stderr);
+ return {
+ present: true,
+ version: said || null,
+ error: said ? null : `${bin} exited ${err?.code ?? "?"} on ${args.join(" ")}`,
+ };
+ }
+ };
+ let r = await run(/** @type {string} */ (t.bin));
+ if (!r.present && t.fallback) {
+ const f = await run(t.fallback);
+ if (f.present) {
+ r = {
+ present: false,
+ version: f.version,
+ error: `only \`${t.fallback}\` is installed (ImageMagick 6); the pipeline calls \`${t.bin}\``,
+ };
+ }
+ }
+ return { ...base, ...r };
+}
diff --git a/common/package.json b/common/package.json
@@ -32,6 +32,8 @@
"./components/virtualizer": "./components/virtualizer.ts",
"./components/*": "./components/*.tsx",
"./lib/detectPlatform.mjs": "./lib/detectPlatform.mjs",
+ "./lib/ports.mjs": "./lib/ports.mjs",
+ "./lib/toolProbe.mjs": "./lib/toolProbe.mjs",
"./ytdlp/platformArgs.mjs": "./ytdlp/platformArgs.mjs",
"./lib/*": "./lib/*.ts",
"./controller/*": "./controller/*.ts",
diff --git a/common/publish/build.ts b/common/publish/build.ts
@@ -177,7 +177,7 @@ export const PREVIEW_SHARES_ARCHIVES_NOTICE =
// Cache-Control set on every uploaded archive. Served through a Cloudflare custom
// domain, this lets the CDN absorb repeated/abusive downloads at the edge instead
// of hitting R2 (each origin GET is a billable Class B op), which is the main cost
-// defense for public archives — see DEPLOY_CLOUDFLARE.md. 1h balances flood
+// defense for public archives — see PUBLISH.md. 1h balances flood
// absorption against re-deployed archives (stable filenames, overwritten in place)
// going stale; raise it if your archives rarely change.
const ARCHIVE_CACHE_CONTROL = "public, max-age=3600";
@@ -230,7 +230,7 @@ export async function runArchiveUploadIntoLog(
`[archives] ${files.length} oversize archive(s) need uploading to R2 bucket ` +
`"${bucket}", but R2 S3 credentials are missing. Set R2_ACCESS_KEY_ID, ` +
`R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID in the environment — see ` +
- `DEPLOY_CLOUDFLARE.md. Aborting before deploy so the site never links to ` +
+ `PUBLISH.md ("Download archives and R2"). Aborting before deploy so the site never links to ` +
`missing files.\n`,
);
return 1;
@@ -377,7 +377,7 @@ export async function runDeployIntoLog(
// ---------------------------------------------------------------------------
// Docker export pipeline (buildPipeline.mode = "docker")
//
-// Three ordered phases (see DEPLOY_DOCKER.md):
+// Three ordered phases (see PUBLISH.md, "Building every site in containers"):
// A) HOST, serial: build:data (shared LMDB + .export-index) then build:archives
// (warm the shared archive cache). One writer of the shared state.
// B) CONTAINERS, parallel (cap maxParallelBuilds): each site's compose + next
@@ -473,7 +473,7 @@ async function runDockerBuildOne(
const gid = typeof process.getgid === "function" ? process.getgid() : null;
if (uid !== null && gid !== null) args.push("-u", `${uid}:${gid}`);
// Optional resource caps so N parallel builds (each next build can use ~8 GB)
- // don't OOM the host. Sized by the operator; see DEPLOY_DOCKER.md.
+ // don't OOM the host. Sized by the operator; see PUBLISH.md.
const mem = process.env.DOCKER_BUILD_MEMORY?.trim();
const cpus = process.env.DOCKER_BUILD_CPUS?.trim();
if (mem) args.push("--memory", mem);
diff --git a/docker/entrypoint.sh b/docker/entrypoint.sh
@@ -47,14 +47,16 @@ mkdir -p \
# ---------------------------------------------------------------------------
# 2. Seed settings.json — with a worker.
#
-# defaults() in common/lib/settings.ts returns `workers: []`, and zero workers
-# means zero transcription slots: auto-transcribe reports `no-workers` and does
-# nothing at all, silently. A fresh container that looks healthy and transcribes
-# nothing is the worst possible first run, so the seed carries exactly one local
-# whisper.cpp worker.
+# With no `workers` key (or no file), getSettings() synthesizes
+# `parallelTranscriptions` (default 2) enabled workers of the default app,
+# whisper.cpp — two CPU whisper slots, and never parakeet in the Vulkan image.
+# (defaults() alone has `workers: []`; zero workers — auto-transcribe silently
+# doing nothing — only happens for a file that says `"workers": []`.) So the
+# seed carries exactly one local worker for the image's engine.
#
# Everything else is left out on purpose. getSettings() merges a partial file
-# over defaults(), so a short seed is a FEATURE: keys we don't write here keep
+# over defaults() (every key but `workers`, above), so a short seed is a
+# FEATURE: keys we don't write here keep
# tracking the app's own defaults as those move, instead of being frozen at
# whatever they were the day this image was built.
#
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -6,6 +6,7 @@
- **"Persist source video" or a whole-recording fetch that cannot get the source no longer marks the video's download failed.** When YouTube's subtitles came down but the source video did not, the video page said "Download failed" over a transcript that is fine. The download now keeps the subtitle pass's result and records only the failed media attempt, with yt-dlp's reason. The run itself now ends failed with that reason; it used to end done with no file, so `fetch_clip` could only say the job "finished but named no file". A partial source file is left for a retry to resume, and the run's log names it.
- **A bucket's retry keeps its log when it empties the bucket.** On a channel's Download stage, "Download with cookies", the partial-download resume and the missing-transcript retry could lose their run log part-way: the video they fetched left the bucket, the page refreshed, and the card disappeared with the log in it. The card now stays, with its log and its button disabled, until the page is reloaded. The Transcribe stage's "Fetch audio" button does the same. (A Diagnostics card still disappears, log and all, when its retry empties it.)
- **A release cut whose commit fails still refreshes the pages.** When the changelog's new heading was written but the commit after it failed, the Cut release form and `pnpm ops cut-release` answered as if nothing had happened and no page showed the new heading until a reload. Both now refresh the changelog pages, and `pnpm ops cut-release` says the file was written. Every refused cut's answer says whether anything was written (`untouched`), and a cut of both changelogs that stopped half-way names the one already cut as well as the failure.
+- **`archilyzer` checks the machine, runs one operation offline, starts the MCP server, and is one command from the repo root.** `pnpm archilyzer <command>` is the short form (`pnpm archilyzer --help` lists them all). `pnpm archilyzer doctor` is a read-only report: Node, the checkout, the corpus and whether each channel's media is reachable, `settings.json`, every tool the paths name plus each enabled worker's engine and model, umtool's report-pipeline tools, and this checkout's ports; it exits 1 only for something the machine is set up to do and cannot. `pnpm archilyzer run <operation> <channel> [ids…]` runs diarization, either attribution pass or digest over one channel as the editor's job does (a job record and log under `.jobs/`, the same summary line, the same refusal for an unmounted drive); sync, the metadata scan, downloads and transcription are refused with the reason, because they run on the editor's paced download queue and worker pool. `pnpm archilyzer mcp` starts the MCP server, so it can be registered as `-- pnpm -C "$PWD" archilyzer mcp`. Every other script in `common/bin/` is a subcommand too (`duplicates`, `posts fetch`, `digest plan`, `verify transcripts`, …), and export's `detect:duplicates` script is now `archilyzer duplicates`. Every environment variable is listed, by audience, in the new `ENVIRONMENT.md`, and `DEPLOY_CLOUDFLARE.md` and `DEPLOY_DOCKER.md` are now one `PUBLISH.md`. Settings and `/sites` no longer call the Docker build mode a follow-up, and say what it is: a label. **Build all sites** builds in containers whenever a container engine answers, whichever mode is set.
## [0.9.4] - 2026-09-28
- **On the Dark ground the sidebar's Archilyzer mark has a thin outline.** Its slate tile now has a 1-pixel ring just outside it, following its rounded corners, in the colour of the mark's unlit lines, so the tile's edge shows against the dark page. Light and Sepia are unchanged, and so is the favicon.
@@ -220,7 +221,7 @@
- **Channel forms now edit site membership directly — pick sites, groups, and create groups inline.** A channel's site membership used to be editable only from the Site form (`/sites/<id>`), and creating a channel silently appended it to the active site (a hidden `activeSite` field) with no group choice and no visibility. Both the **New channel** and channel **Configure** forms now carry a **Sites** section: every configured site listed with a membership checkbox and a compact group dropdown — `(default)`, any existing group, or **"+ New group…"**, which reveals a name input and creates the group on that site (id slugified from the name, reused if it already exists) as part of the save. On create, the active site (`?site=`) is pre-checked, reproducing the old behavior but visibly and overridably; on edit, current memberships pre-check with their groups, and unchecking removes the membership. The checked sites serialize into one hidden `siteMembershipsJson` field (the SiteForm hidden-JSON precedent); the server plans all writes up front — validating group ids, preserving other channels' entries and this channel's `order`, erroring clearly on a since-deleted group, skipping since-deleted sites, and rewriting only sites that actually changed. See the new `editor/app/channels/{lib/siteMemberships.ts,components/SiteMembershipsSection.tsx}`, `editor/app/channels/{actions.ts,components/{ChannelForm,ChannelFormClient}.tsx,new/page.tsx,[slug]/page.tsx}`, and `editor/e2e/channel-site-membership.spec.ts`.
- **The monitor widget can now start a sync and shows sync freshness + scheduler health — each behind its own flag.** The widget was start-a-sync-less and said nothing about how fresh your channels were. Five new opt-in URL flags, all toggleable in the builder and the in-widget gear (all default off, so existing links are unchanged): **(1) a channel-aware Sync button** (`sync=1`) — pinned to one channel (`channel=X`) it runs that channel's streaming sync; otherwise it sweeps all channels (`syncAllChannelsAction`), briefly showing `Queued X · skipped Y`. **(2) A last-sync readout** (`lastsync=1`) — `Last full sync: …` from a new persisted `lastSyncAllAt` marker written at the end of each Sync-all sweep, plus a `Last channel sync: …` line when an individual channel synced more recently. **(3) A scheduler-status strip** (`sched=1`) — `Auto-sync on/off · next … · last run …`, derived from the same `buildScheduleView` the `/scheduler` page uses. **(4) An absolute-time toggle** (`abstime=1`) — readouts show locale timestamps instead of relative "5m ago". **(5) A Sync-all confirm** (`syncask=1`) — a `window.confirm` before a full sweep. The two readouts poll a new lightweight `/api/widget/sync` route (a few scalars, 15s floor) and, with the Sync button/controls, keep the widget from collapsing to "Idle" — sync health is exactly what you check when nothing's running. See `editor/app/widget/lib/config.ts`, `editor/app/widget/components/{MonitorWidget,WidgetControls,WidgetConfigForm}.tsx`, the new `editor/app/api/widget/sync/route.ts`, `common/jobs/syncSchedulerState.ts` (`lastSyncAllAt`), `editor/app/channels/actions.ts`, and `editor/e2e/widget.spec.ts`.
- **The audio-integrity check now tightens its probe interval when a source starts serving corruption, then relaxes as it stabilises.** Previously the integrity probe ran on a fixed cadence (default 60s) for the whole download, so up to ~60s of bytes were downloaded — and discarded — between a corruption event and the checkpoint that caught it. The interval is now adaptive (AIMD, like TCP congestion control, inverted): each **malformed** checkpoint **halves** the live interval (60→30→15→10s, floored at the existing `AUDIO_CHECK_INTERVAL_MIN_SECONDS` of 10s), so a misbehaving source gets probed more aggressively and wastes fewer bytes per rollback; a run of clean checkpoints then **steps it back up** additively (+15s after every 2 clean probes) toward the configured interval. The reduced cadence persists across yt-dlp relaunches for the rest of the download run. Fully backward compatible — a clean download never leaves the configured interval. Tunable via constants in `common/lib/channelConfig.ts` (`AUDIO_CHECK_INTERVAL_BACKOFF_FACTOR_DEFAULT`, `AUDIO_CHECK_INTERVAL_RECOVER_STEP_SECONDS`, `AUDIO_CHECK_INTERVAL_RECOVER_AFTER_CLEAN`) plus test-only env overrides. See `common/ytdlp/audioCheckCadence.ts` (pure AIMD math + `audioCheckCadence.test.ts`) and `common/ytdlp/audioCheckedDownload.ts` (`resolveKnobs`, the watcher loop, and the advance/malformed checkpoint branches).
-- **Docker build mode is now real: build every site in parallel, then deploy them serially.** The `Docker` build mode (Settings → Build pipeline) was previously a stub that fell back to the basic build. It now runs a proper pipeline, driven by a new **Build all sites** control on the Deploy page (one job, one log, one Cancel). The shared, corpus-scale work — the search index, the per-site staging, and the downloadable archive zips — runs **once on the host**; then each site's `compose + next build` runs in its **own container in parallel** (capped by the **Max parallel builds** setting), each writing an isolated per-site `out/` under `export/.export-builds/<siteId>/`; then the built sites **deploy serially** on the host (R2 upload + `wrangler pages deploy`), tolerant of a single site failing. Containers are read-only over the shared corpus/index/archive cache and run as your host user so outputs aren't root-owned. The image (`Dockerfile.build`, tag from **Build image**) is built/reused via Docker layer caching; when no container engine is available the action falls back to a serial host build+deploy. New env knobs: `DOCKER_BIN` (e.g. `podman`), `DOCKER_BUILD_MEMORY`/`DOCKER_BUILD_CPUS` (per-container caps). See `editor/app/deploy/buildDeployCore.ts` (`runDockerBuildAllPhase`/`runDockerDeployAllPhase`), `editor/app/build/buildAction.ts` (`buildAndDeployAllSitesAction`/`buildAllSitesAction`), `editor/app/deploy/components/BuildAllSitesButton.tsx`, `Dockerfile.build`, `docker/build-site.sh`, `common/bin/build-archives.ts`, and **[DEPLOY_DOCKER.md](../DEPLOY_DOCKER.md)**.
+- **Docker build mode is now real: build every site in parallel, then deploy them serially.** The `Docker` build mode (Settings → Build pipeline) was previously a stub that fell back to the basic build. It now runs a proper pipeline, driven by a new **Build all sites** control on the Deploy page (one job, one log, one Cancel). The shared, corpus-scale work — the search index, the per-site staging, and the downloadable archive zips — runs **once on the host**; then each site's `compose + next build` runs in its **own container in parallel** (capped by the **Max parallel builds** setting), each writing an isolated per-site `out/` under `export/.export-builds/<siteId>/`; then the built sites **deploy serially** on the host (R2 upload + `wrangler pages deploy`), tolerant of a single site failing. Containers are read-only over the shared corpus/index/archive cache and run as your host user so outputs aren't root-owned. The image (`Dockerfile.build`, tag from **Build image**) is built/reused via Docker layer caching; when no container engine is available the action falls back to a serial host build+deploy. New env knobs: `DOCKER_BIN` (e.g. `podman`), `DOCKER_BUILD_MEMORY`/`DOCKER_BUILD_CPUS` (per-container caps). See `editor/app/deploy/buildDeployCore.ts` (`runDockerBuildAllPhase`/`runDockerDeployAllPhase`), `editor/app/build/buildAction.ts` (`buildAndDeployAllSitesAction`/`buildAllSitesAction`), `editor/app/deploy/components/BuildAllSitesButton.tsx`, `Dockerfile.build`, `docker/build-site.sh`, `common/bin/build-archives.ts`, and **[DEPLOY_DOCKER.md](../PUBLISH.md#building-every-site-in-containers)**.
## [0.7.3] - 2026-07-07
- **Deploys no longer re-upload unchanged oversize archives to R2.** The deploy step used to stream every over-cap archive to R2 on every deploy, even ones byte-identical to what was already there. Because the export build now reuses an unchanged channel's cached zip verbatim, the upload step first does a cheap `HeadObject` and skips any archive whose R2 object already has the same size — so a redeploy after changing one channel only re-uploads that channel's oversize bundle. See `editor/app/deploy/buildDeployCore.ts`.
@@ -229,7 +230,7 @@
- **Search results scroll smoothly again on large result sets.** The results list windows one card per matching video (only the on-screen cards are mounted), but each visible card and every one of its hit rows was re-rendering on *every* scroll frame — and each hit row re-ran its `<mark>` highlighting, so a single video with hundreds of hits meant hundreds of redundant highlight passes per frame while scrolling. The result cards and individual hit rows are now memoized so an unchanged card/row is skipped during scroll, and opening the modal on a hit only re-renders the two rows whose highlight state actually changes. No visible/behavioral change — same DOM, same results, just far less work per frame. See `common/components/TranscriptSearch.tsx` (`ResultCard`/`HitRow` memoization, `openWithMode` stabilized via `useCallback`).
- **The monitor widget gains a needs-work channel list, more interaction buttons, and an in-place settings gear.** Three additions, all driveable from the widget builder. **(1) A "Needs work" list** (URL flag `act=1`) — a compact, per-channel worklist of videos to download (`↓ N`) or transcribe (`✎ N`), reusing the same `loadActionableSummary` that powers the `/actionable` page via a new `/api/widget/actionable` route; it polls on a 15s floor (the backlog changes on job completions, not seconds) and caps at 6 channels with a `+N more` line. **(2) More interactions** behind the existing `controls=1` switch: each needs-work row gains the same per-channel **Download missing** / **Transcribe pending** buttons as the actionable page (reusing `InlineActionButton`), and the controls row adds **Retry all failed** alongside Pause/Resume + Drain. **(3) An in-place settings gear** (on by default; URL flag `gear=0` to hide, or a **Show settings gear** builder checkbox) — clicking it opens the builder's own form *inside the widget window*, so a pinned widget can be reconfigured live without opening the builder page; edits apply immediately and mirror into the address bar via `history.replaceState`, so a reload preserves them and the link stays copyable. The builder form is extracted into a shared `WidgetConfigForm` used by both the builder and the overlay, and the widget's poller now fetches immediately on (re)subscribe instead of after one interval, so newly-enabled sections render at once. Existing links render unchanged (the two new flags default to their old behavior; the gear is the one new default-visible affordance and is read-only — it mutates no server state). See `editor/app/widget/lib/config.ts`, the new `editor/app/widget/components/WidgetConfigForm.tsx` and `editor/app/api/widget/actionable/route.ts`, `editor/app/widget/components/{MonitorWidget,WidgetControls}.tsx`, `editor/app/widget/builder/components/WidgetBuilder.tsx`, and `editor/e2e/widget.spec.ts`.
- **Every site build now bundles downloadable per-channel transcript & live-chat archive zips.** The archive builders (per-channel `<slug>.zip` / `<slug>.live_chat.zip`) previously only ran as standalone actions that wrote to a non-served directory; now `compose-site` generates them for the site's own channels straight into the served `public/archives/` and writes a `manifest.json` (sizes + counts) that the site's new **Downloads** page reads. **`zip` is now the default archive format** everywhere (was `tar.gz`), and the Build page's format help text tracks the selected format. Generation is **on by default with three opt-out levels**: a global **Generate archive zips on build** toggle in Settings, a per-site **Generate archive zips** toggle (plus an optional **Archive size cap (MB)**) on the site's page, and a per-build **Skip archive zips** checkbox on the Build and Build & Deploy controls (`BUILD_ARCHIVES=0`). See `common/bin/compose-site.ts` (`composeArchives`), `common/controller/archive{Transcripts,LiveChat}.ts` (new `outDir` option), `common/lib/archiveOptions.ts` (default + manifest types), `common/lib/{site,settings}.ts` (opt-out flags), and `editor/app/{deploy/buildDeployCore.ts,build/buildAction.ts,deploy/components/Build{Export,Deploy}Button.tsx,sites/components/SiteForm.tsx,settings/components/SettingsForm.tsx}`.
-- **Oversize archives now overflow to Cloudflare R2 instead of being dropped.** A single file over 25 MB breaks a Cloudflare Pages deploy, so a channel zip over the cap (default 25 MB; `0` = no cap) used to be removed from what's served and flagged `oversize`. Now, when **Archive overflow storage** is configured in Settings (an R2 **bucket** + its **public URL**), `compose-site` stages each oversize archive to `export/.r2-staging/<siteId>/` and records its future public URL in the manifest; the deploy step then uploads it to `<bucket>/<siteId>/archives/<file>` **before** the Pages deploy, so the Downloads page links straight to R2. Uploads go over R2's **S3 API** using the AWS SDK's multipart uploader (`@aws-sdk/lib-storage`) — `wrangler r2 object put` caps a single upload at 300 MiB and real live-chat archives are larger, whereas multipart streams any size. This needs R2 **S3 credentials** in the environment (`R2_ACCESS_KEY_ID`, `R2_SECRET_ACCESS_KEY`, `CLOUDFLARE_ACCOUNT_ID`); if they're missing when there's something to upload, the deploy fails before the Pages step (so the site never links to a missing file). With no bucket configured the old drop-and-flag behavior is unchanged (uploads run only in the editor's Deploy / Build & deploy actions, not a raw `pnpm deploy`). The **combined "whole site" archives were removed** — they duplicated the per-channel content and were always the first to blow the cap. Each R2 upload also now sets `Cache-Control: public, max-age=3600` so a Cloudflare custom domain caches downloads at the edge — the main defense against download-abuse cost (R2 egress is free; only origin reads are billable, and cached hits skip the origin). New **[DEPLOY_CLOUDFLARE.md](../DEPLOY_CLOUDFLARE.md)** documents the full R2 setup plus the Cloudflare custom-domain / caching / rate-limiting / bot config for instance operators. See `common/lib/settings.ts` (`archiveStorage`), `common/bin/compose-site.ts` (staging + manifest `url`), `common/lib/archiveOptions.ts` (`ArchiveManifestEntry.url`), and `editor/app/deploy/buildDeployCore.ts` (`runArchiveUploadIntoLog`, `ARCHIVE_CACHE_CONTROL`) / `deploy/deployAction.ts` / `build/buildAction.ts`.
+- **Oversize archives now overflow to Cloudflare R2 instead of being dropped.** A single file over 25 MB breaks a Cloudflare Pages deploy, so a channel zip over the cap (default 25 MB; `0` = no cap) used to be removed from what's served and flagged `oversize`. Now, when **Archive overflow storage** is configured in Settings (an R2 **bucket** + its **public URL**), `compose-site` stages each oversize archive to `export/.r2-staging/<siteId>/` and records its future public URL in the manifest; the deploy step then uploads it to `<bucket>/<siteId>/archives/<file>` **before** the Pages deploy, so the Downloads page links straight to R2. Uploads go over R2's **S3 API** using the AWS SDK's multipart uploader (`@aws-sdk/lib-storage`) — `wrangler r2 object put` caps a single upload at 300 MiB and real live-chat archives are larger, whereas multipart streams any size. This needs R2 **S3 credentials** in the environment (`R2_ACCESS_KEY_ID`, `R2_SECRET_ACCESS_KEY`, `CLOUDFLARE_ACCOUNT_ID`); if they're missing when there's something to upload, the deploy fails before the Pages step (so the site never links to a missing file). With no bucket configured the old drop-and-flag behavior is unchanged (uploads run only in the editor's Deploy / Build & deploy actions, not a raw `pnpm deploy`). The **combined "whole site" archives were removed** — they duplicated the per-channel content and were always the first to blow the cap. Each R2 upload also now sets `Cache-Control: public, max-age=3600` so a Cloudflare custom domain caches downloads at the edge — the main defense against download-abuse cost (R2 egress is free; only origin reads are billable, and cached hits skip the origin). New **[DEPLOY_CLOUDFLARE.md](../PUBLISH.md#download-archives-and-r2)** documents the full R2 setup plus the Cloudflare custom-domain / caching / rate-limiting / bot config for instance operators. See `common/lib/settings.ts` (`archiveStorage`), `common/bin/compose-site.ts` (staging + manifest `url`), `common/lib/archiveOptions.ts` (`ArchiveManifestEntry.url`), and `editor/app/deploy/buildDeployCore.ts` (`runArchiveUploadIntoLog`, `ARCHIVE_CACHE_CONTROL`) / `deploy/deployAction.ts` / `build/buildAction.ts`.
- **Archive downloads can now be served securely without owning a domain (`r2-proxy/` Worker).** Serving oversize R2 archives with cost/abuse protection previously implied a Cloudflare **custom domain** (for CDN caching + rate-limiting rules). New self-contained Cloudflare Worker at `r2-proxy/` serves the bucket on a free `*.workers.dev` subdomain instead — the same domain-free model as Pages' `*.pages.dev`, addressing the privacy/expense of registering a domain. It's a pure passthrough (request path `<siteId>/archives/<file>.zip` → bucket key), so **one Worker serves every site** — deploy it once, not per site. It adds edge caching (Cache API, honoring the object's `Cache-Control`), native per-IP rate limiting (free binding), path allow-listing (`*/archives/*.zip` only), and `Range`/resumable-download support. Point the editor's **Archive overflow public URL** at the `workers.dev` URL and nothing else changes (uploads/manifest are identical). `DEPLOY_CLOUDFLARE.md` now documents both paths (Worker vs custom domain), with the Worker as the recommended no-domain option. See `r2-proxy/{src/index.ts,wrangler.toml,package.json,README.md}` and `editor/app/settings/components/SettingsForm.tsx` (public-URL hint).
- **The Duplicates page is now a per-site toggle and hides itself when empty.** Each site's editor page gains a **Show the Duplicates page** checkbox (on by default). `compose-site` writes the site-filtered `duplicates.json` only when the toggle is on *and* there's at least one in-scope cluster, and the export Header keys its Duplicates nav link off a new `hasDuplicates()` — so the link and page disappear both when a site opts out and when it simply has no detected duplicates. See `common/lib/site.ts` (`duplicates` flag), `common/bin/compose-site.ts` (gated write), `export/app/lib/duplicates.ts` (new), `export/app/components/Header.tsx`, and `editor/app/sites/{components/SiteForm.tsx,actions.ts}`.
- **You can now change a channel's slug (its id) — deliberately, from the Danger zone.** A channel's slug *is* its on-disk directory name (`transcripts/channels/<slug>/`), so it used to be fixed at creation ("Slug is fixed once a channel is created"). A new **Rename** form in the channel's Danger zone lifts that: enter a new slug and **type the current slug to confirm** (same friction as delete), and the rename is blocked while the channel has running/queued jobs (the in-memory registry keys by slug). Because the slug is a directory name, the rename does a **full migration** of every slug-keyed store so nothing silently breaks: it moves the channel dir (config, data, playlist, snapshot, shards, failed lists) **and** the saved-video store dir — rewriting each `saved-video.json` pointer's absolute `dir` so persisted source videos still resolve — then retargets every site.json membership, the sync scheduler's per-channel backoff state, and any job bookmarks. The two filesystem moves run first and roll back on failure; the metadata updates that follow are atomic and best-effort (surfaced as warnings). Renaming **changes the channel's public URL** (the old one 404s), which the form warns about. The slug grammar is also now validated on create. See `common/controller/renameChannel.ts`, `common/controller/channels.ts` (`isValidChannelSlug`), `common/lib/savedVideo-server.ts` (`rewriteSavedVideoDir`), `common/jobs/bookmarks.ts` (`renameChannelInBookmarks`), `editor/app/channels/{actions.ts,components/RenameChannelForm.tsx,[slug]/page.tsx}`, and `editor/e2e/channel-rename.spec.ts`.
diff --git a/editor/app/channels/[slug]/digestActions.ts b/editor/app/channels/[slug]/digestActions.ts
@@ -66,6 +66,9 @@ export async function digestChannelAction(
order?: string,
limitCount?: number,
force?: boolean,
+ // Only these videos — a replay of an ids-scoped run (`archilyzer run digest
+ // <channel> <ids…>`) must not widen to the channel.
+ ids?: string[],
): Promise<StreamActionResult> {
return runDigestChannelJob({
paths: getPaths(),
@@ -75,6 +78,7 @@ export async function digestChannelAction(
order,
limitCount,
force,
+ ids,
onDone: () => safeRevalidate([`/channels/${slug}`]),
});
}
diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts
@@ -69,6 +69,9 @@ const bool = (v: unknown): boolean | undefined =>
typeof v === "boolean" ? v : undefined;
const num = (v: unknown): number | undefined =>
typeof v === "number" ? v : undefined;
+// An id scope: the strings of an array, or undefined (= the whole channel).
+const strings = (v: unknown): string[] | undefined =>
+ Array.isArray(v) ? v.filter((k): k is string => typeof k === "string") : undefined;
// Flag-style params + the captured queueKey for a spec.
function params(spec: JobSpec): {
@@ -129,6 +132,7 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = {
str(p.order),
num(p.limitCount),
bool(p.force),
+ strings(p.ids),
);
},
"digest-channel-remote": (spec) => {
@@ -140,6 +144,7 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = {
str(p.order),
num(p.limitCount),
bool(p.force),
+ strings(p.ids),
);
},
"whisper-all": (spec) => {
@@ -305,10 +310,7 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = {
const kindIds = Array.isArray(p.kindIds)
? p.kindIds.filter((k): k is string => typeof k === "string")
: undefined;
- const ids = Array.isArray(p.ids)
- ? p.ids.filter((k): k is string => typeof k === "string")
- : undefined;
- return backfillChannelAction(spec.slug, queueKey, kindIds, ids);
+ return backfillChannelAction(spec.slug, queueKey, kindIds, strings(p.ids));
},
"check-kept-deleted": (spec) => {
const { queueKey } = params(spec);
diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx
@@ -298,12 +298,12 @@ export function SettingsForm({ initial }: Props) {
<fieldset className="flex flex-col gap-3 border border-border rounded p-3">
<legend className="px-1 text-sm font-medium">Build pipeline</legend>
<p className="text-xs text-muted-foreground">
- How the static export is built. <strong>Basic</strong> runs the build
- in <code>export/</code> and serializes builds on one queue.{" "}
- <strong>Docker</strong> will run each site's build in an isolated
- container for safe parallel multi-site builds — the container pipeline
- is a follow-up, so Docker currently falls back to a basic build. The
- mode can also be toggled on the{" "}
+ A single site builds in <code>export/</code>, one at a time on one
+ queue. <strong>Build all sites</strong> builds every site at once, each
+ in its own container (capped by Max parallel builds), then deploys them
+ one by one, whenever a container engine answers, and serially on the
+ host when none does. The mode below is a label for now: no build reads
+ it. See PUBLISH.md. It can also be toggled on the{" "}
<a href="/sites" className="underline">
Sites
</a>{" "}
@@ -317,9 +317,7 @@ export function SettingsForm({ initial }: Props) {
className="rounded border border-border bg-card px-2 py-1 text-sm"
>
<option value="basic">Basic — serial build queue</option>
- <option value="docker">
- Docker — isolated parallel builds (follow-up)
- </option>
+ <option value="docker">Docker — isolated parallel builds</option>
</select>
</label>
<Field
@@ -327,7 +325,7 @@ export function SettingsForm({ initial }: Props) {
name="maxParallelBuilds"
defaultValue={String(initial.buildPipeline.maxParallelBuilds)}
type="number"
- hint="Cap on concurrent per-site container builds in Docker mode (1–16). Ignored in Basic mode, which is always serial."
+ hint="Cap on concurrent per-site container builds (1–16) when Build all sites runs in containers — whenever a container engine answers, whichever mode is set."
/>
<Field
label="Docker image tag"
diff --git a/editor/app/sites/components/BuildAllSitesButton.tsx b/editor/app/sites/components/BuildAllSitesButton.tsx
@@ -10,12 +10,13 @@ import { JobLane } from "./JobLane";
type Lane = { deploy: boolean; skipArchives: boolean; key: number };
// One-click orchestrated pipeline over EVERY site as a SINGLE managed job (one
-// log, one Cancel): in Docker mode it runs the shared data phase + archive warm
-// once, fans the per-site builds out in parallel (capped by maxParallelBuilds),
-// then deploys the built sites serially. Without a container engine it falls back
-// to a serial host build+deploy. Distinct from the per-site panel below, which
-// launches one separate job per selected site.
-export function BuildAllSitesButton({ dockerMode }: { dockerMode: boolean }) {
+// log, one Cancel): whenever a container engine answers — whatever
+// buildPipeline.mode says, which nothing reads — it runs the shared data phase +
+// archive warm once, fans the per-site builds out in parallel (capped by
+// maxParallelBuilds), then deploys the built sites serially. Without one it falls
+// back to a serial host build+deploy. Distinct from the per-site panel below,
+// which launches one separate job per selected site.
+export function BuildAllSitesButton() {
const [deploy, setDeploy] = useState(true);
const [skipArchives, setSkipArchives] = useState(false);
const [lane, setLane] = useState<Lane | null>(null);
@@ -55,16 +56,17 @@ export function BuildAllSitesButton({ dockerMode }: { dockerMode: boolean }) {
</label>
</div>
<p className="text-xs text-muted-foreground">
- {dockerMode
- ? "Docker pipeline: shared data phase runs once, per-site builds run in parallel, then deploys run serially."
- : "Docker mode is off — this runs a serial host build+deploy fallback (one site at a time)."}
+ When a container engine answers (whichever build mode is set): the shared
+ data phase runs once, per-site builds run in parallel in containers, then
+ deploys run serially. With none, a serial host build and deploy, one site
+ at a time.
</p>
{lane && (
<JobLane
key={lane.key}
title={lane.deploy ? "Build & deploy all sites" : "Build all sites"}
subtitle={
- dockerMode ? "Docker pipeline · parallel builds" : "Serial host fallback"
+ "Containers when an engine answers, else serial on the host"
}
trigger={() =>
lane.deploy
diff --git a/editor/app/sites/components/BuildModeToggle.tsx b/editor/app/sites/components/BuildModeToggle.tsx
@@ -4,8 +4,11 @@ import { useState, useTransition } from "react";
import type { BuildMode } from "yt-dlp-transcript-common/lib/settings";
import { setBuildModeAction } from "../lib/buildModeAction";
-// Segmented Basic | Docker control. Persists the choice as the global build-mode
-// default (setBuildModeAction → settings.json) so every subsequent build uses it.
+// Segmented Basic | Docker control. Persists the choice (setBuildModeAction →
+// settings.json `buildPipeline.mode`). TODAY IT IS A LABEL: no build path reads
+// it — Build all / Build & deploy all use containers whenever `docker version`
+// answers (buildAction.ts, dockerAvailable) — and the note below says so rather
+// than promising a switch.
export function BuildModeToggle({ mode: initialMode }: { mode: BuildMode }) {
const [mode, setMode] = useState<BuildMode>(initialMode);
const [pending, startTransition] = useTransition();
@@ -53,11 +56,11 @@ export function BuildModeToggle({ mode: initialMode }: { mode: BuildMode }) {
);
})}
</div>
- {mode === "docker" && (
- <span className="text-xs text-warning">
- Docker builds are a follow-up — runs the basic build for now.
- </span>
- )}
+ <span className="text-xs text-muted-foreground">
+ The mode is a label for now: Build all sites uses containers whenever a
+ container engine answers, and builds serially on the host when none
+ does, in either mode.
+ </span>
</div>
);
}
diff --git a/editor/app/sites/components/BuildSitesPanel.tsx b/editor/app/sites/components/BuildSitesPanel.tsx
@@ -13,9 +13,9 @@ export type SiteOption = {
type Lane = { siteId: string; title: string; deploy: boolean; key: string };
// Batch build (and optionally deploy) several sites at once. Each selected site
-// launches its own managed job, rendered as its own live JobLane. In Basic mode
-// the jobs share the build/deploy queue and run one at a time (the export/ tree
-// is shared); Docker mode (a follow-up) unlocks true parallelism.
+// launches its own managed job, rendered as its own live JobLane. The jobs share
+// the build/deploy queue and run one at a time (the export/ tree is shared),
+// whichever build mode is set; "Build all sites" is the parallel path.
export function BuildSitesPanel({
sites,
serial,
@@ -127,8 +127,9 @@ export function BuildSitesPanel({
)}
{serial && (
<p className="text-xs text-muted-foreground">
- Basic mode runs these one at a time (the build output tree is shared).
- Use “Build all sites” above in Docker mode for true parallel builds.
+ These run one at a time (the build output tree is shared). “Build
+ all sites” above builds every site in parallel, in containers, whenever
+ a container engine answers.
</p>
)}
diff --git a/editor/app/sites/page.tsx b/editor/app/sites/page.tsx
@@ -171,7 +171,7 @@ export default async function SitesPage() {
</p>
</div>
<BuildModeToggle mode={buildMode} />
- <BuildAllSitesButton dockerMode={buildMode === "docker"} />
+ <BuildAllSitesButton />
<div className="mt-4">
<h3 className="font-semibold">Or pick specific sites</h3>
diff --git a/editor/e2e/deploy-page.spec.ts b/editor/e2e/deploy-page.spec.ts
@@ -68,7 +68,7 @@ test("Build & deploy is enabled only when the active site has a Cloudflare proje
).toBeVisible();
});
-test("build-mode toggle persists the choice and shows the Docker follow-up note", async ({
+test("build-mode toggle persists the choice and says the mode is a label", async ({
page,
}) => {
await writeSite("testsite", { cloudflareProject: "proj" });
@@ -93,7 +93,7 @@ test("build-mode toggle persists the choice and shows the Docker follow-up note"
timeout: 1_000,
});
}).toPass({ timeout: 15_000 });
- await expect(page.getByText(/Docker builds are a follow-up/i)).toBeVisible();
+ await expect(page.getByText(/The mode is a label for now/i)).toBeVisible();
// Wait for the choice to actually reach settings.json before reloading.
// BuildModeToggle updates its own state OPTIMISTICALLY — setMode(next) flips
@@ -150,5 +150,5 @@ test("batch panel: selecting sites enables the launch button and reflects deploy
).toBeVisible();
// Basic mode (the default) notes that the batch runs serially.
- await expect(page.getByText(/Basic mode runs these one at a time/i)).toBeVisible();
+ await expect(page.getByText(/These run one at a time/i)).toBeVisible();
});
diff --git a/export/package.json b/export/package.json
@@ -5,14 +5,13 @@
"type": "module",
"scripts": {
"dev": "next dev --port ${EXPORT_DEV_PORT:-3000}",
- "build:index": "NODE_OPTIONS=--max-old-space-size=8192 tsx ../common/bin/build-index.ts",
- "build:stats": "NODE_OPTIONS=--max-old-space-size=8192 tsx ../common/bin/build-stats.ts",
- "build:templates": "tsx ../common/bin/build-chart-templates.ts",
+ "build:index": "NODE_OPTIONS=--max-old-space-size=8192 tsx ../common/bin/archilyzer.ts index",
+ "build:stats": "NODE_OPTIONS=--max-old-space-size=8192 tsx ../common/bin/archilyzer.ts build stats",
+ "build:templates": "tsx ../common/bin/archilyzer.ts build templates",
"build:data": "pnpm run build:index && pnpm run build:stats && pnpm run build:templates",
- "build:archives": "tsx ../common/bin/build-archives.ts",
- "detect:duplicates": "NODE_OPTIONS=--max-old-space-size=8192 tsx ../common/bin/duplicate-shorts.ts",
- "compose:site": "tsx ../common/bin/compose-site.ts",
- "compose:hub": "tsx ../common/bin/compose-hub.ts",
+ "build:archives": "tsx ../common/bin/archilyzer.ts build archives",
+ "compose:site": "tsx ../common/bin/archilyzer.ts compose site",
+ "compose:hub": "tsx ../common/bin/archilyzer.ts compose hub",
"build": "tsx ../common/bin/archilyzer.ts build site",
"build:hub": "tsx ../common/bin/archilyzer.ts build hub",
"start": "serve out",
diff --git a/homepage/content/README.md b/homepage/content/README.md
@@ -6,7 +6,7 @@ exists for whoever edits the docs next.
## Why these are hand-written rather than rendered from the root docs
-The obvious move is to render `SETUP.md`, `DEPLOY_CLOUDFLARE.md` and friends
+The obvious move is to render `SETUP.md`, `PUBLISH.md` and friends
directly, and keep one copy. That was rejected for a decisive reason:
> `SETUP.md` says `git clone <this-repo-url>`. **There is no public repository.**
@@ -31,8 +31,8 @@ its public counterpart needs the same change.
| `docs/what-is-archilyzer.md` | `README.md` | package list, pipeline modes |
| `docs/install.md` | `SETUP.md` | tool versions, env-var table, backend list, the Windows path |
| `docs/operate.md` | `README.md`, `SCHEDULED_SYNC.md` | editor routes, scheduler settings |
-| `docs/deploy-cloudflare.md` | `DEPLOY_CLOUDFLARE.md` | the 25 MB Pages limit, R2 options |
-| `docs/deploy-docker.md` | `DEPLOY_DOCKER.md` | phase structure, settings names |
+| `docs/deploy-cloudflare.md` | `PUBLISH.md` (Cloudflare, R2, cost-abuse) | the 25 MB Pages limit, R2 options |
+| `docs/deploy-docker.md` | `PUBLISH.md` (building every site in containers) | phase structure, settings names |
| `docs/ai-and-mcp.md` | `mcp/README.md` | tool names, `corpus.json` shape |
| `docs/faq.md` | — (written for this site) | claims about cost and hardware |
diff --git a/homepage/package.json b/homepage/package.json
@@ -5,8 +5,8 @@
"type": "module",
"scripts": {
"dev": "next dev --port ${HOMEPAGE_DEV_PORT:-3030}",
- "build:index": "NODE_OPTIONS=--max-old-space-size=8192 tsx ../common/bin/build-index.ts",
- "compose": "tsx ../common/bin/compose-homepage.ts",
+ "build:index": "NODE_OPTIONS=--max-old-space-size=8192 tsx ../common/bin/archilyzer.ts index",
+ "compose": "tsx ../common/bin/archilyzer.ts compose homepage",
"build:data": "pnpm run build:index && pnpm run compose",
"prebuild": "pnpm run build:data",
"build": "next build",
diff --git a/package.json b/package.json
@@ -5,6 +5,7 @@
"license": "MIT",
"type": "module",
"scripts": {
+ "archilyzer": "pnpm --filter yt-dlp-transcript-common exec tsx bin/archilyzer.ts",
"build:index": "pnpm --filter yt-dlp-transcript-common exec tsx bin/archilyzer.ts index",
"sync:tick": "pnpm --filter yt-dlp-transcript-common exec tsx bin/archilyzer.ts sync tick",
"build:export": "pnpm --filter export run build",
diff --git a/plans/release-11.md b/plans/release-11.md
@@ -394,6 +394,275 @@ mounted is fine as a rule, keep `untouched` optional tonight).
the new row; with `persistKept` counting every return, the new persistKept test fails; the other
8 in the two files pass.
+### Slice O6, as shipped — one-core Phase 4 slice 3 (2026-09-28) — checkpoint A
+
+Branch `r11/phase-4-s3` off `main` `2162db92` (no later `main` to merge before the first commit),
+worktree `/home/user/Projects/r11-phase-4-s3`, port block #6, one Opus implementer. The spec is
+[`one-core.md`](one-core.md) "Phase 4", items 2 (the rest) and 3. Checkpoint A is everything but
+the `E2E_` prefix cleanup, which is checkpoint B, after O1–O5 land.
+
+**The CLI's last subcommands.**
+- **`archilyzer doctor [--json]`** (`common/bin/doctor.ts`). One read-only report over what was
+ spread across `paths.ts`, umtool's doctor and `scripts/worktree.mjs`:
+ - workspace: node against next's `>=20.9.0`, the checkout, `node_modules`, the path overrides set;
+ - corpus: channel count, each channel's media through `inspectChannelMedia`, the LMDB index by
+ `stat` only;
+ - settings: `settings.json` present, a JSON object, loads (`settingsFromFile`, new in
+ `settings.ts`: getSettings' body for a named file);
+ - tools: every binary `paths.ts` names, plus each enabled local worker's engine, `parakeet-cli`
+ and model (looked up on PATH, not run);
+ - umtool's report-pipeline table (read from `umtool/lib/tools.mjs` by a runtime import of the
+ file; common does not depend on umtool);
+ - the port block (asked of `scripts/worktree.mjs ports`), each port free or in use by a TCP
+ connect.
+ - A FAIL exits 1. It is something the machine is configured to do and cannot: a settings file
+ that is not a JSON object (every process silently reads defaults), an enabled worker's engine or
+ model missing BESIDE a corpus, yt-dlp/ffmpeg/ffprobe missing beside a corpus, a binary an env
+ override names explicitly, node too old, no `node_modules`. Everything else warns or notes. A
+ clone with no corpus is not broken, and an engine missing there is a warning.
+ - Strictly read-only: no LMDB open, no mkdir, no settings write, no bind. The one process-state
+ change is a `chdir` around umtool's table (it resolves `facecrop.py` from the cwd), restored
+ at once.
+- **The probe exists once.** `common/lib/toolProbe.mjs` is umtool's `probeOne` lifted out (plain
+ ESM, exported); `umtool/lib/tools.mjs` keeps its table and calls it. `probeTools()` output was
+ byte-identical before and after (`o6-umtool-probe-{before,after}.json`, `cmp`).
+- **`archilyzer run <operation> <channel> [ids…] [--lane local|remote]`**
+ (`common/bin/run-operation.ts`) calls `runOperationChannelJob`, the body the /channels
+ buttons, the stage cards and the lane runners call. So a run is a real `backfill-channel` /
+ `digest-channel-*` job: its record and log under `.jobs/`, the same summary line, the channel
+ snapshot flushed at the end, and `runManagedFunction`'s media guard (a refusal before any record
+ exists).
+ - **Runs:** diarization, attribution-diarized, attribution-text, digest (the four registry
+ operations).
+ - **Refused with a sentence read off the descriptor:** sync, metadata-scan, download (the
+ per-platform download queue paces every request to one source inside the editor), and
+ transcription (the editor's worker pool).
+ - **Refused up front:** an unknown operation (with both lists), an unknown channel, an
+ operation switched off in settings, a paused lane (the batch would idle-wait for a resume
+ only the editor can give), and `--lane` off digest. Ids with no data dir are named.
+ - Ctrl-C cancels the job. Exit 0 done, 1 failed or refused, 2 usage, 130 cancelled.
+ - Different from the editor's run, and said in the file's header: it does not see the editor's
+ transcription activity, so the backfill lane's yield-to-transcription does not apply.
+ - To carry ids, `runDigestChannelJob` gains `ids` (as the backfill job has; into
+ `spec.params`), `runOperationChannelJob` passes `ids` to both runners, and the editor's
+ digest replay forwards them through `digestChannelAction`, so a replayed scoped run cannot
+ widen to the channel.
+- **`archilyzer mcp [args…]`** (`common/bin/mcp.ts`) starts the MCP server as `pnpm --filter
+ yt-dlp-transcript-mcp exec tsx src/index.ts` does: mcp's tsx, cwd `mcp/`, env passed through. It
+ is a child process, not an import. Nothing is printed on stdout.
+- **`pnpm archilyzer <command>`** is a new root script for the long form. pnpm prints its `$ …`
+ line on stderr, so `-- pnpm -C "$PWD" archilyzer mcp` is a clean MCP registration (initialize
+ handshake checked through it).
+- **Every bin is a subcommand.** `_cli.ts` gains PASSTHROUGH rows. They are matched on the leading
+ words only, and everything after the path is handed on verbatim.
+ - In-process: `build stats|templates|archives`, `docs env|files`, and `brand media` (with its
+ argv).
+ - As children with their own flags (`_spawnBin.ts`): `duplicates` (with the 8 GB heap its script
+ had), `posts fetch|check`, `diarize backfill`, `digest plan|validate`, `reconcile video-dirs`,
+ `verify transcripts`, `transcribe` (`transform.ts`; deleted with its row in the review
+ round, `9424d357`) and `migrate channel-priority`.
+ - A test fails when a file in `common/bin/` has no row.
+- **package.json.**
+ - Kept: the scripts the publish pipeline runs by name: export's `build:index`, `build:stats`,
+ `build:templates`, `build:archives`, `compose:site` and `compose:hub` (`build.ts`'s steps and
+ docker Phase A), and homepage's `build:index` and `compose`. They keep their names and heaps
+ and now call `archilyzer <row>`. Each was checked with `pnpm run <script> --help` (it prints
+ its row), and `build:data` + `compose:site` + `compose:hub` were smoke-run on a scratch corpus
+ (all exit 0).
+ - Removed: export's `detect:duplicates` (no referrer; it is now `archilyzer duplicates`).
+ - Kept, with the reason: root `build:index` / `sync:tick`. They are one-line CLI aliases, and
+ `sync:tick` is the documented cron line (SCHEDULED_SYNC.md, the editor's sync console copy,
+ SETTINGS.md), so removing it would silently break an operator's crontab.
+ - `docker/build-site.sh` and `publish-site.sh` still parse (`sh -n`) and name `build site`,
+ which exists.
+
+**Ports exist once.** `common/lib/ports.mjs` is the table (name, base, what for), with
+`OFFSET_STEP`, `portsForOffset` and `portFor`.
+- `scripts/worktree.mjs` imports it, and gains the three e2e ports it never offset: `HUB_PORT`,
+ `ORIGIN_B_PORT`, `HUB_A_PORT`.
+- `sync tick`'s default URL and the doctor read it.
+- Two places cannot import a module, and `ports.test.ts` reads them as text and fails on a
+ mismatch or an undeclared port:
+ - a package.json script's `${NAME:-N}` (a shell line);
+ - the `--ports NAME:N` list handed to queue-lock (its syntax is shared with older checkouts).
+- The playwright configs and e2e helpers still spell their fallbacks, held by the same test. B
+ points them at the module, since it edits them anyway (homepage's is O2's tonight).
+- Not in the table: the client default URLs (`mcp/src/fetchClip.ts`, umtool's tag route,
+ `archilyzer-ops.mjs`) and the container's ports.
+
+**Config and docs.**
+- **Env vars, declared once.** `common/lib/envVars.ts` lists every variable the apps' code
+ reads, by audience:
+ - `paths` (exactly what getPaths reads);
+ - `runtime`;
+ - `port` (generated from ports.mjs);
+ - `internal` (set by the pipeline);
+ - `docker` (the `ARCHILYZER_*` set);
+ - `test`.
+- `ENVIRONMENT.md` is generated from it (`archilyzer docs env [--check]`). `envVars.test.ts`
+ holds the list to the code in both directions: every `process.env.X` / `env.X` read under
+ common/, editor/, export/, homepage/, mcp/src and scripts/ is declared, and every entry is
+ still named somewhere outside the list (a mention, not a proven read). *(Review S1: as first
+ shipped, that second test counted `envVars.ts` itself as a mention and could not fail; fixed in
+ `8bea5844`, which also checks every `ARCHILYZER_*` name in docker/, the Dockerfiles and the
+ compose files is declared.)* umtool's own knobs stay in umtool/docs until Phase 5.
+- **PUBLISH.md absorbs DEPLOY_CLOUDFLARE.md and DEPLOY_DOCKER.md.** It covers what gets
+ published, the editor / `pnpm ops` / CLI table, Pages, previews, R2, the cost-abuse defenses,
+ containers and the MCP registration. Both old files are removed. Every link to them now names
+ PUBLISH.md, except in plans/ and the released changelog bullets: README, SETUP, AGENTS,
+ RUNNING_IN_DOCKER, r2-proxy ×3, homepage/content/README.md, and build.ts's R2 refusal plus
+ three comments.
+- **Stale claims corrected.**
+ - README's and SETUP's hand-kept env tables are now a pointer to ENVIRONMENT.md and `doctor`.
+ - SETUP said `pnpm e2e` runs on port 3001; it is 3011.
+ - CONTRIBUTING's "CLI shims" named a `retry-failures.ts` that does not exist; the section is
+ now "The archilyzer CLI".
+ - RUNNING_IN_DOCKER named whisper.cpp as the only engine; it now names parakeet.cpp beside it.
+ - **Settings → Build pipeline still called Docker "a follow-up" that "falls back to a basic
+ build"**, and its option read "(follow-up)". So did `settingsSchema.ts`'s description
+ (SETTINGS.md regenerated) and its BuildMode comment. *(Review S2: the first rewrite
+ overstated the mode as a switch — no build reads `buildPipeline.mode`. In `366553eb` the
+ Settings hints, the schema texts, /sites' toggle note, the Build all paragraph and lane
+ subtitle and the per-site panel note all say the mode is a label and Build all uses
+ containers whenever `docker version` answers; `deploy-page.spec.ts` in step.)*
+ - "Transcode as a stage": no doc claims one any more. The remaining "transcode" rows are
+ ffmpeg's audio extraction.
+ - `PARALLEL_TRANSCRIBE_LIMIT` was already in no doc. The only mention is the released
+ changelog bullet that records its removal.
+- **The two stale hub hints were already fixed.** `SettingsForm.tsx`'s `homepageUrl` hint and
+ `homepage.ts`'s comments name `archilyzer-hub`, since release 7's review fix `bc2d9fb6`.
+ STATE's note is stale; nothing to do.
+
+| sha | what |
+|---|---|
+| `eeeef338` | `common:` `lib/ports.mjs` (+ `ports.test.ts` 4), `worktree.mjs` and `sync-tick` read it; `lib/toolProbe.mjs`, umtool's `tools.mjs` calls it |
+| `9be897c6` | `common:` `lib/envVars.ts` (+ `envVars.test.ts` 6), `bin/env-docs.ts`, `ENVIRONMENT.md`, `docs env` row |
+| `5c16a75c` | `common:` `archilyzer doctor` (+ `doctor.test.ts` 8), `settingsFromFile` |
+| `5da3e9e1` | `common:` the digest channel job's `ids`; the dispatcher passes ids; the editor's digest replay forwards them |
+| `3bfe2ab2` | `common:` `run`, `mcp`, passthrough rows, `_spawnBin.ts`, every bin a row (+ `run-operation.test.ts` 6, `mcp.test.ts` 1, `_cli.test.ts` +3) |
+| `e7de1d96` | `export, homepage:` the pipeline's scripts call the CLI; `detect:duplicates` removed |
+| `a132c213` | `editor, common:` root `pnpm archilyzer`; the Docker build-mode copy (Settings, schema, SETTINGS.md); build.ts names PUBLISH.md |
+| `58fd7d72` | `docs:` PUBLISH.md; the two deploy docs removed; README/SETUP/CONTRIBUTING/AGENTS/RUNNING_IN_DOCKER/r2-proxy/homepage README |
+| `1a65926b` | `common:` the mcp test's exit race and the run tests' timeouts (found by the bite checks) |
+| _this_ | `plans:` this record; the `[Unreleased]` bullet in `editor/CHANGELOG.md` |
+
+**Gates**, all from the worktree root. The logs are `o6-*.log` in the job's scratch dir.
+- **tsc** (`pnpm -r --no-bail --workspace-concurrency=1 exec tsc --noEmit`) was clean four times:
+ 91 s before `eeeef338`, 58 s before `5da3e9e1`, 66 s before `a132c213`, and 95 s on
+ `1a65926b`. The three commit groups in between are prefix-closed subsets of a checked tree: no
+ file in a commit imports one from a later commit. `58fd7d72` is docs only.
+- **common 2,066/2,066** (2,038 + 28: ports 4, envVars 6, doctor 8, run 6, mcp 1, `_cli` 3), 47 s.
+- **editor unit 85/85**, 4 s.
+- **`test:scripts` 174 + 1 skip**, 9 s.
+- **mcp 269/269**, 18 s.
+- **`next build`:**
+ - editor ok, 36 s;
+ - umtool ok, 17 s (`tools.mjs` now imports `yt-dlp-transcript-common/lib/toolProbe.mjs`);
+ - export ok, 27 s (no dangling `export/public` links).
+- **e2e:**
+ - **Editor** (`o6-specs.txt`: `digest.spec.ts backfill.spec.ts jobs-retry.spec.ts`, which
+ also matches `availability-backfill.spec.ts`), no queue wait: **39 passed, 0 failed, 3.2
+ min**.
+ - **umtool** `dashboard.spec.ts` (`SONG_DIR=~/reports/quartering-uh-song/data`, through
+ `wt run`): run 1 was **6 passed, 1 failed, 1.1 min**. The failure was "the projects panel
+ says exactly what /browse's header says": `page.goto` aborted during the cold `next dev`
+ compile (35 s). It does not touch the tool probe. The two probe tests (the tools panel, and
+ `umtool doctor` exiting 1 on a bad path) passed. Rerun: **7 passed, 25.8 s**, after 19 s in
+ the queue.
+ - Nothing else an e2e exercises changed: the editor suite never runs the export build scripts
+ to completion (`deploy-page.spec.ts:130` says so), hence the scratch smoke-run.
+- **The CLI by hand, on scratch corpora only:**
+ - `run diarization chan` exited 0 in 2 s, with the summary line and two job records (the run
+ and the snapshot).
+ - `run sync chan` exited 2 with the sentence (1 since review L6: a refusal, not a usage error).
+ - `mcp --local <dir>` answered initialize, both directly and via `pnpm -C … archilyzer mcp`.
+ - `doctor` on the worktree: no failures. Parakeet workers ×3 (engine, cli, model), every
+ binary, umtool's table (python absent: info), block #6 with every port free.
+- **Nothing ran against the primary's `transcripts/`.**
+
+**They bite** (each mutation in a scratch copy of the tip, `o6-bite.log`):
+- ports: an export dev default drifting to 3009, and an undeclared `NEW_STUB_PORT` in a
+ playwright config, each fail 1.
+- envVars:
+ - an undeclared read in `workerToken.ts` fails 1;
+ - a declared var nothing reads: as first shipped this "failed 1" only because ENVIRONMENT.md
+ went stale — the mention test itself passed (review S1). After `8bea5844`, with
+ ENVIRONMENT.md regenerated, the mention test fails 1;
+ - an undeclared `ARCHILYZER_NEW_KNOB` in `docker/entrypoint.sh` fails 1 (review L5);
+ - a hand-edited ENVIRONMENT.md fails 1;
+ - a new override in `paths.ts` fails 2.
+- doctor:
+ - a `mkdirSync` before returning fails 3 (the tree snapshots);
+ - a non-JSON settings file as a warning fails 1;
+ - a corpus that does not need the media tools fails 1.
+- run:
+ - the dispatcher dropping `ids` fails 1 ("ids scope the run");
+ - external ops not refused fails 2;
+ - ids none of which is on disk, not refused, fails 1 (review L6);
+ - a paused lane not refused gives pass 2, cancelled 4, exit 1. The held job leaves only
+ unref'd poll timers and node cancels the rest.
+- cli: `verify transcripts`' row removed fails 1 (every bin reachable); passthrough disabled
+ fails 1.
+- mcp: the row renamed fails 1. It hung before `1a65926b`, which the bite found.
+
+**Found and left** (after the review round):
+- **For the operator: should Build all / Build & deploy all honour `buildPipeline.mode`?** Today
+ they use containers whenever `docker version` answers and the mode is a label (review S2). The
+ parent ruled copy-only tonight; honouring it (basic → serial host even when docker answers) is
+ a behaviour change for its own slice.
+- **`run` does not see the editor's lanes** (review L1). Beside the editor's lane on the same
+ channel it does the same videos twice — wasted CPU, not damage (atomic writes). Said in the
+ header and the usage line; a `.jobs/` check for a running job of the same kind and slug is not
+ implemented.
+- **Until :3001 restarts, its digest replay predates `ids`** (review L2): Retry on an ids-scoped
+ `archilyzer run digest` job from the old editor widens to the whole channel (money, on the
+ metered lane). Morning runbook: restart before retrying a CLI digest job.
+- **The doctor's default engine binary comes from `app.defaultBin()`** (the global `getPaths()`),
+ not the injected paths (review L7). The same in production; wrong only under injected test
+ paths. Left: a fix would re-derive each app's default beside the registry.
+- **No spec replays a digest job**, so the replay forwarding `ids` is type-checked only.
+- `mcp/README.md`'s and AGENTS.md's `claude mcp add` examples were left alone (O1 owns the first;
+ the prompt said leave both). The new form is in README and PUBLISH.md.
+- `homepage/content/docs/*` are hand-derived copies by design. Only the drift table in
+ `homepage/content/README.md` was updated.
+- Resolved in the review round: AGENTS.md's, the entrypoint comment's and RUNNING_IN_DOCKER's
+ "no workers → zero workers" claim (Q5); the two released changelog links (Q1, retargeted,
+ words unchanged); `archilyzer transcribe` (Q2, deleted); PUBLISH.md's homepage row (Q4).
+- **Optional, not done:** SETUP.md absorbing SCHEDULED_SYNC.md and WORKTREES.md; PLAN.md's phase
+ table becoming a pointer to STATE.md.
+- **Checkpoint B** (the `E2E_` prefix cleanup, and the playwright configs importing ports.mjs) is
+ next, after O1–O5 land on `main`.
+
+**Review fixes (checkpoint A)** — the review (`o6-review.md`) was SHIP AFTER FIXES: no blocker,
+two should-fixes, seven lows; the parent ruled on the questions. First `main` `c1d4790a` (O4 + O3)
+was merged: conflicts only in `editor/CHANGELOG.md` (one `[Unreleased]`: O4's and O3's bullets,
+then O6's) and this file (O4's and O3's records, then this one).
+
+| sha | what |
+|---|---|
+| `fcdaea12` | merge `main` `c1d4790a` |
+| `8bea5844` | S1: the mention test leaves `envVars.ts` out and can fail; L5: docker/'s `ARCHILYZER_*` names must be declared (+1 test) |
+| `366553eb` | S2: the build-mode copy says the mode is a label (Settings, schema + SETTINGS.md, /sites toggle, Build all, per-site panel; `deploy-page.spec.ts`) |
+| `ef84f535` | Q1: the two released changelog links → PUBLISH.md anchors, words unchanged |
+| `9424d357` | Q2: `transform.ts` + the `transcribe` row deleted; L6: usage 2 / refusal 1, all-absent ids refused (+1 test); L1 in the header and usage |
+| `bec01bf9` | Q5: the seeded-worker claim (AGENTS.md, the entrypoint comment, RUNNING_IN_DOCKER); L4: WORKTREES.md → ENVIRONMENT.md#ports, `pnpm dev:*` in ports.mjs, umtool's config comment |
+| `36db91e2` | S2 / Q4 / L3 in PUBLISH.md |
+| _this_ | `plans:` this record |
+
+Gates on the merged tree, from the worktree root (`o6-gateR.log`, `o6-e2e-editorR.log`):
+- **tsc clean** before `8bea5844` (on the merge: 102 s; again with S1+S2: 76 s) and before
+ `9424d357` (33 s, incremental). `ef84f535`, `bec01bf9` and `36db91e2` change no TypeScript
+ but `ports.mjs`' description strings.
+- **common 2,081/2,081** (main's merged tests plus this round's 2), 45 s. **editor unit 85/85**.
+ **`test:scripts` 175 + 1 skip** (O4's +1). **mcp 269/269**.
+- **`docs env --check`, `settings example --check`, `docs files --check`: all exit 0.**
+- `docker/entrypoint.sh` still parses (`sh -n`).
+- **Editor e2e** `deploy-page.spec.ts digest.spec.ts backfill.spec.ts jobs-retry.spec.ts`
+ (`o6-specsR.txt`; `backfill.spec.ts` also matches `availability-backfill.spec.ts`), no queue wait: **43 passed, 0 failed, 2.9 min** (39 before + deploy-page.spec's 4, the build-mode sentence among them).
+- **They bite** (`o6-bite.log`, "review round"): a declared `ZZ_NOBODY_READS_ME` with
+ ENVIRONMENT.md regenerated fails exactly the mention test (1); an undeclared
+ `ARCHILYZER_NEW_KNOB` in the entrypoint fails 1; all-absent ids not refused fails 1.
+
## Rollout
Nothing is rolled out tonight. The morning runbook lists what is owed: the :3001 editor restart,
diff --git a/r2-proxy/README.md b/r2-proxy/README.md
@@ -22,7 +22,7 @@ Then set the editor's **Settings → Archive overflow public URL** to the deploy
`https://<name>.<you>.workers.dev` URL.
Full setup and the alternative custom-domain path are documented in
-[../DEPLOY_CLOUDFLARE.md](../DEPLOY_CLOUDFLARE.md).
+[../PUBLISH.md](../PUBLISH.md#securing-downloads-against-cost-abuse).
## Scripts
diff --git a/r2-proxy/src/index.ts b/r2-proxy/src/index.ts
@@ -7,7 +7,7 @@
// request path straight to the bucket key — there is nothing per-site about it.
// Deploy it ONCE to a free `<name>.workers.dev` subdomain (no custom domain, no
// domain purchase, no WHOIS), then point every site's "Archive overflow public
-// URL" at that single subdomain. See ../DEPLOY_CLOUDFLARE.md.
+// URL" at that single subdomain. See ../PUBLISH.md.
//
// Why a Worker instead of the raw r2.dev URL: it gives us tunable, in-code rate
// limiting (the cost/abuse backstop — Cloudflare's dashboard rate-limit rules
diff --git a/r2-proxy/wrangler.toml b/r2-proxy/wrangler.toml
@@ -4,7 +4,7 @@
# cd r2-proxy && pnpm dlx wrangler deploy
#
# One bucket + one Worker serves every export site (keys are namespaced by site
-# id). See ../DEPLOY_CLOUDFLARE.md for the full setup.
+# id). See ../PUBLISH.md for the full setup.
name = "archilyzer-exports"
main = "src/index.ts"
diff --git a/scripts/worktree.mjs b/scripts/worktree.mjs
@@ -10,24 +10,10 @@ import { execFileSync, spawn } from "node:child_process";
import fs from "node:fs";
import path from "node:path";
import { pathToFileURL } from "node:url";
+import { OFFSET_STEP, portsForOffset } from "../common/lib/ports.mjs";
-// Base ports (offset 0 == main worktree). Mirrors the hardcoded defaults in
-// editor/export package.json scripts and the Playwright configs.
-const PORT_BASES = {
- EDITOR_PORT: 3001, // editor real dev/start
- PORT: 3011, // editor test server + Playwright editor baseURL
- EXPORT_PORT: 3010, // export server launched by editor e2e
- EXPORT_DEV_PORT: 3000, // export real dev
- EXPORT_E2E_PORT: 3020, // export's own Playwright suite
- OLLAMA_STUB_PORT: 11435, // digest-lane stub server in the editor e2e suite
- HOMEPAGE_DEV_PORT: 3030, // homepage (hub) real dev
- HOMEPAGE_PORT: 3031, // homepage static `serve out` (start:homepage)
- HOMEPAGE_E2E_PORT: 3040, // homepage's own Playwright suite
- UMTOOL_PORT: 3050, // um-clip triage tool real dev
- UMTOOL_E2E_PORT: 3051, // umtool's own Playwright suite
- EDITOR_STUB_PORT: 3052, // stub editor the umtool e2e suite fetches clips from
-};
-const OFFSET_STEP = 100;
+// The port table — names, bases and the offset step — is common/lib/ports.mjs,
+// the one copy. This file only decides WHICH offset a checkout gets.
function git(args, opts = {}) {
// With stdio:"inherit" execFileSync returns null (output not captured).
@@ -87,16 +73,6 @@ function offsetForIndex(index) {
return index * OFFSET_STEP;
}
-// Compute the port env map for a given offset. Does not consult process.env.
-function portsForOffset(offset) {
- const env = {};
- for (const [key, base] of Object.entries(PORT_BASES)) {
- env[key] = String(base + offset);
- }
- env.PLAYWRIGHT_BASE_URL = `http://localhost:${PORT_BASES.PORT + offset}`;
- return env;
-}
-
// Index of the worktree containing `dir` (default: cwd) in the worktree list.
function indexForDir(dir = process.cwd()) {
const trees = listWorktrees();
diff --git a/umtool/lib/tools.mjs b/umtool/lib/tools.mjs
@@ -1,21 +1,20 @@
// Which external tools are on this machine, and which pipeline needs which.
//
// Plain ESM with no app imports so `umtool doctor` runs from a terminal, and so
-// the app's /api/doctor and the CLI cannot disagree about what was probed.
+// the app's /api/doctor and the CLI cannot disagree about what was probed. The
+// PROBE itself (run the version flag, read the version, ENOENT = absent) is
+// common/lib/toolProbe.mjs, shared with `archilyzer doctor`, which reports this
+// table too; only the table lives here.
//
// THE ONE RULE: this is never run during a render, and never from a page
// render. Probing seven binaries is ~100 ms of fork/exec, which is nothing once
// and a tax on every request if it leaks into a page. So the app keeps a cached
// report with a TTL and the dashboard shows the cache or "not checked" plus a
// button; only the button and the CLI probe.
-import { execFile } from "node:child_process";
-import { stat } from "node:fs/promises";
import path from "node:path";
-import { promisify } from "node:util";
+import { probeTool } from "yt-dlp-transcript-common/lib/toolProbe.mjs";
import { SONG_SCRATCH } from "./paths.mjs";
-const execFileP = promisify(execFile);
-
/** Mirrors lib/faces.ts, which re-exports these so the two cannot drift. */
export const facedetPython = () =>
process.env.FACEDET_PYTHON ?? path.join(SONG_SCRATCH, "facedet", "bin", "python");
@@ -43,52 +42,12 @@ export const TOOLS = () => [
{ id: "facecrop.py", file: facecropPy(), neededBy: ["faces"], required: false },
];
-const firstLine = (s) => String(s ?? "").split("\n").find((l) => l.trim()) ?? "";
-/** `ffmpeg version 7.1.1 …` -> `7.1.1`; `2025.08.11` -> itself. */
-const versionOf = (text) => {
- const line = firstLine(text);
- const m = line.match(/(\d+\.\d+(?:\.\d+)*(?:[-_.][A-Za-z0-9]+)*)/);
- return m ? m[1] : line.slice(0, 60) || null;
-};
-
-async function probeOne(t) {
- const base = { id: t.id, bin: t.file ?? t.bin, neededBy: t.neededBy, required: !!t.required };
- if (t.file) {
- const ok = await stat(t.file).then((s) => s.isFile(), () => false);
- return { ...base, present: ok, version: null, error: ok ? null : `${t.file} is missing` };
- }
- const run = async (bin) => {
- try {
- const { stdout, stderr } = await execFileP(bin, t.args, { timeout: 5000, maxBuffer: 1 << 20 });
- return { present: true, version: versionOf(stdout || stderr), error: null };
- } catch (err) {
- // ENOENT is "not on this machine". Any other exit means the binary RAN --
- // an unusual version flag, say -- which is presence, honestly reported.
- if (err?.code === "ENOENT") return { present: false, version: null, error: `${bin}: not found` };
- const said = versionOf(err?.stdout || err?.stderr);
- return { present: true, version: said || null, error: said ? null : `${bin} exited ${err?.code ?? "?"} on ${t.args.join(" ")}` };
- }
- };
- let r = await run(t.bin);
- if (!r.present && t.fallback) {
- const f = await run(t.fallback);
- if (f.present) {
- r = {
- present: false,
- version: f.version,
- error: `only \`${t.fallback}\` is installed (ImageMagick 6); the pipeline calls \`${t.bin}\``,
- };
- }
- }
- return { ...base, ...r };
-}
-
/**
* Run every tool's version flag. ~100 ms in total, in parallel.
* @returns {Promise<{ checkedAt: number, tools: Array<{id:string,bin:string,present:boolean,version:string|null,error:string|null,neededBy:string[],required:boolean}>, ok: boolean }>}
*/
export async function probeTools() {
- const tools = await Promise.all(TOOLS().map(probeOne));
+ const tools = await Promise.all(TOOLS().map((t) => probeTool(t)));
return {
checkedAt: Date.now(),
tools,
diff --git a/umtool/playwright.config.ts b/umtool/playwright.config.ts
@@ -6,8 +6,8 @@ const PORT = Number(process.env.UMTOOL_E2E_PORT ?? 3051);
// The editor stub's port. Named, because the queue lock's port PREFLIGHT only
// checks the ports it is given: a bare PORT+1 was outside it, so a second
// checkout's stub could already hold the port and this run would drive it. It
-// is in scripts/worktree.mjs PORT_BASES (so a worktree gets its own) and in
-// package.json's --ports spec (so the preflight sees it).
+// is in common/lib/ports.mjs (so scripts/worktree.mjs gives a worktree its own)
+// and in package.json's --ports spec (so the preflight sees it).
const STUB_PORT = Number(process.env.EDITOR_STUB_PORT ?? PORT + 1);
// The package is "type": "module", so there is no __dirname here.
const FIXTURE = path.join(path.dirname(fileURLToPath(import.meta.url)), ".e2e-song");