Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit c5dcbf938e801884375de2f90ab117cc3ad89fce
parent 02778c25f480a3ae837cab28dd5ac32b1a38f299
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu, 20 Aug 2026 01:19:15 -0400

Merge branch 'feat/umtool-projects'

Brings the umtool project bench and the report-to-video pipeline onto main,
including the cue resolver that lets both run against a published archive with
no local corpus.

The two sides are disjoint by construction — umtool/, scripts/report-to-video/
and four workspace files on one side, common/ and editor/ on the other — so this
merged without a conflict.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

Diffstat:
M.gitignore | 9++++++++-
Mpackage.json | 2+-
Mpnpm-lock.yaml | 17+++++++++++++++++
Mpnpm-workspace.yaml | 1+
Ascripts/report-to-video/README.md | 786+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Ascripts/report-to-video/build-video.mjs | 1838+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Ascripts/report-to-video/check-availability.mjs | 182+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Ascripts/report-to-video/compose-chrome.mjs | 599+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Ascripts/report-to-video/cues.mjs | 270+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Ascripts/report-to-video/cues.test.mjs | 297+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Ascripts/report-to-video/ledger-totals.mjs | 540+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Ascripts/report-to-video/ledger-totals.test.mjs | 705+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Ascripts/report-to-video/package.json | 25+++++++++++++++++++++++++
Ascripts/report-to-video/render-cards.mjs | 1530+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Ascripts/report-to-video/resolve-windows.mjs | 241+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Ascripts/report-to-video/verify-build.mjs | 110+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/app/api/browse/decisions/route.ts | 3++-
Mumtool/app/api/browse/poster/route.ts | 30++++++++++++++++++++++++++++++
Aumtool/app/api/browse/projects/route.ts | 34++++++++++++++++++++++++++++++++++
Mumtool/app/api/mix/files/route.ts | 23++++++++++++++++++-----
Aumtool/app/api/report/build/route.ts | 141+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/app/api/report/claim/route.ts | 65+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/app/api/report/clip/route.ts | 75+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/app/api/report/cues/route.ts | 26++++++++++++++++++++++++++
Aumtool/app/api/report/fetch/route.ts | 49+++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/app/api/report/peaks/route.ts | 64++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/app/api/report/raw/route.ts | 95+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/app/api/report/window/route.ts | 52++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/app/browse/[...path]/page.tsx | 88+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Dumtool/app/browse/[song]/[cut]/page.tsx | 122-------------------------------------------------------------------------------
Dumtool/app/browse/[song]/page.tsx | 254-------------------------------------------------------------------------------
Aumtool/app/browse/at/page.tsx | 53+++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/app/browse/decisions/page.tsx | 29+++++++++++++++--------------
Mumtool/app/browse/page.tsx | 278+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------------------
Mumtool/app/globals.css | 55++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mumtool/app/mix/page.tsx | 49++++++++++++++++++++++++++++++++++++++++++++-----
Aumtool/bin/umtool.mjs | 594+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components.json | 21+++++++++++++++++++++
Mumtool/components/MixBench.tsx | 119+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Aumtool/components/projects/ClaimBench.tsx | 481+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/projects/ClaimBenchPage.tsx | 101+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/projects/ClipBench.tsx | 549+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/projects/ClipBenchPage.tsx | 100+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/projects/CutPage.tsx | 125+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/projects/ProjectGrid.tsx | 104+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/projects/ProjectView.tsx | 70++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/projects/ReportBuildChain.tsx | 308+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/projects/ReportProject.tsx | 338+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/projects/SongProject.tsx | 255+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/projects/SweepProject.tsx | 73+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/ui/badge.tsx | 49+++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/ui/button.tsx | 42++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/README.md | 91+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/authoring.md | 125+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/browse.md | 88+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/build.md | 109+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/claim-bench.md | 119+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/cli.md | 89+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/clip-bench.md | 93+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/decisions.md | 106+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/e2e.md | 81+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/folders.md | 90+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/index.md | 75+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/mix-from-a-project.md | 84+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/projects.md | 115+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/quirks.md | 227+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/docs/report-video.md | 263+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/e2e/browse.spec.ts | 10++++++----
Aumtool/e2e/build.spec.ts | 202+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/e2e/claim-bench.spec.ts | 182+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/e2e/clip-bench.spec.ts | 209+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/e2e/deck.spec.ts | 6++++--
Mumtool/e2e/fixtures/make-fixture.mjs | 407++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mumtool/e2e/mix.spec.ts | 79+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/e2e/projects.spec.ts | 495+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/lib/browse.ts | 26+++++++++++---------------
Mumtool/lib/decisions.ts | 46++++++++++++++++++++++++++--------------------
Mumtool/lib/jobs.ts | 122+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------
Mumtool/lib/media.ts | 75+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Aumtool/lib/mix-preset.ts | 91+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/lib/mix.ts | 11++++++++---
Aumtool/lib/paths.mjs | 122+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/lib/paths.ts | 108++++++++++++++++++++++++++-----------------------------------------------------
Aumtool/lib/project-types.ts | 130+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/projects.ts | 436+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/projects/core.mjs | 137+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/projects/index-db.mjs | 223+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/projects/kinds.mjs | 193+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/projects/report.mjs | 800+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/projects/song-ids.mjs | 35+++++++++++++++++++++++++++++++++++
Aumtool/lib/projects/song.mjs | 101+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/projects/sweep.mjs | 49+++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/projects/walk.mjs | 195+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/report/driver.mjs | 145+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/report/manifest.mjs | 270+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/report/serve.mjs | 68++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/lib/trim.ts | 18+++++++++++++++++-
Aumtool/lib/utils.ts | 14++++++++++++++
Mumtool/next.config.ts | 7+++++++
Mumtool/package.json | 8++++++++
Mumtool/playwright.config.ts | 8++++++++
101 files changed, 18230 insertions(+), 619 deletions(-)

diff --git a/.gitignore b/.gitignore @@ -137,5 +137,12 @@ yarn-error.log* # one project). umtool/.e2e-song/ umtool/.next/ -umtool/.next-e2e/ +# Any alternate dist dir, not just the e2e one. +# +# Tailwind v4 auto-detects its sources and honours .gitignore -- so a dist dir +# that is NOT ignored gets scanned, its binary turbopack cache yields candidate +# class names like `p-[var(-sM0or-Z)]`, and the generated stylesheet fails to +# parse. Every page then 500s on a CSS error with nothing wrong in the CSS. +# Cost an e2e run's worth of confusing red to find. +umtool/.next-*/ umtool/test-results/ diff --git a/package.json b/package.json @@ -21,7 +21,7 @@ "e2e": "node scripts/worktree.mjs run -- pnpm --filter editor run e2e", "wt": "node scripts/worktree.mjs", "e2e:sharded": "node scripts/run-sharded-e2e.mjs", - "test:scripts": "node --test scripts/*.test.mjs", + "test:scripts": "node --test scripts/*.test.mjs scripts/report-to-video/*.test.mjs", "lint": "pnpm --filter export run lint" }, "devDependencies": { diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml @@ -302,8 +302,19 @@ importers: specifier: ^5.6.0 version: 5.9.3 + scripts/report-to-video: {} + umtool: dependencies: + class-variance-authority: + specifier: ^0.7.1 + version: 0.7.1 + clsx: + specifier: ^2.1.1 + version: 2.1.1 + lmdb: + specifier: ^3.5.4 + version: 3.5.4 next: specifier: 16.2.3 version: 16.2.3(@babel/core@7.29.0)(@playwright/test@1.59.1)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) @@ -313,6 +324,12 @@ importers: react-dom: specifier: 19.2.4 version: 19.2.4(react@19.2.4) + report-to-video: + specifier: workspace:* + version: link:../scripts/report-to-video + tailwind-merge: + specifier: ^3.6.0 + version: 3.6.0 yt-dlp-transcript-common: specifier: workspace:* version: link:../common diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml @@ -4,6 +4,7 @@ packages: - export - homepage - mcp + - scripts/report-to-video - umtool allowBuilds: diff --git a/scripts/report-to-video/README.md b/scripts/report-to-video/README.md @@ -0,0 +1,786 @@ +# report-to-video + +Turns a cited sweep report into a video: the clips run in chronological order and +let the source speak for itself, with thin chrome carrying the citation and a +timeline of where you are. A video rendering of the reports we already write. + +Two scripts and a manifest: + +| file | lifetime | what it is | +|---|---|---| +| `build-video.mjs` | stable | manifest → mp4. Fetches clips, snaps cuts to silence, letterboxes them into the chrome, crossfades. | +| `render-cards.mjs` | stable | draws the timeline footer and marker, plus optional card stills. Imported by `build-video.mjs`. | +| `resolve-windows.mjs` | stable | widens clip windows from cue spans to whole sentences. Run once after authoring a manifest. | +| `<report>/video.manifest.json` | per report | the edit decision list. **This is the regeneration source of truth**, not the report. | + +``` +node scripts/report-to-video/resolve-windows.mjs ~/reports/<slug>/video.manifest.json --write +node scripts/report-to-video/build-video.mjs ~/reports/<slug>/video.manifest.json +``` + +Output lands in `<report dir>/out/`: `cards/`, `clips-raw/`, `segments/`, and the +finished `<slug>.mp4`. First worked example: `~/reports/ferret-rescue/`. + +## Why there is a manifest at all + +**A sweep report does not contain enough information to cut a video from.** Its +citations carry a single start second and nothing else — `momentUrl()` +(`common/lib/momentUrl.ts:106`) takes one `seconds` and floors it, and the MCP +`Snippet` type (`mcp/src/search.ts:41`) has no `end` field. There is no clip +length anywhere in a report. + +The end times do exist, they are just never rendered: every cue in +`transcripts/channels/<slug>/data/<id>/transcript.cues.json` is `{start, end, text}`. +So the manifest is built by matching each quote back to its covering cues and +recording the real window. That is also what handles ellipsis-joined citations — +a report quote like `"…" … "…"` is often two separate cue spans presented as one. + +The manifest additionally pins the provenance (share link, corpus handle, match +counts, the narrowing queries) so a rebuild months later is reproducible and the +video's own claims about its coverage can be checked. + +## Where a clip actually gets cut + +Three stages, because a cue span is the wrong answer twice over. + +1. **Cue span** — the raw window covering the quote, from `transcript.cues.json`. +2. **Sentence widening** (`resolve-windows.mjs`) — walk outward to the nearest cue + ending in `.`, `?` or `!`. A cue boundary is where the *caption line wrapped*, + so cutting there drops the run-up that makes a quote intelligible. + + Two asymmetries matter. The **start** takes a lead-in only if a real sentence + opening sits within `--max-lead` (8 s); otherwise it takes none, because a + half-sentence run-up is the irrelevant context you were trying to avoid, not + context. The **end** is never clamped to a budget — stopping partway through a + sentence is the exact mid-thought ending this removes — so `--max-tail` (12 s) + only bounds how far it looks before giving up and using the cue end. + + **Some uploads have no punctuation at all.** Older ASR in this corpus emits + unpunctuated cue text for whole videos, and sentence detection then has nothing + to find: widening degrades to the raw cue span at both ends. That is not a + silent failure you can ignore — it is what produced a clip opening mid-thought + on "higher than they can afford and because", and what let another clip's tail + run through its neighbour. For those videos, pick the window by reading the + cues and set it by hand; the de-overlap and silence passes still apply. + + **`lockStart` / `lockEnd` pin an edge** to exactly what the manifest says, and + neither widening nor de-overlap will move it. Reach for it when the utterance's + real trailing pause does not line up with its last cue's end — the snap picks + the *nearest* silence, and in speech over game audio the nearest one is often a + gap between syllables rather than the pause at the end of the thought. +3. **Silence snapping** (`build-video.mjs`) — a sentence boundary in the + *transcript* still isn't a boundary in the *audio*, so clips clip words in + half. Fetch `fetchPad` seconds wider than needed, run `silencedetect` over the + result, and move each cut to the nearest silence within `snapWindow`. Starts + land on a silence's END (just before speech resumes), ends on a silence's START + (just after speech stops). No silence close enough → keep the exact point; a + tight cut beats a cut in the wrong place. The build logs `start✓ end✓` per clip + so you can see which snapped. + + **The silence threshold is relative, and it has to be.** These are game + streams: the gaps between words are full of game audio and music — quiet, but + nowhere near silent. A fixed absolute threshold sits below the noise floor and + finds nothing. On one measured clip: mean volume −21 dB, **0** silences at + −32 dB, **25** at −26 dB. So each clip is measured with `volumedetect` first + and the threshold set `silenceRelDb` (default 6) below its own mean. If a rebuild + suddenly reports mostly `start– end–`, this is the knob. + +The trim happens during the burn-in encode, so snapping costs nothing extra. + +4. **De-overlap** (`resolve-windows.mjs`) — widening is per-clip and blind to its + neighbours, so two clips cut from the *same* video can end up overlapping, and + the overlap plays as the same footage twice. Any earlier clip whose tail runs + into a later clip's start is trimmed back to that start. This is not a rare + edge case: it fired on the first report, where a 2024 upload has **no + punctuation at all** in the relevant stretch, so sentence detection found + nothing and the tail ran the full budget straight through the next clip. + +## Regenerating and changing a video + +Everything is cached by content, so iteration is cheap: + +- **Reorder, drop or add clips** — edit `timeline`, re-run. Cached clips are not + refetched, so a re-cut costs an encode, not a download. +- **Change a clip's window** — edit `start`/`end`, re-run. A raw file's window is + in its name, so the cache is content-addressed; and since a request is satisfied + by any cached file that **contains** it, a nudge inside the existing pad costs + nothing at all. Only a window that escapes every cached file downloads again. +- **Preview one entry** — `--only <id>` builds a single segment and stops. +- **Work offline** — `--skip-fetch` fails loudly instead of downloading, so you + can confirm you are working entirely from cache. +- **Fetch one clip, wide** — `--fetch-only <id> --pad 20` puts a generous window + in the cache without building anything. This is what the umtool clip bench runs, + and containing-window reuse is what makes that fetch double as the build's cache. +- **Force a refetch** — delete `out/clips-raw/`, or pass `--no-reuse` to require an + exact-window file. +- **Iterate on the rail** — `--rail-only` re-runs only the rail chain over a cached + `out/<slug>.prerail.mp4`; `--preview <start> <dur>` does the same over a window. + `--no-rail` builds the cut without one. See + [The claim rail](#the-claim-rail-renderrail). + +Containing-window reuse was retrofitted, and the waste it removes is measurable: +`ferret-rescue/out/clips-raw` holds **31 files for 10 clips** because every window +edit before this downloaded the same material again — one source is there four +times over overlapping windows. The tightest containing file wins, not the widest, +because silence detection decodes the whole file and a 40 s file costs more than +the 24 s one that would also have done. + +After changing any window, re-run `resolve-windows.mjs --write` before building. +It is a **fixed point**: running it on an already-resolved manifest reports +`0 window(s) changed` and rewrites nothing. That property is load-bearing and was +not free — the manifest stores times rounded to 2 dp, so a value read back can sit +a hair below the cue end it came from, which lands the end lookup on the previous +cue and runs the search on to the *next* sentence. Left alone, every re-run grew +the same clip. Hence `EPS` in the end lookup and the 0.05 s deadband on applying a +change. + +## Driven from umtool + +The three CLIs are the source of truth and stay usable on their own; umtool drives +them rather than reimplementing them, so the UI and the terminal can never disagree +about a window, a format string or the Rumble retry. Three additions exist for that: + +- **`--progress ndjson`** — one JSON object per line instead of prose: + `start`, `card`, `clip`, `fetch`, `snap`, `segment`, `entry-failed`, `concat`, + `chapters`, `note`, `done`, `error`. The event set is exactly what was already + being printed; making it a format switch is what stops a wording change from + breaking the driver. +- **`--continue-on-error`** — record a failed entry and carry on. A dead source at + clip 14 of 19 otherwise throws away thirteen fetches already paid for. The run + still **refuses to concatenate** and exits non-zero: a finished file that quietly + lost a citation is worse than no file. +- **`buildVideo({manifestPath, opts, out, only, fetchOnly})`** is exported, and + `widen()` from `resolve-windows.mjs` already was. umtool imports `widen()` so the + bench's "extend to sentence end" is the CLI's own function, and **spawns** the + build — a 40-minute chain of yt-dlp and ffmpeg inside a request handler has no + cancellation story. + +`check-availability.mjs` is the fourth CLI and belongs at the *front* of a build: + +``` +node scripts/report-to-video/check-availability.mjs <manifest.json> +``` + +It runs `yt-dlp --simulate` once per distinct `(channel, video)` — no bytes +downloaded — classifies each failure (`deleted`, `private`, `restricted`, +`members-only`, `geo-blocked`, `no-cues`, `maybe_missing`), and writes +`out/availability.json` with a timestamp. This is the one fact about a manifest +that goes stale in *both* directions: a source can die after the manifest is +written, and a source annotated "gone" can come back. `no-cues` is called out +separately because it is a different bug — usually the Rumble two-ids trap, where +the manifest names the MCP video id while the cue file lives under the URL slug. + +## Manifest shape + +`timeline` is an ordered list; entries are `card` or `clip`. + +```jsonc +{ "type": "card", "id": "ch3", "style": "chapter", "seconds": 4.0, + "kicker": "March – November 2025", "heading": "Then: the county", + "sub": "Six months for the first approval" } + +{ "type": "clip", "id": "c04", "video": "uyz1_FIqIEk", + "start": 32989.56, "end": 32994.19, // cue-accurate, from transcript.cues.json + "cite": 32989, // the second shown in the attribution line + "quote": "The pre-application screening was approved by the county, dude." } +``` + +Two per-clip fields exist for compilations that span sources or need a hand-cut +window: + +- **`channel`** — the archived channel this clip's cue file lives under, overriding + `provenance.channelSlug`. A compilation about one person routinely spans several + mirror channels (`HasanAbiVODs` / `…VODs3` / `…VODsbackup`), and cue files are + keyed by channel, so a single manifest-wide slug cannot find them all. +- **`lock`** — exempt this clip from `resolve-windows`. Sentence-widening exists to + stop clips ending mid-thought, but that is exactly wrong when the author has + deliberately cut a quote short: a single cue often holds a whole paragraph, so + trimming to "I hate this country so much sometimes" and dropping the rest of the + sentence is an editorial decision that widening would silently undo. `lock` also + handles the reverse case — a clip whose lead-in would drag in seconds of some + *other* audio (a news package playing before the speaker starts). + +Clips also carry `section` and (auto-set) `sectionEnter`. Card styles — `title`, +`timeline`, `status`, `bullets`, `sources` — still work, but the ferret-rescue cut +uses none of them. `render` holds resolution, fps, fonts, palette and the knobs +(`fetchPad`, `snapWindow`, `silenceRelDb`, `transition`, `slideSeconds`, +`headerHeight`, `footerHeight`, and the optional `rail`); `provenance` holds the +sweep's scope and counts. Two further entry `type`s, `scroll` and `chart`, close a +cut off a top-level `ledger[]` — see [The claim rail](#the-claim-rail-renderrail). + +## Chrome, not cards + +**The ferret-rescue cut has no cards at all** — no title, no chapter breaks, no +closing slate. It is a cited timeline and nothing else: the clips run in +chronological order and the source material carries the argument. Cards remain +supported for reports that want them, but the default posture is that anything +drawn is an interruption which has to earn its place. + +Nothing is drawn *over* the picture either. The video is **letterboxed between** +thin chrome rather than overlaid by it: + +- **Header (`headerHeight`, 56 px).** The citation only — cleaned title · upload + date · timestamp. No quote: the clip is already saying it, and burning in a + transcription of speech you can hear is noise. +- **Footer (`footerHeight`, 100 px).** The timeline: one node per milestone, each + with a label and a month/year stamp beneath it. Drawn once by + `renderFooterAssets()`. +- **The marker slides.** On the first clip of each section (`sectionEnter`, set + automatically), the fill bar and the amber marker animate from the previous node + to the current one over `slideSeconds`. Everywhere else they hold position. The + motion is ffmpeg expressions on `crop`/`overlay`, so it costs nothing beyond the + encode that was happening anyway. + + > **This used to be half true.** Until 2026-08-19 only the marker moved. The + > fill bar was a `drawbox` whose width was `if(lt(t,0.9),…)` — but **`drawbox` + > has no time variable**: its `t` is the box *thickness*, and with `t=fill` + > that is effectively `INT_MAX`, so the comparison was always false and the + > width expression collapsed to its end value. (`drawbox=w='t*10':h=8:t=2` and + > `drawbox=w=20:h=8:t=2` produce an identical YAVG.) + > + > It was worse than "always full", because **`drawbox` reads `w=0` as *the + > input width***. Section 0's fill is 0 px, so every clip in the first section + > drew the bar across the **whole frame** — the progress track read 100 % + > complete on the opening clip of every cut that has a footer. Measured on + > `quartering-flagging-takedowns/n02`: 1708 accent px on the track row before, + > 0 after (the correct value), with the amber marker parked on node 0. + > + > It is now a `_bar.png` strip of `2·trackLen × 3` — accent on the left half, + > transparent on the right — translated under a fixed-width `crop`, which + > *does* evaluate `x` per frame. Verified on `ferret-rescue/c07`: 608 → 698 → + > 798 → 912 px across t = 0.0 … 0.9 s, then parked. + +Clips carry `section` (index into `timelineNodes`); nodes supply `label` and +`date`. Ordering clips chronologically is the author's job — the manifest plays in +the order it is written. + +A consequence worth knowing: 16:9 source into the reduced height leaves narrow +pillarbox bars. That is the price of never covering the picture, and it is why the +chrome is kept as thin as it is. + +**Both bands are optional, and turning them off is a real setting, not a hack.** +`footerHeight: 0` (or an empty `timelineNodes`) drops the timeline; `headerHeight: 0` +drops the citation line. With both at zero the clips fill the whole frame and +nothing is drawn at all — the filtergraph loses the overlays rather than compositing +invisible ones, and `renderFooterAssets` returns early instead of drawing PNGs +nobody uses. Two reasons this comes up: + +- A timeline footer only means something if the clips *are* a progression through + time. A cut ordered by argument rather than by date should not draw one. +- The header prints the **upload date of the archived copy**, which for a VOD-mirror + channel is often years after the stream (a Nov 2019 stream re-uploaded in Apr 2023 + reads `… November 6, 2019 … · 2023-04-06`). The title usually carries the true + date, so nothing is false, but on a cut spanning many re-uploads it reads badly. + +The `hasan-hate-america` cut runs with both off. Restoring them is a two-value edit. + +Note: commas inside an ffmpeg filter expression have to survive filtergraph +parsing — wrapping the expression in single quotes is what protects them. + +- **Segments crossfade** (`transition`, default 0.5 s). This forces a full + re-encode of the timeline via `xfade`/`acrossfade` — the concat demuxer can only + stream-copy hard cuts. Pass `--no-xfade` for a fast hard-cut build while + iterating; the last pass can add the transitions back. + +## Two cuts from one manifest (`--variant`) + +A sweep finds more claims than a cut can show footage for. There are two +defensible answers to that and they make different videos, so the manifest +describes both and one filter picks between them. + +| variant | output | ledger | what the viewer sees | +|---|---|---|---| +| `sourced` (default) | `out/<slug>.mp4` | claims with a clip | every row on screen has footage behind it | +| `full` | `out/<slug>-full.mp4` | every claim | the unclipped ones are stacked onto `ledger` cards | + +`selectVariant(manifest, variant)` runs **immediately after the manifest is read** +and is the whole mechanism. Three rules, in this order: + +1. a timeline entry tagged `"variant": "full"` survives only in that variant; +2. a claim survives only if the entry its `entryId` names survived — which is what + makes `sourced` a sourced-only ledger, because in `full` every claim is pinned, + to a clip or to a `ledger` card; +3. `card.variants[<name>]` field overrides are merged in and the key dropped. + +Nothing downstream learns about variants. `ledgerTotals`, the rail, the chart +band, `scheduleClaims`, the chapters and the closing cards already take the +ledger and the timeline as inputs. + +**Rule 3 exists because copy can be false in one cut.** A title card saying "48 +dated claims" is a lie in a cut that shows nineteen, and the closing sources card +says two claims stayed ambiguous — both of which happen to be unclipped, so in +`sourced` there are none. + +### Output layout + +``` +out/ + clips-raw/ SHARED — the only expensive thing in a build + availability.json SHARED — a fact about the manifest, not about a cut + <slug>.mp4 sourced + <slug>-full.mp4 full + sourced/{cards,segments,qr,chrome,schedule.json} + full/{cards,segments,qr,chrome,schedule.json} +``` + +`clips-raw` is shared deliberately: `sourced`'s clips are a subset of `full`'s, so +no clip is ever fetched twice. `sourced` writes `out/<slug>.mp4` because that is +the path umtool's build probe already looks for. + +`compose-chrome.mjs` and `verify-build.mjs` both take `--variant` for the same +reason: the band plots the ledger the cut carries, and verifying the whole +manifest against one variant's file would report a missing chapter for every +entry the other cut has. + +### The `ledger` entry type + +`full`'s answer to the claims no clip covers. One card per **run of consecutive +unclipped claims**, so 29 claims cost 13 cards and about 66 seconds. + +```jsonc +{ "type": "ledger", "id": "L10", "variant": "full", "seconds": 4.8, + "kicker": "November 2024", + "heading": "Ten and eight, named separately, in one breath", + "sub": "…", // optional + "claims": ["c12", "m14"] } // ledger ids, in ledger order +``` + +Each row draws `date · scope pill · why it is not footage · the quote · his +figure`, plus a right-hand **arithmetic column**: `media / coffee / publica` as +they stand, the implied total, the layer this claim just moved lit, and its delta. +The arithmetic is **read from `ledgerTotals`, never recomputed** — one walk, or +the card and the band disagree about the same sum. + +Rows **reveal in sequence** behind an opaque `pal.bg` rectangle walking down the +card: the rail curtain's device, exact because the card ground is flat. +`seconds` is derived (`2.2 + 1.3·rows`) rather than authored, because the rail +pins to the same clock — see below. + +**"Why it is not footage" comes from a probe, not from a hand-written kicker.** +`check-availability.mjs` now probes every **ledger** source as well as every clip +source, so a row says `source deleted`, `source unreachable` or `not clipped` +because `yt-dlp --simulate` said so on a recorded date. Several unclipped claims +are cut from videos this cut clips elsewhere, i.e. demonstrably live; saying "the +upload is gone" about one of those is the kind of error that discredits the whole +compilation. + +**Pins gain a within-segment offset.** A claim on a `ledger` card is pinned to its +own row's reveal (`ledgerRevealAt(r)`), not to the segment's mid-dissolve. +Otherwise four rail rows land on one frame, and the pin-order guard's strict +monotonicity breaks for no reason. In `full` this pins the rail almost exactly: +every claim has a segment, so `scheduleClaims` interpolates almost nothing. + +**`status` is retired.** It existed to quote a claim whose source had gone; a +`ledger` card does the same thing better, alongside the arithmetic the claim moves +and with the reason coming from the probe. + +## The claim rail (`render.rail`) + +A cut whose whole point is *which company a number was about* has a problem: the +dates and the figures are **spoken**, and shown only in the header's citation +line. A viewer can hear "nearly ten" three times without ever seeing that the +three refer to three different payrolls. + +`render.rail` adds a persistent **vertical ledger down the right edge**. It +appends one row per claim as the video runs, keeps a live per-company tally +beside it, and lists **every claim the sweep found** — not just the ones with a +clip behind them. Claims with no clip are dimmed (muted ink, hollow dot) and pass +with no audio; they are what stops the rail from implying the cut is the corpus. + +It is **entirely opt-in**. With no `render.rail` key the filtergraph is the one +that was there before, and output is byte-for-byte unchanged. + +```jsonc +"render": { + "rail": { + "width": 420, // picture shrinks to width - 420 + "rowHeight": 46, // one claim row; window height is a whole multiple + "tallyRowHeight": 44, + "tallyTop": 96, // y of the tally block inside the rail column + "pad": 22, + "slide": 0.55, // seconds per row-change ease + "rule": "#2A322F", + "tracks": [ // one per company; ORDER is the rail/legend order + { "key": "media", "label": "The Quartering · media", "color": "#22AB83" } + ] + } +} +``` + +and a top-level `ledger[]`, in **playback order** — each track's claims contiguous +and date-sorted within the track: + +```jsonc +{ "id": "m02", "date": "2022-09-17", "company": "media", + "value": 4, // null for a qualitative claim; the tally ignores those + "display": "4", // the badge + "label": "counts them out: one, two, three, four", + "quote": "…", "src": "…", + "hedged": false, // a hedge word, not a figure -> hollow dot on the chart + "plotted": true, // appears in the step chart + "entryId": "a01" } // pins the row to that timeline entry's segment +``` + +Pinned entries carry a `claim` back-reference so the link reads both ways. + +### How it is put together + +Everything that moves is **one tall strip walked by a fixed-size `crop`**, not a +per-state still, because swapping stills can only cut and a crop can ease. Five +strips, all bounded `-loop 1 -framerate <fps> -t <total+2>`: + +| strip | size | what it is | +|---|---|---| +| `_rail_chrome.png` | `RW × RHGT` | opaque panel, title, rules, **and the tally swatches and labels**. Runs the whole video — no `enable=` gates | +| `_rail_log.png` | `RW × N·rowHeight` | every claim, stacked, no padding | +| `_rail_curtain.png` | `RW × LOGH` | opaque `pal.bg` | +| `_rail_hl.png` | `RW × rowHeight` | the amber current-row marker | +| `_rail_tally.png` | `ΣlaneW × rows·cellH` | one COLUMN per lane — four rolling cells and the roster line | +| `_rail_qr.png` | `TILEW × segments·TILEH` | one provenance tile per segment | + +### The tally rolls one number at a time + +It used to be a column of four-row slabs walked by one `crop`: when the coffee +company's number changed, all four rows moved, and "The Quartering" slid up the +screen for a reason that had nothing to do with it. Text that has not changed +must not move. + +So the **swatch and the company label went into the static chrome**, and each +track got its own **rolling cell** — number, delta triangle, `as of <date>` and a +population chip, right-aligned in a ~170 px column. All the lanes live side by +side in **one** PNG, so it is still one input: five `crop`s at different `x`, five +overlays. + +**The direction of a roll is decided by the strip's LAYOUT, not by the ramp.** + +| | rows | the crop | what you see | +|---|---|---|---| +| rise | `[old, new]` | walks **down** | content moves **up** | +| fall | `[new, old]` | walks **up** | content moves **down** | + +Between transitions a one-frame `gte()` step repositions to the next pair's +starting row. That step is invisible **because both endpoint rows hold identical +content** — which is why every pair repeats the value it starts from instead of +sharing a row with its neighbour, and why the delta chip rides on both rows and +therefore stays on screen until the next change. + +``` +y_j(t) = r_j0 + Σ_k [ (a_k − b_{k−1})·gte(t,t_k) + (b_k − a_k)·ease(t_k) ] +``` + +Every term is cumulative and saturating — the rail's hard rule. + +A repeated identical figure still rolls, upward: he said it again on a new date, +and the `as of` line underneath is what changed. + +### The roster line + +A fifth lane under the tally, `2 editors · 1 designer`, in `pal.muted`. It moves +**only when the rendered line changes**, which in this corpus means it stands +still through October and December 2023 while the total above it goes from three +to four. That is the finding, drawn rather than asserted. + +It comes from an optional `roles` field on a ledger claim, and `rosterAt()` / +`rosterLine()` in `ledger-totals.mjs` are the one implementation, because a +chapter card states the same thing in words. + +```jsonc +"roles": [{ "role": "video editor", "count": 2, "verbatim": "two video editors" }] +``` + +`verbatim` is his words; `role` and `count` are our reading. `roles` is **not** one +of the six adjudication fields — most claims are a number and nothing else, and +gating the inbox on a field a handful of entries can carry would leave it +permanently red. + +### The QR moved into the rail's foot + +It used to float over the bottom-right of the **picture**, which is the one part +of the frame this cut promises never to draw on. It is now a bordered tile parked +at the foot of the rail column — `pal.accent` rule, `SCAN → JERALYZER` above the +code, what it opens below it — and one more strip: one tile per segment, +crop-walked with **instantaneous `gte()` steps** at segment mid-dissolves. A code +that eased into place would spend the ease unscannable. + +The tile overlays **after the curtain**, which is what stops the parked curtain +painting over it. + +> **A card cannot carry the report's share link.** That link carries all 23 +> channel filters and is ~1.4 k characters: a version-40 symbol, 177 modules in a +> 132 px tile, about 0.7 px per module. Cards get `provenance.qrLink` — the same +> query without the channel list, ~200 chars, 63 modules, verified scannable at +> this size — and fall back to `provenance.siteOrigin`. Manifests with **no** +> rail keep the old per-clip overlay in `buildClipSegment`, byte for byte. + +> **`railGeometry` used to derive its height from `render.footerHeight`.** With +> the chart band on, the band reserves 200 px and `footerHeight` says 100, so the +> rail column ran a hundred pixels — about two rows of its log window — past the +> line every other renderer letterboxes to. It reads `reservedFooterHeight(render)` +> now, the same fix the closing cards needed for the same reason. + +**The curtain is why one strip is enough.** With the window parked at the top +while the list is still filling, rows `i+1 … K-1` would show claims the video has +not made yet. The curtain is an opaque rectangle riding just below the last +revealed row; once the list is full it parks exactly one window-height down, +which is the bottom of the rail column — permanently outside the window. It is +`pal.bg` precisely so that parking there is invisible against the footer band. +Curtain and log **must share the same eased `P`**, or the curtain lags the rows +mid-slide and unrevealed claims flash into view. + +**Ramps are cumulative and saturating, never gated.** A piecewise sum of +`gte(t,sᵢ)·lt(t,sᵢ₊₁)·…` terms flashes to `y=0` for one frame at any boundary +gap, because every gate evaluates false at once and the sum collapses. Terms that +rise to their delta and stay cannot do that. + +**Scheduling.** Playback is ONE chronology across every company, and the ledger is +sorted the same way, so a claim's position in the rail *is* its position in time. +A claim with a clip behind it is pinned to that clip's segment; a claim on a +`ledger` card is pinned to its own row's reveal; the rest are spread evenly +between their neighbouring pins. A pin that runs backwards is refused — the +ledger and the timeline disagreeing about the order of events is a manifest bug, +and the whole cut rests on the two agreeing. + +Segment-level pins land at **`starts[i] + D/2`** — mid-dissolve, where the picture +is already crossfading and a ±3-frame error is invisible. + +The chain attaches **after the last `xfade` node**, inside the concat pass. `t` +there is absolute and continuous from 0, and nothing downstream of the last xfade +is dissolved — so it already has post-pass semantics without a second encode, +which would re-quantize crf-20 output at exactly the content that hurts most +(antialiased text on flat colour). + +### `--rail-only` and `--preview` + +`--rail-only` re-runs just the rail over a cached `out/<slug>.prerail.mp4`, +building that file from the existing segments the first time. Seconds instead of +the full concat. The hard-cut base is a separate file +(`<slug>.prerail-hardcut.mp4`), because the two timelines are different lengths +and a cached base from the wrong mode is a stale-cache trap the length assertion +would otherwise have to explain. + +**It is mandatory for `--no-xfade`**, not an optimisation: `concatHardCut` is +`-c copy` and a stream-copy mux cannot host a filtergraph at all. That path +concats to `.prerail.mp4`, asserts its length against `segmentOffsets().total`, +and then applies the rail. + +`--preview <start> <dur>` renders a window. `-ss` restarts `t` near zero, which +would put every absolute-time ramp in the wrong place — so the preview path +inserts `setpts=PTS+<start>/TB` before the rail chain and rebases afterwards. +Getting that wrong makes a working rail look broken. + +## The end sequence: `scroll` and `chart` + +Two timeline entry `type`s that exist to close a cut, both driven off the same +`ledger[]`: + +- **`scroll`** — the whole ledger as one tall PNG, walked by an animated `crop`. + **One chronological line, a column per company**: it used to group by company, + which re-told the cut's own order backwards and hid the only thing worth seeing + there — that the four payrolls were being described in the same weeks. Company + is read from COLUMN POSITION, so colour is the secondary encoding. + `hold` (default 2 s) buys a still moment at both ends; + `clip()` in the expression provides it for free, and crop's own clamping + degrades an off-by-a-few content height into a static last frame, not an error. +- **`chart`** — the four-series step chart over the `plotted` claims, authored as + SVG and rasterized with `rsvg-convert` (deterministic about output size in a + way ImageMagick's RSVG delegate is not). Fonts inside the SVG resolve through + **fontconfig, not `render.fontRegular`** — use the family name the Pango cards + use. + +The chart's wipe **cannot** be `crop=w='<ramp>'`: crop's `w` is config-time and +`t` is undefined there. It is a curtain instead — an opaque `pal.bg` rectangle +slid rightwards off the plot, which is exact because the card ground is flat. + +**Colour is not the only encoding on that chart, and that is a requirement.** No +four-colour categorical palette clears the data-viz all-pairs CVD gate (three +slots is the documented ceiling), so every series also carries a distinct dash +pattern and a direct end-of-line label with a leader elbow. The four hues are the +published artifact's, re-validated against this video's darker ground (`#0F1312`) +on the *adjacent* pairlist — the pairlist for line charts — where all five checks +pass (worst adjacent CVD ΔE 10.2 against a ≥8 target; normal-vision ΔE 17.4 +against a ≥15 floor). + +**`hideRail: true`** on a closing card slides the whole rail column off to the +right over that card's dissolve — one offset expression shared by every rail +overlay, so the column moves as one object — and lets the card render at the full +`width` instead of `contentWidth`. Not an `enable=` pop: a column that vanishes +between two frames reads as a dropped frame. The slide is cumulative and +saturating like every other ramp, so the rail does not come back; every card +after the first `hideRail` one should carry the flag too, or it lays out inside a +content width whose rail is no longer there. + +Both kinds go through the same `fps=,setsar=1` and the same `encodeArgs` as every +other segment. They have to: `xfade` rejects a mismatched link with *"First input +link parameters do not match"*, and that surfaces at concat time, after every +fetch has been paid for. + +## The ledger is adjudicated, and both totals depend on it + +`ledger[].company` used to be an **undocumented interpretation**, and four +different hazards were riding on it: + +| hazard | example | why it mattered | +|---|---|---| +| **scope ambiguity** | *"I have 10 employees, my coffee company employees… my editors"* | all-companies or coffee-only, depending on where the comma falls | +| **derived, not stated** | *"10 at coffee brand coffee, I've got eight staff for the live stream"* recorded as **18** | he never says 18 | +| **population drift** | 5 *"full-time salaried"*, 10 *"employees"*, 10 *"all basically contractors"* | different denominators, one series | +| **synthetic values** | 10.5 for *"about 10 people, 11 people"* | a midpoint we invented and attributed to him | + +So every ledger entry now carries six adjudicated fields — `scope`, +`scopeBasis`, `scopeConfidence`, `population`, `valueKind`, `flags` — settled +against **±90 s of surrounding context, never the quote alone**. A first-person +quote is routinely the host reading someone else's words or being sarcastic, and +neither is visible inside the quote. One claim in this corpus is a guest's +payroll rather than his, and it reads identically until you listen either side. + +`scopeConfidence: "unresolved"` is a legitimate outcome and **feeds neither +total**. + +**The rule that follows:** the *stated* series may contain only **a figure he +utters as a single number for a named scope**. Sums and midpoints are ours, and +live in the *implied* series, which says so on screen. + +`ledger-totals.mjs` is the one implementation of that arithmetic — the umtool +inbox, the chart band and the closing card all import it, so none of them can +disagree. It **refuses to run on an unadjudicated ledger**, because both totals +lie if you act on one. **Six** named predicates compute incoherence rather than +asserting it (`contradicts_component`, `same_day_conflict`, `self_negating`, +`population_mismatch`, `not_his_number`, `status_flip`). Deliberately **not** a +predicate: a large rise or fall between claims. Fluctuation is the subject, not a +defect. + +`status_flip` is the sixth: the same people described as staff and then as +contractors, or the reverse. Three details in it are load-bearing. + +- **`employees` and `people` are in NEITHER camp.** They are what he says when he + is not making a claim about status at all, and reading them as one side or the + other manufactures a reversal out of a change of vocabulary. +- **An `all` claim is comparable with any company; two companies are not + comparable with each other.** Without that asymmetry the corpus's clearest + reversal is invisible: December 2024's ten are the *channel's*, May 2025's ten + or eleven are *everything's*. +- **Only against the most recent comparable claim that carries a camp.** Fire on + every earlier pair and one 2022 "all 1099 and not full-time" flags each of the + next seven claims in turn — seven findings where there is one. Bounded this + way it fires at the TRANSITIONS, which is what a flip-flop is. + +It is not gated on the claim having a figure. *"That's why all my workers are +contract workers"* names no number and is the single clearest status claim here. + +Work the adjudication in umtool, at `/browse/<project>/claim/<id>`. Sign-off is +"the inbox is empty": `claim-unadjudicated` is **blocking**. + +## `render.chromeEngine: "hyperframes"` — the chart band + +Opt-in, and absent it the ffmpeg chrome path is byte-for-byte unchanged. + +The chart band **replaces** the footer node track, which only moved at section +handovers — precisely the fault it exists to fix. It takes the footer's ground +and 100 px more, and the picture loses that height (1500×924 → 1500×824). + +`compose-chrome.mjs` emits a HyperFrames project per region and renders it to a +**lossless RGBA PNG sequence**; `build-video.mjs` overlays the frames. Three +things about that are load-bearing: + +- **Footage never enters Chrome.** HyperFrames pre-extracts source video to JPEG + q95, which is unacceptable when the picture *is* the cited evidence. Only + chrome is composed there — about 37 % of full-frame pixels rather than 100 %. +- **The PNG regions overlay BEFORE the rail chain, not after.** The rail chain + ends in `format=yuv420p`, and overlaying an alpha sequence onto yuv420p is the + same alpha-subsampling trap the rail already documents, one layer later. +- **The playhead is driven by `out/schedule.json`**, which the build writes. + Recomputing claim times here would be a second implementation of + `segmentOffsets()` and would drift the first time `transition` changed. Same + rule as `widen()`: imported, never reimplemented. + +One clip-path sweeps the whole plot rather than a `stroke-dashoffset` per series. +The obvious build animates each path's dash offset, and it looks right for the +strokes and wrong for everything else: the gap band between the two totals is a +filled polygon with no stroke to offset, so it appears whole the moment it fades +in and the chart is showing an answer the playhead has not reached. + +**The flag colour is not the palette's amber.** `#E8A33F` sits at ΔE 12.4 from +the coffee series' `#D2732F` at *normal* vision — below the 15 floor — so a flag +badge beside a coffee mark was hard to tell from the coffee mark. `#E0E24A` +replaces it and adds **no new worst pair**: the worst CVD pair +(`#C55F9C↔#22AB83`, ΔE 5.4 deutan) and the worst normal-vision pair +(`#C55F9C↔#D2732F`, ΔE 16.3) are identical with and without it. The implied +total's `#EDF0EC` fails the categorical lightness and chroma checks *by design* — +it is an aggregate, not a categorical peer, so it is encoded by weight and +consumes no palette slot. + +## Things that cost time to find out + +**yt-dlp picks VP9 at `height<=720`, and that is a trap.** `--download-sections` +combined with `--force-keyframes-at-cuts` re-encodes, so a VP9 pick means +libvpx-vp9 — 27 seconds to cut a 5-second clip. It also writes a `.webm` and +appends that extension to whatever `-o` you gave, so the file never lands where +you asked and the run fails looking for it. Pin H.264/AAC in mp4 and the same cut +takes ~12 seconds. `build-video.mjs` does this and keeps a rename fallback for +the case where a fallback format still forces another container. + +**`--force-keyframes-at-cuts` is not optional here.** Without it the cut snaps to +the nearest preceding keyframe and can start seconds early. That is fine for a +human scrubbing a VOD; it is not fine when the clip *is* the citation. +`PlayerProvider.tsx:478` builds the copyable clip command without this flag — +correct for its purpose, wrong for ours. + +**`--ignore-config` is mandatory.** The operator's own yt-dlp config redirects +output to `~/Podcasts` and attaches thumbnail/metadata post-processors. Every +managed yt-dlp call in this repo passes `--ignore-config` for the same reason. + +**yt-dlp exit 101 is success**, not failure — it means a clean early stop. The +repo encodes this at `common/ytdlp/downloadOneManaged.ts:355`; a new caller has to +replicate it. + +**Local media will not help you.** Only 5 of 1,804 PirateSoftware video dirs hold +any media at all, and none are ones a report is likely to cite. Clips are a +network fetch. `metadata.info.json` and `transcript.cues.json` *are* present for +every video, so titles, dates, durations, webpage URLs and cue timings all come +from disk with no probe. + +**Upstream availability is load-bearing.** A clip can only be fetched while the +source is still up. Check `platform state` via `get_video_metadata` before +committing to a clip — a `deleted`/`maybe_missing` source needs a quote card +instead of footage. The archive outlives its sources, so a video built from an +old report will be *less* complete than the report unless this is handled +deliberately. + +**ImageMagick's `-size` leaks into the Pango group.** `-size 1920x1080 xc:BG` +followed by `( ... pango:@file )` renders the text into a full-frame box, which +pins it to the top and wraps at the frame edge instead of the text column. Reset +`-size` inside the parens. + +**ffmpeg `drawtext` does not wrap and hates punctuation.** Both are solved the +same way: wrap to a column count in JS, write to a file, and use +`textfile=`. Nothing then needs escaping. Stream titles also need emoji and +`!command` suffixes stripped or they render as tofu in the attribution line. + +**Segments are encoded to identical parameters on purpose** so the final +concatenation is a stream copy via the concat demuxer. Mismatched streams are the +usual reason a naive concat produces a broken or audio-desynced file. + +## Not done yet + +- **Narration is silent by design.** Cards carry the connective text; the only + audio is the clips'. A TTS layer would attach per card (`seconds` already gives + it a duration to fill) — deliberately deferred rather than designed out. +- **Snapping is silence-based, not word-based.** It finds gaps in the audio, which + is usually the same thing as a word boundary but is not guaranteed to be — + a speaker who does not pause gets the unsnapped cut. Forced alignment against + the transcript would be exact; `silencedetect` is a tenth of the work and + handles the cases that were actually audible. +- **Only the chart band is a HyperFrames region.** `chromeRegions()` returns one + entry. The rail and the header are still drawn by the ffmpeg chain, and porting + them is the rest of the job — the rail needs the ghost/hop convention (a row + with no clip fades in dimmed under a dashed rule and the amber highlight *hops + over* it to the next cited row) which the strip builders cannot express. Until + then `railFilterChain` and `renderFooterAssets` stay; they must not be retired + on the strength of the band alone. +- **`timelineNodes` / `section` / `sectionEnter` are still live.** They lose their + only consumer when the ffmpeg footer goes, not when the band arrives — so they + retire with `renderFooterAssets`, in that same commit. +- **Manifests are written by hand** from verified cue data. Deriving a first-draft + manifest automatically from a report's citations is the obvious next step; the + report parse is straightforward (`> "quote"` followed by + `— [title @ h:mm:ss](…?v=slug%2Fid&t=sec)`), the cue-matching is the real work. diff --git a/scripts/report-to-video/build-video.mjs b/scripts/report-to-video/build-video.mjs @@ -0,0 +1,1838 @@ +#!/usr/bin/env node +// build-video.mjs — render a cited sweep report into a narrated-by-text video. +// +// Takes a video manifest (see README.md next to this file) and produces one mp4: +// text cards state the findings, clips let the source say it in their own voice, +// and every clip carries a burned-in quote plus its attribution. +// +// Pipeline, per manifest entry: +// card -> still PNG (render-cards.mjs) -> N seconds of video + silent audio +// clip -> yt-dlp --download-sections (WIDE) -> silence-snap -> trim + burn +// then the segments are crossfaded together into the finished file. +// +// Three things worth knowing about how clips are cut: +// +// 1. Windows come from the manifest as absolute [start, end] seconds, derived +// from transcript.cues.json (which carries an END per cue). A sweep report +// only ever records a single start second, so windows cannot be recovered +// from the report alone. +// 2. Those windows are widened to sentence boundaries by resolve-windows.mjs, +// so a clip carries the run-up that makes the quote make sense. +// 3. A cue boundary is still not a *speech* boundary — cutting there clips +// words in half. So we fetch wider than needed and snap the real cut to a +// silence found in the audio. That is what makes clips start and end +// between words rather than through them. +// +// Fetched clips are cached by (video, start, end); re-running is cheap and only +// changed entries re-download. Delete out/clips-raw to force a refetch. +// +// In the app: not used. On the CLI: +// node scripts/report-to-video/build-video.mjs <manifest.json> [options] +// +// Options: +// --out <dir> Output root (default: manifest dir + /out) +// --variant <name> Which cut to build (sourced | full; default sourced) +// --skip-fetch Fail instead of downloading anything not already cached +// --only <id> Build a single entry's segment and stop (for iterating) +// --no-xfade Hard cuts instead of crossfades (much faster; concat copy) +// --progress ndjson One JSON event per line instead of prose (for umtool) +// --continue-on-error Record a failed entry and carry on, instead of aborting +// --fetch-only <id> Fetch one clip's window into clips-raw and stop +// --pad <s> Override render.fetchPad (the clip bench fetches wide) +// --site-origin <url> Archive to read cue windows from when there is no local +// corpus (defaults to the manifest's provenance.siteOrigin) +// --resolve-site-ids On a published-id miss, find the record by scanning the +// channel's shards. Slow; see cues.mjs. +// --cue-source <which> auto (default) | local | http. The two can disagree +// once a corpus moves past its last publish — see cues.mjs. +// --no-rail Skip the claim rail even when the manifest configures one +// --rail-only Re-run just the rail over out/<slug>.prerail.mp4 +// --preview <s> <d> Rail-only, over a <d>-second window starting at <s> +// +// Requires: yt-dlp, ffmpeg/ffprobe, ImageMagick with Pango. + +import { execFile } from "node:child_process"; +import { promisify } from "node:util"; +import { mkdir, writeFile, readFile, access, readdir, rename } from "node:fs/promises"; +import path from "node:path"; + +import { + renderCard, renderFooterAssets, renderRailAssets, renderScrollCard, renderChartCard, + renderLedgerCard, ledgerRevealAt, ledgerSeconds, + cardWidth, contentWidth, reservedFooterHeight, +} from "./render-cards.mjs"; +import { createCueSource, siteOriginFromManifest } from "./cues.mjs"; + +const execFileP = promisify(execFile); + +const YTDLP = process.env.YTDLP_BIN ?? "yt-dlp"; +const FFMPEG = process.env.FFMPEG_BIN ?? "ffmpeg"; +const FFPROBE = process.env.FFPROBE_BIN ?? "ffprobe"; +const QRENCODE = process.env.QRENCODE_BIN ?? "qrencode"; + +// Cue windows and per-video metadata come from a local corpus when there is one +// and from the published archive otherwise, so this runs in a clone with no +// `transcripts/` directory. Built once main() has the manifest (it carries the +// archive origin); see cues.mjs. +let CUES = null; + +const exists = (p) => access(p).then(() => true, () => false); + +// ---- variants ------------------------------------------------------------ +// ONE manifest, two cuts, one filter, applied once. +// +// The question the two variants answer differently is what to do with a claim +// the sweep found but no clip covers. `sourced` refuses to put it on screen at +// all -- every row the viewer sees has footage behind it. `full` gives each one +// a slot on a stacked ledger card, so nothing is dropped and the arithmetic of +// each layer is shown rather than asserted. +// +// Both end on the same three numbers. That is the point of shipping both: if +// the totals moved when the unsourced rows came off, the thesis would rest on +// rows nobody can check. +// +// The filter runs IMMEDIATELY after the manifest is read, and nothing +// downstream learns about variants. `ledgerTotals`, the rail, the chart band, +// `scheduleClaims`, the chapters and the scroll already take the ledger and the +// timeline as inputs, so selecting is the whole of the mechanism. +export const VARIANTS = ["sourced", "full"]; +/** The cut a caller means when it does not say. `out/<slug>.mp4`. */ +export const DEFAULT_VARIANT = "sourced"; + +/** + * The manifest as one variant sees it. + * + * Three things happen, in this order: + * + * 1. A timeline entry tagged `variant` survives only in that variant. (The + * stacked ledger cards are `variant: "full"`.) + * 2. A claim survives only if the entry it is pinned to survived. That single + * rule is what makes `sourced` a sourced-only ledger: in `full` every claim + * is pinned -- to a clip or to a ledger card -- so nothing is dropped. + * 3. `card.variants[<name>]` field overrides are merged in. The title and + * sources cards have to state their own scope honestly, and "50 dated + * claims" is simply false in `sourced`. + */ +export function selectVariant(manifest, variant = DEFAULT_VARIANT) { + if (!VARIANTS.includes(variant)) { + throw new Error(`unknown variant \`${variant}\` — one of ${VARIANTS.join(", ")}`); + } + const timeline = (manifest.timeline ?? []) + .filter((e) => !e.variant || e.variant === variant) + .map((e) => { + if (!e.variants) return e; + const { variants, ...rest } = e; + return { ...rest, ...(variants[variant] ?? {}) }; + }); + const kept = new Set(timeline.map((e) => e.id)); + const ledger = (manifest.ledger ?? []).filter((c) => c.entryId && kept.has(c.entryId)); + return { ...manifest, variant, timeline, ledger }; +} + +/** + * Where a variant's own working files live. + * + * `clips-raw` stays at the ROOT and is shared: it holds the only expensive + * thing in the build (network fetches), and `sourced`'s clips are a subset of + * `full`'s, so a shared cache means no clip is ever fetched twice. Everything + * else is per-variant, because every one of them differs between the two cuts. + * + * `sourced` writes `out/<slug>.mp4` -- the path umtool's build probe already + * looks for -- and `full` writes `out/<slug>-full.mp4` beside it. + */ +export function variantPaths(outRoot, slug, variant) { + return { + root: outRoot, + dir: path.join(outRoot, variant), + rawDir: path.join(outRoot, "clips-raw"), + final: path.join(outRoot, variant === "sourced" ? `${slug}.mp4` : `${slug}-${variant}.mp4`), + }; +} + +// ---- progress protocol --------------------------------------------------- +// This has two audiences: a human watching a terminal, and umtool's build driver +// reading the pipe. Rather than have the driver scrape prose (which would make +// every wording change a breaking change), `--progress ndjson` switches every +// line to one JSON object. The event set is exactly what was already being +// printed -- this is a formatting switch, not new instrumentation. +// +// Events: start, card, clip, fetch, snap, segment, entry-failed, concat, +// chapters, note, done. +const HUMAN = { + start: (e) => `${e.title} — ${e.entries} entr(ies)`, + card: (e) => `card ${e.id}`, + clip: (e) => `clip ${e.id} (${e.video}) §${e.section}${e.sectionEnter ? " ⟶" : ""}`, + fetch: (e) => + e.reuse + ? ` fetch ${e.id}: ${e.reuse} already covers ${hms(e.from)}–${hms(e.to)} — no download` + : e.cached + ? null + : ` fetch ${e.id}: ${e.video} ${hms(e.from)}–${hms(e.to)}`, + snap: (e) => + ` snap ${e.id}: ${e.start ? "start✓" : "start–"} ${e.end ? "end✓" : "end–"} ` + + `(${Number(e.seconds).toFixed(1)}s)`, + segment: () => null, + "entry-failed": (e) => ` ** ${e.id} failed: ${e.message}`, + concat: (e) => `${e.mode === "xfade" ? "crossfading" : "hard-cutting"} ${e.n} segments…`, + chapters: (e) => `chapters: ${e.n} marker(s) -> ${e.file}`, + note: (e) => e.message, + done: (e) => + e.duration === undefined + ? `built ${e.out}` + : `\n${e.out}\nduration=${e.duration}\nsize=${e.size}`, +}; + +let EMIT = (ev, fields = {}) => { + const line = HUMAN[ev]?.({ ev, ...fields }); + if (line) console.log(line); +}; + +export function setProgressMode(mode) { + EMIT = + mode === "ndjson" + ? (ev, fields = {}) => process.stdout.write(JSON.stringify({ ev, ...fields }) + "\n") + : (ev, fields = {}) => { + const line = HUMAN[ev]?.({ ev, ...fields }); + if (line) console.log(line); + }; +} + +function hms(total) { + const s = Math.floor(total); + const h = Math.floor(s / 3600); + const m = Math.floor((s % 3600) / 60); + const sec = s % 60; + return h > 0 + ? `${h}:${String(m).padStart(2, "0")}:${String(sec).padStart(2, "0")}` + : `${m}:${String(sec).padStart(2, "0")}`; +} + +// Stream titles here are full of emoji and !commands. drawtext renders them as +// tofu with a text font, and they add nothing to an attribution line. +function cleanTitle(title) { + return title + .replace(/[\u{1F000}-\u{1FFFF}\u{2600}-\u{27BF}\u{FE0F}]/gu, "") + .replace(/\s*[!@]\S+/g, "") + .replace(/\s{2,}/g, " ") + .replace(/[\s·|-]+$/, "") + .trim(); +} + +// drawtext does not wrap. Break to a character budget, write to a file, and use +// textfile= so nothing needs shell or filter escaping. +function wrap(text, cols) { + const words = text.split(/\s+/); + const lines = []; + let line = ""; + for (const w of words) { + if (line && (line + " " + w).length > cols) { + lines.push(line); + line = w; + } else { + line = line ? line + " " + w : w; + } + } + if (line) lines.push(line); + return lines.join("\n"); +} + +// The published shard record carries the same fields as a local cue file, so this +// reads identically whichever source answered. +async function videoMeta(videoId, channelSlug, hints = {}) { + const d = await CUES.load(channelSlug, videoId, hints); + return { title: d.title, uploadDate: d.uploadDate, webpageUrl: d.webpageUrl, duration: d.duration }; +} + +// The CONTAINER's duration is max(video, audio), and the audio is longer: the +// AAC encoder pads the front with ~21 ms of decoder delay, and a video duration +// is rarely an exact multiple of the frame interval. Either way the excess is +// small — and it ACCUMULATES through segmentOffsets, which subtracts one +// transition per segment and hands the result to xfade, the chapter marks and +// (now) the rail. A few hundred ms of drift by segment 20 is enough to land a +// rail row-change on the wrong side of a cut. +// +// The video stream's frame COUNT is the number the timeline actually runs on, +// so derive the duration from it. nb_frames is absent on some demuxers; fall +// back to the container rather than failing a build over a probe. +async function probeDuration(file, fps) { + if (fps) { + const { stdout } = await execFileP(FFPROBE, [ + "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=nb_frames", + "-of", "default=nw=1:nk=1", file, + ]); + const n = Number(stdout.trim()); + if (Number.isFinite(n) && n > 0) return n / fps; + } + const { stdout } = await execFileP(FFPROBE, [ + "-v", "error", "-show_entries", "format=duration", + "-of", "default=nw=1:nk=1", file, + ]); + return Number(stdout.trim()); +} + +// yt-dlp exits 101 on a clean early stop (break-on-existing / max-downloads). +// The repo treats that as success everywhere else; do the same here. +const ytdlpOk = (err) => err?.code === 101; + +// ---- the clip cache ------------------------------------------------------ +// A raw clip's window is IN ITS NAME, which makes the file immutable and the +// cache content-addressed. The original lookup was for the exact name, so any +// change to a window -- a hand edit, a widen, a nudge in the clip bench -- was a +// fresh download of material already on disk. Measured on ferret-rescue: 31 +// files for 10 clips, one source fetched four times over overlapping windows. +// +// So: satisfy a request from ANY cached file that contains it. The TIGHTEST +// container wins, because detectSilence decodes the whole file and a 40s file +// costs more than the 14s one that would also have done. The clip bench fetches +// deliberately wide, and this is what makes that generous fetch become the +// build's cache rather than a second one. +const WINDOW_RE = /^(\d+(?:\.\d+)?)-(\d+(?:\.\d+)?)$/; + +// A window read back from a 2 dp manifest can sit a hair outside the file that +// produced it; the same tolerance resolve-windows.mjs uses for the same reason. +const WIN_EPS = 0.02; + +export async function cachedWindowsFor(rawDir, video) { + let names; + try { + names = await readdir(rawDir); + } catch { + return []; + } + const prefix = `${video}_`; + const out = []; + for (const name of names) { + if (!name.startsWith(prefix) || !name.endsWith(".mp4")) continue; + // The remainder must be exactly `a-b`, which is what stops a video id that + // is a prefix of another (or one containing `_`) from claiming its files. + const m = WINDOW_RE.exec(name.slice(prefix.length, -4)); + if (!m) continue; + out.push({ name, path: path.join(rawDir, name), from: Number(m[1]), to: Number(m[2]) }); + } + return out; +} + +/** The tightest cached file containing [from, to], or null. */ +export async function findContainingWindow(rawDir, video, from, to) { + const windows = await cachedWindowsFor(rawDir, video); + let best = null; + for (const w of windows) { + if (w.from > from + WIN_EPS || w.to < to - WIN_EPS) continue; + if (!best || w.to - w.from < best.to - best.from) best = w; + } + return best; +} + +async function fetchClip(entry, meta, render, rawDir, opts) { + // Deliberately over-fetch: the snapping pass below needs room on both sides to + // find a silence, and a clip that has no slack can only be cut where the cue + // happened to break — which is what put words in half in the first place. + const pad = opts.pad ?? render.fetchPad ?? 3.0; + const from = Math.max(0, entry.start - pad); + const to = entry.end + pad; + + // Shared across variants, and deliberately so: this is the only expensive + // thing in a build, and the two cuts overlap almost entirely. + const name = `${entry.video}_${from.toFixed(2)}-${to.toFixed(2)}.mp4`; + const dest = path.join(rawDir, name); + if (await exists(dest)) { + EMIT("fetch", { id: entry.id, video: entry.video, from, to, cached: true }); + return { path: dest, fetchStart: from, cached: true }; + } + if (!opts.noReuse) { + const hit = await findContainingWindow(rawDir, entry.video, from, to); + if (hit) { + EMIT("fetch", { + id: entry.id, video: entry.video, from, to, cached: true, reuse: hit.name, + }); + // fetchStart is the CACHED file's start, not the requested one -- every cut + // downstream is expressed relative to it, so reuse is transparent. + return { path: hit.path, fetchStart: hit.from, cached: true }; + } + } + if (opts.skipFetch) throw new Error(`--skip-fetch set and no cached window covers ${name}`); + + const maxH = render.maxHeightSource; + const fmt = [ + `bv*[vcodec^=avc1][height<=${maxH}]+ba[acodec^=mp4a]`, + `bv*[ext=mp4][height<=${maxH}]+ba[ext=m4a]`, + `b[ext=mp4][height<=${maxH}]`, + `b[height<=${maxH}]`, + ].join("/"); + + const argsWith = (extra) => [ + // The operator's own yt-dlp config redirects output and attaches thumbnail + // and metadata post-processors; without this the clips land elsewhere. + "--ignore-config", + "--no-playlist", + "--download-sections", `*${from.toFixed(2)}-${to.toFixed(2)}`, + // Without this the cut snaps to the nearest preceding keyframe, which can be + // seconds early — fine for scrubbing, not fine when the clip IS the citation. + "--force-keyframes-at-cuts", + ...extra, + // Pin H.264/AAC in mp4. Left alone yt-dlp picks VP9+Opus at these heights, + // and since --force-keyframes-at-cuts re-encodes, that means libvpx-vp9 — + // 27s to cut a 5s clip. It also writes .webm and appends that to -o. + "-f", fmt, + "--merge-output-format", "mp4", + "-o", dest, + "--", meta.webpageUrl, + ]; + + const attempt = async (extra) => { + try { + await execFileP(YTDLP, argsWith(extra), { maxBuffer: 1 << 26 }); + return null; + } catch (err) { + return ytdlpOk(err) ? null : err; + } + }; + + EMIT("fetch", { id: entry.id, video: entry.video, from, to, cached: false }); + let err = await attempt([]); + + // Rumble delivers HLS whose segments are named `.tar`, and ffmpeg 8's picky + // extension check rejects those outright — "URL ... is not in + // allowed_segment_extensions" — killing the fetch with exit 183. Rumble ships + // no progressive format to fall back to, so without this every Rumble-sourced + // clip is unbuildable. + // + // It has to be a RETRY, not a default: -extension_picky lives on the HLS + // demuxer, so passing it against a progressive URL (YouTube's googlevideo mp4) + // makes ffmpeg abort with "Option extension_picky not found" — i.e. adding it + // unconditionally trades a Rumble failure for a YouTube one. + if (err && /allowed_segment_extensions|allowed_extensions/.test(String(err.stderr ?? err.message ?? ""))) { + EMIT("note", { id: entry.id, message: ` ${entry.id}: HLS segment extension rejected, retrying with -extension_picky 0` }); + err = await attempt(["--downloader-args", "ffmpeg_i:-extension_picky 0"]); + } + if (err) { + throw new Error(`yt-dlp failed for ${entry.id} (${entry.video}): ${err.stderr ?? err.message}`); + } + if (!(await exists(dest))) { + // If a fallback format still forced another container, yt-dlp writes + // "<dest>.<realext>". Adopt it rather than failing the run. + const dir = path.dirname(dest); + const base = path.basename(dest); + const stray = (await readdir(dir)).find((f) => f.startsWith(base + ".")); + if (!stray) throw new Error(`yt-dlp reported success but produced no file for ${entry.id}`); + await rename(path.join(dir, stray), dest); + } + return { path: dest, fetchStart: from, cached: false }; +} + +// Parse ffmpeg's silencedetect output into [{s, e}] intervals, in seconds +// relative to the start of the given file. +async function detectSilence(file, render) { + const minDur = render.silenceMinDur ?? 0.09; + + // The threshold has to be RELATIVE to the clip, not absolute. These are game + // streams: the gaps between words are full of game audio and music, so they + // are quiet but nowhere near silent. A fixed -32 dB sits below the noise floor + // of a typical clip here and finds literally zero silences (measured: mean + // volume -21 dB, 0 hits at -32 dB, 25 hits at -26 dB). Measure the clip first + // and cut a few dB under its own mean instead. + const { stderr: volLog } = await execFileP( + FFMPEG, + ["-nostdin", "-i", file, "-af", "volumedetect", "-f", "null", "-"], + { maxBuffer: 1 << 26 }, + ).catch((e) => ({ stderr: e.stderr ?? "" })); + const meanMatch = (volLog ?? "").match(/mean_volume:\s*(-?[\d.]+) dB/); + const mean = meanMatch ? Number(meanMatch[1]) : -24; + const noise = Math.max(-45, Math.min(-18, mean - (render.silenceRelDb ?? 6))); + + // ffmpeg exits 0 here, so stderr comes back on the resolved result. + const { stderr } = await execFileP( + FFMPEG, + ["-nostdin", "-i", file, "-af", `silencedetect=noise=${noise.toFixed(1)}dB:d=${minDur}`, "-f", "null", "-"], + { maxBuffer: 1 << 26 }, + ).catch((e) => ({ stderr: e.stderr ?? "" })); + const log = stderr ?? ""; + + const out = []; + let open = null; + for (const line of log.split("\n")) { + const s = line.match(/silence_start:\s*(-?[\d.]+)/); + if (s) open = Number(s[1]); + const e = line.match(/silence_end:\s*(-?[\d.]+)/); + if (e && open !== null) { + out.push({ s: open, e: Number(e[1]) }); + open = null; + } + } + return out; +} + +// Snap a desired cut to the nearest silence, so the clip begins and ends between +// words instead of through one. Returns the desired point unchanged when no +// silence is close enough — better a tight cut than a cut in the wrong place. +function snap(desired, intervals, kind, window) { + let best = null; + for (const iv of intervals) { + // Starting: we want to resume just before speech does -> the silence's END. + // Ending: we want to stop just after speech does -> the silence's START. + const point = kind === "start" ? iv.e : iv.s; + const d = Math.abs(point - desired); + if (d > window) continue; + if (!best || d < best.d) best = { d, point }; + } + if (!best) return { at: desired, snapped: false }; + const lead = kind === "start" ? -0.10 : 0.18; + return { at: Math.max(0, best.point + lead), snapped: true }; +} + +// Encoder quality is manifest-driven so a cut can trade size for fidelity without +// editing this file. Defaults reproduce the original hardcoded settings exactly. +const encodeArgs = (render) => [ + "-c:v", "libx264", + "-preset", render.preset ?? "medium", + "-crf", String(render.crf ?? 20), + "-pix_fmt", "yuv420p", + "-r", String(render.fps), + "-c:a", "aac", + "-b:a", render.audioBitrate ?? "160k", + "-ar", String(render.audioRate), + "-ac", String(render.audioChannels), + "-movflags", "+faststart", +]; + +// The rail pass runs over an ALREADY ENCODED file, so its audio is already the +// finished AAC. Re-encoding it would cost a whole generation for nothing — and +// would make "the rail does not touch the audio" untrue. +const encodeArgsVideoOnly = (render) => [ + "-c:v", "libx264", + "-preset", render.preset ?? "medium", + "-crf", String(render.crf ?? 20), + "-pix_fmt", "yuv420p", + "-r", String(render.fps), + "-c:a", "copy", + "-movflags", "+faststart", +]; + +async function buildClipSegment(entry, meta, render, dirs, opts, chrome, nodes, provenance) { + const outDir = dirs.dir; + const { path: raw, fetchStart } = await fetchClip(entry, meta, render, dirs.rawDir, opts); + const seg = path.join(outDir, "segments", `${entry.id}.mp4`); + const pal = render.palette; + const { width, height } = render; + + // Desired cut points, expressed relative to the over-fetched file. + const wantA = entry.start - fetchStart; + const wantB = entry.end - fetchStart; + const win = render.snapWindow ?? 1.6; + + const sil = await detectSilence(raw, render); + const a = snap(wantA, sil, "start", win); + const b = snap(wantB, sil, "end", win); + // Never let snapping invert or collapse the window. + const cutA = Math.min(a.at, wantB - 1); + const cutB = Math.max(b.at, cutA + 1); + EMIT("snap", { id: entry.id, start: a.snapped, end: b.snapped, seconds: cutB - cutA }); + + const quotePath = path.join(outDir, "segments", `${entry.id}.quote.txt`); + const attribPath = path.join(outDir, "segments", `${entry.id}.attrib.txt`); + // Written for reference/diffing only — the quote is no longer drawn on screen. + await writeFile(quotePath, wrap(`“${entry.quote}”`, 92), "utf8"); + + const d = meta.uploadDate; + const date = `${d.slice(0, 4)}-${d.slice(4, 6)}-${d.slice(6, 8)}`; + await writeFile( + attribPath, + `${cleanTitle(meta.title)} · ${date} @ ${hms(entry.cite ?? entry.start)}`, + "utf8", + ); + + // The picture is the point. Nothing is drawn over it: the video is letterboxed + // between a thin citation header and a thin timeline footer, so the source + // material plays unobstructed and the additions stay subtle. + const HH = render.headerHeight ?? 56; + // The picture lives left of the rail column; the rail's own pixels are painted + // by the rail chain at concat time, over ground this pad leaves for it. + const VW = contentWidth(render); + const FH = chrome.footerHeight; + const hasFooter = FH > 0 && chrome.footer; + // headerHeight:0 drops the citation line too, leaving the clips alone on screen. + // Worth having: a cut whose sources are listed elsewhere does not need to carry + // its own attribution burnt into every frame. + const hasHeader = HH > 0; + const VH = height - HH - FH; + const trackAbsY = height - FH + chrome.trackY; + + // Where the progress marker travels this clip. Only the first clip of a + // section moves it; the rest hold it in place. + const T = render.slideSeconds ?? 0.9; + const xTo = chrome.xs[entry.section]; + const xFrom = entry.sectionEnter ? chrome.xs[Math.max(0, entry.section - 1)] : xTo; + // Commas inside a filter option have to survive filtergraph parsing; single + // quotes around the expression is what protects them. + const ramp = (a, b) => + a === b ? String(b) : `'if(lt(t,${T}),${a}+(${b}-${a})*t/${T},${b})'`; + const markX = ramp(xFrom - chrome.markerRadius, xTo - chrome.markerRadius); + + // The fill bar CANNOT be a drawbox with a `t`-dependent width. drawbox has no + // time variable at all: its `t` is the box THICKNESS, and with `t=fill` that + // is effectively INT_MAX, so the old `if(lt(t,0.9),…)` was always false and + // the bar was always drawn at its final width. (Proof: `drawbox=w='t*10'` and + // `drawbox=w=20` produce an identical YAVG.) Only the amber marker ever moved. + // + // So do it the way the rail does: a 2*LEN-wide strip, accent on the left half + // and transparent on the right, translated under a fixed-width crop. crop's + // x IS per-frame in `t`, and it clamps, so the ends are self-parking. + const fillA = xFrom - chrome.x0; + const fillB = xTo - chrome.x0; + const fillExpr = fillA === fillB + ? String(fillB) + : `${fillA}+(${fillB - fillA})*clip(t/${T},0,1)`; + + const base = [ + `scale=${VW}:${VH}:force_original_aspect_ratio=decrease`, + `pad=${VW}:${VH}:(ow-iw)/2:(oh-ih)/2:color=${pal.bg}`, + // Widen back to the full frame, leaving the rail column (if any) as ground. + `pad=${width}:${VH}:0:0:color=${pal.bg}`, + `pad=${width}:${height}:0:${HH}:color=${pal.bg}`, + "setsar=1", + `fps=${render.fps}`, + ...(hasHeader + ? [ + `drawbox=x=90:y=${Math.round((HH - 24) / 2)}:w=4:h=24:color=${pal.accent}:t=fill`, + [ + `drawtext=textfile='${attribPath}'`, + `fontfile='${render.fontRegular}'`, + "fontsize=22", + `fontcolor=${pal.muted}`, + "x=118", + `y=${Math.round((HH - 26) / 2)}`, + ].join(":"), + ] + : []), + ].join(","); + + // A manifest with a RAIL draws the code in the rail's foot instead, as one + // more strip: bottom-right of the frame, bordered, one per clip. It used to + // float over the bottom-right of the picture — the one part of the frame this + // cut promises never to draw on. Manifests with no rail keep the old overlay, + // byte for byte. + const qr = + render.qr === false || render.rail ? null : await qrForEntry(entry, provenance, render, outDir); + const qrM = render.qr?.margin ?? 28; + + // Bound the bar strip SHORTER than the clip. An overlay secondary that outruns + // the main extends the output, and the fix for that (shortest=1) would instead + // truncate the clip to the strip. Ending early is free: overlay's default + // eof_action=repeat holds the strip's last frame, which is the parked bar. + const barT = Math.max(0.2, cutB - cutA - 0.25); + + const inputs = ["-ss", cutA.toFixed(3), "-to", cutB.toFixed(3), "-i", raw]; + let nextIdx = 1; + let footerIdx, markerIdx, barIdx, qrIdx; + if (hasFooter) { + footerIdx = nextIdx++; inputs.push("-i", chrome.footer); + markerIdx = nextIdx++; inputs.push("-i", chrome.marker); + // A PNG fed with a plain -i through an ANIMATED crop is frozen: the crop + // sees one frame at t=0 and repeatlast repeats the already-cropped result. + // -loop 1 -framerate is what makes the strip a video the crop can walk. + barIdx = nextIdx++; + inputs.push( + "-loop", "1", "-framerate", String(render.fps), "-t", barT.toFixed(3), + "-i", chrome.bar, + ); + } + if (qr) { qrIdx = nextIdx++; inputs.push("-i", qr.png); } + + const parts = hasFooter + ? [ + `[0:v]${base}[b]`, + `[b][${footerIdx}:v]overlay=0:${height - FH}[f]`, + `[${barIdx}:v]crop=w=${chrome.trackLen}:h=3:x='${chrome.trackLen}-(${fillExpr})':y=0[bar]`, + `[f][bar]overlay=x=${chrome.x0}:y=${trackAbsY - 1}[g]`, + `[g][${markerIdx}:v]overlay=x=${markX}:y=${trackAbsY - chrome.markerRadius}[q]`, + ] + : [`[0:v]${base}[q]`]; + + // Sit above the footer when there is one, so the code never straddles the chrome. + parts.push( + qr + ? `[q][${qrIdx}:v]overlay=x=${VW}-w-${qrM}:y=H-h-${FH + qrM}[v]` + : `[q]null[v]`, + ); + + await execFileP( + FFMPEG, + [ + "-nostdin", "-v", "error", "-y", + ...inputs, + "-filter_complex", parts.join(";"), + "-map", "[v]", "-map", "0:a", + ...encodeArgs(render), + seg, + ], + { maxBuffer: 1 << 24 }, + ); + return seg; +} + +// ---- QR provenance code -------------------------------------------------- +// A compilation asks the viewer to take the edit on trust. The QR is the antidote: +// it resolves to this clip's exact START in the archive's own viewer, so anyone can +// pull up the surrounding hour and check that the cut is fair. Per clip, because a +// single code for the whole video would send everyone to the first citation. +// +// Two rules learned the hard way: it must be FULLY OPAQUE (a translucent QR will +// not scan) and it must keep its quiet zone (the white border is part of the +// symbol, not decoration). +async function qrForEntry(entry, provenance, render, outDir) { + const q = render.qr ?? {}; + // A mirror's LOCAL slug is not the id the site serves, and a clip taken from a + // copy whose archived transcript is broken should point at the copy that reads — + // so an explicit per-clip citeUrl always wins over the derived one. + const url = + entry.citeUrl ?? + `${provenance.siteOrigin}/?v=${encodeURIComponent( + `${entry.channel ?? provenance.channelSlug}/${entry.video}`, + )}&t=${Math.floor(entry.start)}`; + const png = path.join(outDir, "qr", `${entry.id}.png`); + await execFileP(QRENCODE, [ + "-o", png, + "-s", String(q.scale ?? 4), + "-m", String(q.quiet ?? 3), + "-l", q.ecc ?? "M", + url, + ]); + return { png, url }; +} + +async function buildCardSegment(card, render, outDir, nodes) { + const png = await renderCard(card, render, outDir, nodes); + const seg = path.join(outDir, "segments", `${card.id}.mp4`); + const dur = String(card.seconds); + + await execFileP( + FFMPEG, + [ + "-nostdin", "-v", "error", "-y", + // Without -framerate the image demuxer runs at its 25 fps default and the + // `-vf fps=30` below DUPLICATES a frame — at the segment's first frame, + // which is exactly where the next xfade seam lands. + "-loop", "1", "-framerate", String(render.fps), "-t", dur, "-i", png, + "-f", "lavfi", "-t", dur, + "-i", `anullsrc=channel_layout=stereo:sample_rate=${render.audioRate}`, + "-vf", `fps=${render.fps},setsar=1`, + ...encodeArgs(render), + "-shortest", + seg, + ], + { maxBuffer: 1 << 24 }, + ); + return seg; +} + +// =========================================================================== +// The claim rail +// =========================================================================== +// A persistent vertical ledger down the right edge, appending one row per claim +// as the video runs. It is folded into the concat pass rather than added as a +// second encode: the chain attaches AFTER the final xfade node, which already +// has post-pass semantics (nothing downstream of the last xfade is dissolved, +// and `t` there is absolute and continuous from 0). A separate pass would +// re-quantize crf-20 output, and antialiased text on flat colour is exactly the +// content that costs most. +// +// Five hard-won rules are load-bearing here; breaking any one produces a hang, +// a silently wrong-length file or a frozen overlay: +// +// 1. crop's w/h are CONFIG-TIME (`t` is undefined there) but x/y are +// per-frame. So every moving part is a fixed-size window walking a strip. +// 2. crop clamps x/y into range, so over-scroll is safe and self-parking. +// 3. A PNG on a plain -i through an animated crop is FROZEN. Every strip +// needs `-loop 1 -framerate <fps> -t <bound>`. +// 4. An UNBOUNDED `-loop 1` input deadlocks ffmpeg once several are chained. +// Hence `-t` on all five. +// 5. An overlay secondary longer than the main EXTENDS the output. The strips +// are deliberately longer (bound = total + 2), so `shortest=1` is required +// on EVERY rail overlay, not just the first. +// +// And two rendering ones: overlay's default `format=yuv420` subsamples alpha as +// well as chroma, which fringes 14–23 px rail text — so every rail overlay is +// `format=yuv444`, with a single `format=yuv420p` before the encoder. + +/** + * Fill an evenly-spaced schedule between known anchors. + * + * `known` holds the entries that are pinned to a segment; everything else is + * distributed linearly between its neighbouring pins, with `lo`/`hi` acting as + * virtual anchors just outside the run. + */ +function distribute(known, n, lo, hi) { + const at = new Array(n).fill(null); + for (const [i, t] of known) at[i] = t; + const pts = [[-1, lo], ...[...known].sort((a, b) => a[0] - b[0]), [n, hi]]; + for (let k = 0; k < pts.length - 1; k += 1) { + const [a, ta] = pts[k]; + const [b, tb] = pts[k + 1]; + for (let j = a + 1; j < b; j += 1) at[j] = ta + ((tb - ta) * (j - a)) / (b - a); + } + return at; +} + +/** + * When each ledger row appears, in finished-timeline seconds. + * + * ONE CHRONOLOGY. The cut plays in date order across every company, and the + * ledger is sorted the same way, so a claim's position in the rail IS its + * position in time. That collapses what used to live here: there is no longer a + * per-company span to bound a claim to, no contiguity rule to enforce, and no + * risk of scheduling a media claim over a coffee clip -- because "over a coffee + * clip" now means "later in the same chronology", which is exactly right. + * + * What remains is the part that was always doing the work: a claim WITH a clip + * behind it is pinned to that clip's segment, and the rest are spread evenly + * between their neighbouring pins. Monotonicity holds by construction, since + * both the pins and the rows are in date order. + * + * Every state change lands at `starts[i] + D/2` -- MID-DISSOLVE -- where the + * picture is already crossfading and a ±3-frame error is invisible. + */ +export function scheduleClaims(ledger, entries, starts, D, endBound) { + const segOf = new Map(entries.map((e, i) => [e.id, i])); + const mid = (seg) => starts[seg] + D / 2; + + // A stacked ledger card carries several claims, and each one has a MOMENT + // inside that card: the reveal of its own row. Pinning all of them to the + // segment's mid-dissolve would land four rail rows on one frame and, worse, + // break the pin-order guard's strict monotonicity for no reason. So a claim + // on such a card is pinned to its own row's reveal. + const within = new Map(); + for (const e of entries) { + if (e.type !== "ledger") continue; + (e.claims ?? []).forEach((cid, r) => within.set(`${e.id}|${cid}`, ledgerRevealAt(r))); + } + + const known = new Map(); + ledger.forEach((c, i) => { + const seg = c.entryId ? segOf.get(c.entryId) : undefined; + if (seg === undefined) return; + const off = within.get(`${c.entryId}|${c.id}`); + known.set(i, off === undefined ? mid(seg) : starts[seg] + off); + }); + if (!known.size) throw new Error("ledger: no claim is pinned to a clip, so nothing anchors the rail"); + + // A pin that runs backwards means the ledger and the timeline disagree about + // the order of events, which is a manifest bug rather than something to + // silently smooth over -- the whole cut rests on the two agreeing. + const pins = [...known].sort((a, b) => a[0] - b[0]); + for (let i = 1; i < pins.length; i += 1) { + if (pins[i][1] <= pins[i - 1][1]) { + throw new Error( + `ledger: ${ledger[pins[i][0]].id} is pinned to ${ledger[pins[i][0]].entryId}, which plays ` + + `before ${ledger[pins[i - 1][0]].id}'s clip — the ledger is not in the cut's order`, + ); + } + } + + // The first card is the title; the rail's own run opens just after it. + const times = distribute(known, ledger.length, mid(0), endBound); + + for (let i = 1; i < times.length; i += 1) { + if (times[i] <= times[i - 1]) times[i] = times[i - 1] + 1 / 30; + } + return times; +} + +/** + * The rail's filtergraph, as one builder with two call sites — the concat pass + * and `--rail-only` — so the two paths cannot drift. + * + * Ramps are CUMULATIVE AND SATURATING, never gated. A piecewise sum of + * `gte(t,s)*lt(t,s')*…` terms flashes to y=0 for one frame at any boundary gap, + * because every gate evaluates false at once and the sum collapses. Terms that + * rise to their delta and stay there cannot do that. + */ +export function railFilterChain(rail, assets, times, render, inLabel, firstInputIdx, bound, opts = {}) { + const g = assets.geom; + const SLIDE = rail.slide ?? 0.55; + const fps = render.fps; + + const P = (s) => `clip((t-${s.toFixed(3)})/${SLIDE},0,1)`; + // smoothstep() does not exist in ffmpeg's expression language. This is it. + const ease = (s) => { const p = P(s); return `${p}*${p}*(3-2*${p})`; }; + // Signed, and explicitly so. Joining terms with "+" was fine while every + // delta was a positive row height; a rolling cell FALLS as often as it rises, + // and `…+-40*x` is at best relying on ffmpeg's unary minus. + const sum = (y0, terms) => + terms.reduce( + (acc, t) => `${acc}${t.d < 0 ? "-" : "+"}${Math.abs(t.d)}*${t.f}`, + String(y0), + ); + const ramp = (y0, steps) => + sum(y0, steps.filter((st) => st.delta !== 0).map((st) => ({ d: st.delta, f: ease(st.at) }))); + + const { K, ROWH, RW, RX, RTOP, LOGH, LOGTOP, TALLYTOP } = g; + + const logY = ramp(0, times.map((t, i) => ({ at: t, delta: i + 1 > K ? ROWH : 0 }))); + // The curtain and the log MUST share the same eased P, or the curtain visibly + // lags the rows mid-slide and unrevealed claims flash into view. + const curtainY = ramp(LOGTOP, times.map((t, i) => ({ at: t, delta: i + 1 <= K ? ROWH : 0 }))); + const hlY = ramp(LOGTOP, times.map((t, i) => ({ at: t, delta: i > 0 && i < K ? ROWH : 0 }))); + + /** + * One lane's y, in the strip's own pixels. + * + * y(t) = r0 + Σ_k [ (a_k − b_{k−1})·gte(t,t_k) + (b_k − a_k)·ease(t_k) ] + * + * The first term is the instantaneous reposition to the next pair's starting + * row; the second is the roll itself. Both are CUMULATIVE AND SATURATING, + * which is the rail's hard rule: a gated piecewise sum flashes to y=0 for one + * frame at any boundary gap, because every gate goes false at once. + */ + const laneY = (lane) => { + const terms = []; + let prevB = 0; + lane.steps.forEach((st, i) => { + if (!st) return; + const at = times[i]; + const jump = (st.a - prevB) * lane.cellH; + const roll = (st.b - st.a) * lane.cellH; + if (jump !== 0) terms.push({ d: jump, f: `gte(t,${at.toFixed(3)})` }); + if (roll !== 0) terms.push({ d: roll, f: ease(at) }); + prevB = st.b; + }); + return sum(0, terms); + }; + + // The rail leaves by SLIDING OFF to the right, not by an enable= pop. One + // offset expression shared by every overlay, so the column moves as one + // object; `overlay`'s x is per-frame in `t`, which is what makes that + // possible at all. Cumulative and saturating, like everything else here. + const hideAt = opts.hideAt ?? null; + const OFF = hideAt == null ? "" : `+${RW + 8}*${ease(hideAt)}`; + const X = (x) => (OFF ? `'${x}${OFF}'` : String(x)); + + const i0 = firstInputIdx; + const files = [assets.chrome, assets.log, assets.curtain, assets.hl, assets.tally]; + if (assets.qr) files.push(assets.qr.path); + const inputs = files.flatMap((f) => [ + "-loop", "1", "-framerate", String(fps), "-t", bound.toFixed(3), "-i", f, + ]); + + const chain = [ + `[${i0 + 1}:v]crop=w=${RW}:h=${LOGH}:x=0:y='${logY}'[rlog]`, + `${inLabel}[${i0}:v]overlay=x=${X(RX)}:y=${RTOP}:format=yuv444:shortest=1[rr0]`, + `[rr0][rlog]overlay=x=${X(RX)}:y=${LOGTOP}:format=yuv444:shortest=1[rr1]`, + // The highlight goes UNDER the curtain: while the list is still filling, the + // row it marks has not been revealed yet, and the curtain is what hides it. + `[rr1][${i0 + 3}:v]overlay=x=${X(RX)}:y='${hlY}':format=yuv444:shortest=1[rr2]`, + `[rr2][${i0 + 2}:v]overlay=x=${X(RX)}:y='${curtainY}':format=yuv444:shortest=1[rr3]`, + ]; + + // One crop per lane out of the SINGLE tally strip. Four numbers that roll + // independently and a roster line that mostly does not, for one more input + // than the slab cost. + // + // `split` first, and it is NOT optional: a filtergraph link may be consumed + // exactly once, so five crops reading `[N:v]` is a parse error, not a + // shortcut. This is the whole reason the lanes share one PNG and still cost + // one input. + chain.push( + `[${i0 + 4}:v]split=${assets.lanes.length}${assets.lanes.map((_, j) => `[ts${j}]`).join("")}`, + ); + let lab = "[rr3]"; + assets.lanes.forEach((lane, j) => { + const isRoster = lane.kind === "roster"; + const h = isRoster ? g.ROSTERH : g.TALLYROWH; + const y = isRoster ? g.ROSTERTOP : TALLYTOP + j * g.TALLYROWH; + const x = isRoster ? RX + g.ROSTERX : RX + g.CELLX; + chain.push( + `[ts${j}]crop=w=${lane.w}:h=${h}:x=${lane.x}:y='${laneY(lane)}'[rc${j}]`, + `${lab}[rc${j}]overlay=x=${X(x)}:y=${y}:format=yuv444:shortest=1[rt${j}]`, + ); + lab = `[rt${j}]`; + }); + + // The provenance tile LAST, so the parked curtain cannot paint over it. + if (assets.qr) { + const qrY = sum(0, assets.qr.steps.map((st) => ({ d: st.delta, f: `gte(t,${st.at.toFixed(3)})` }))); + chain.push( + `[${i0 + 5}:v]crop=w=${g.TILEW}:h=${g.TILEH}:x=0:y='${qrY}'[rqr]`, + `${lab}[rqr]overlay=x=${X(RX + g.PAD)}:y=${g.TILETOP}:format=yuv444:shortest=1[rq]`, + ); + lab = "[rq]"; + } + + chain.push(`${lab}format=yuv420p[vout]`); + + return { inputs, chain: chain.join(";"), outLabel: "[vout]" }; +} + +/** + * The chrome as PNG-sequence overlays, for `render.chromeEngine: "hyperframes"`. + * + * OPT-IN, and absent it nothing below runs -- the ffmpeg chrome path is left + * byte-for-byte alone, which is the same bargain the rail was added under. + * + * The five ffmpeg traps the rail documents apply here unchanged, and two of them + * bite harder with an image sequence: + * + * * `format=yuv444` on EVERY overlay. overlay's default yuv420 subsamples + * ALPHA as well as chroma, which fringes small text -- and the band is + * nothing but small text. + * * `shortest=1` on EVERY overlay. A secondary longer than the main EXTENDS + * the output; the sequence is rendered to the same length as the concat, but + * a one-frame rounding difference either way must not change the duration. + * * One `format=yuv420p` before the encoder, once, at the end. + * + * A finite image sequence needs no `-t`: unlike `-loop 1` it ends by itself, so + * the deadlock the rail's five chained loops hit cannot happen here. + */ +export function chromeOverlayChain(render, regions, inLabel, firstInputIdx, opts = {}) { + const { outLabel = "[hfout]", final = true } = opts; + const inputs = []; + const parts = []; + let lab = inLabel; + regions.forEach((r, i) => { + inputs.push( + "-framerate", String(render.fps), + "-start_number", "1", + "-i", path.join(r.frames, "frame_%06d.png"), + ); + const idx = firstInputIdx + i; + const last = i === regions.length - 1; + const out = last && !final ? outLabel : `[hf${i}]`; + parts.push(`${lab}[${idx}:v]overlay=x=${r.x}:y=${r.y}:format=yuv444:shortest=1${out}`); + lab = out; + }); + if (final) parts.push(`${lab}format=yuv420p[vout]`); + return { + inputs, + chain: parts.join(";"), + outLabel: final ? "[vout]" : outLabel, + count: regions.length, + }; +} + +/** + * Where each rendered chrome region sits in the frame. + * + * The chart band REPLACES the footer node track rather than joining it: the + * track only moved at section handovers, which is precisely the fault the band + * exists to fix. So it takes the footer's ground and 100px more of it, and the + * picture loses that height. + */ +export function chromeRegions(render, outDir) { + const H = render.chart?.height ?? 200; + return [ + { + name: "chart", + frames: path.join(outDir, "chrome", "chart-frames"), + x: 0, + y: render.height - H, + width: contentWidth(render), + height: H, + }, + ]; +} + +/** + * The footer's stand-in when the chrome is drawn in a browser. + * + * It reserves the band's HEIGHT and draws nothing, so every segment letterboxes + * to the same picture box the overlay expects and the ground under the band is + * the palette background. `footer: null` is what switches the whole ffmpeg + * footer -- image, marker and fill bar -- off; the degenerate shape is the one + * renderFooterAssets already returns for a manifest with no nodes, so this path + * is not new. + */ +function reservedFooter(render) { + return { + footer: null, marker: null, bar: null, trackLen: 0, + footerHeight: render.chart?.height ?? 200, + trackY: 0, xs: [], x0: 0, markerRadius: 0, + }; +} + +/** + * Everything the rail chain needs that depends on the built segments. Returns + * null when the manifest does not ask for a rail — which is what keeps this + * whole feature opt-in and every existing report byte-for-byte unchanged. + */ +async function buildRailPlan(manifest, render, entries, segments, D, outDir) { + const rail = render.rail; + if (!rail || !manifest.ledger?.length) return null; + const { starts, total } = await segmentOffsets(segments, D, render.fps); + const assets = await renderRailAssets( + render, manifest.ledger, outDir, entries, manifest.provenance, + ); + // Every claim must be on the board before the closing ledger scroll reads it + // back, so the last section's spare rows are spread up to that segment. + const endIdx = entries.findIndex((e) => e.type === "scroll" || e.type === "chart"); + const endBound = endIdx > 0 ? starts[endIdx] : total; + const times = scheduleClaims(manifest.ledger, entries, starts, D, endBound); + + // The QR tile changes at the MID-DISSOLVE of every segment, instantaneously + // — a code that eased into place would spend the ease unscannable, and the + // picture is already crossfading there. + if (assets.qr) { + assets.qr.steps = entries.slice(1).map((_, i) => ({ + at: starts[i + 1] + D / 2, + delta: assets.geom.TILEH, + })); + } + + // Where the rail leaves. The closing ledger is a full-width card and the rail + // is the one thing on screen it would have to be read around, so the column + // slides off over that card's dissolve and does not come back. + const hideIdx = entries.findIndex((e) => e.hideRail); + const hideAt = hideIdx > 0 ? starts[hideIdx] : null; + + // The schedule, written down. + // + // The chart band has to sweep in step with the rail -- a playhead that tracks + // the current moment is the whole point of it -- and the only way it and the + // rail can be guaranteed to agree is for one of them to compute the schedule + // and the other to READ it. Recomputing from segment durations would be a + // second implementation of segmentOffsets(), and it would drift the first time + // the crossfade changed. Same rule as widen(): imported, never reimplemented. + await writeFile( + path.join(outDir, "schedule.json"), + JSON.stringify( + { + fps: render.fps, + transition: D, + total, + endBound, + segments: entries.map((e, i) => ({ id: e.id, type: e.type, start: starts[i] })), + claims: manifest.ledger.map((c, i) => ({ id: c.id, at: times[i], entryId: c.entryId ?? null })), + }, + null, + 2, + ) + "\n", + ); + + EMIT("note", { + message: `rail: ${manifest.ledger.length} claims, ${times.filter((_, i) => manifest.ledger[i].entryId).length} pinned, ` + + `window ${assets.geom.K} rows` + (hideAt == null ? "" : `, hides at ${hideAt.toFixed(1)}s`), + }); + return { assets, times, total, hideAt }; +} + +// ---- the end sequence ---------------------------------------------------- +// Two segment kinds that exist only to close the cut: the whole ledger read +// back in one scroll, then the same claims plotted. Both go through the SAME +// `fps=,setsar=1` and the SAME encodeArgs as every other segment — xfade +// rejects a mismatched link with "First input link parameters do not match", +// which would surface only at concat time, after every fetch has been paid for. + +async function buildScrollSegment(card, render, outDir, ledger) { + const { path: png, contentHeight, width: VW } = await renderScrollCard(card, render, ledger, outDir); + const seg = path.join(outDir, "segments", `${card.id}.mp4`); + const pal = render.palette; + const { width, height } = render; + const HH = render.headerHeight ?? 56; + // The band's height, not the manifest's footerHeight — otherwise the last + // 100px of the scroll play underneath the chart. + const FH = reservedFooterHeight(render); + const winH = height - HH - FH; + const dur = String(card.seconds); + + // clip() buys a free hold at BOTH ends, and crop's own clamping degrades an + // off-by-a-few contentHeight into a static last frame rather than an error. + const hold = card.hold ?? 2.0; + const travel = Math.max(0, contentHeight - winH); + const denom = Math.max(0.1, card.seconds - 2 * hold); + + await execFileP( + FFMPEG, + [ + "-nostdin", "-v", "error", "-y", + "-loop", "1", "-framerate", String(render.fps), "-t", dur, "-i", png, + "-f", "lavfi", "-t", dur, "-i", `color=c=${pal.bg}:s=${width}x${height}:r=${render.fps}`, + "-f", "lavfi", "-t", dur, + "-i", `anullsrc=channel_layout=stereo:sample_rate=${render.audioRate}`, + "-filter_complex", [ + `[0:v]crop=w=${VW}:h=${winH}:x=0:y='${travel}*clip((t-${hold})/${denom.toFixed(3)},0,1)'[win]`, + `[1:v][win]overlay=x=0:y=${HH}:shortest=1,fps=${render.fps},setsar=1[v]`, + ].join(";"), + "-map", "[v]", "-map", "2:a", + ...encodeArgs(render), + "-shortest", + seg, + ], + { maxBuffer: 1 << 24 }, + ); + return seg; +} + +/** + * A stacked ledger card: rows revealed in sequence by a walking curtain. + * + * The curtain is an opaque `pal.bg` rectangle that starts covering every row + * and steps down one row-height per reveal. Same device as the rail's, and for + * the same reason: the card ground is flat, so an opaque rectangle over it is + * an exact in-place wipe with no per-pixel filter. + * + * The ramp is CUMULATIVE AND SATURATING, like every other ramp here. + */ +async function buildLedgerSegment(card, render, outDir, ledger, avail) { + const geo = await renderLedgerCard(card, render, ledger, outDir, avail); + const seg = path.join(outDir, "segments", `${card.id}.mp4`); + const pal = render.palette; + const { width, height } = render; + // `seconds` is DERIVED, not authored: the pins that land claims on their own + // rows read the same clock, so a hand-set duration would silently move them. + const dur = String(card.seconds ?? ledgerSeconds(geo.rows)); + + const curtainH = height; + const curtain = path.join(outDir, "cards", `${card.id}.curtain.png`); + await execFileP("magick", [ + "-size", `${geo.width}x${curtainH}`, `xc:${pal.bg}`, curtain, + ]); + + const SLIDE = render.rail?.slide ?? 0.55; + const ease = (at) => { + const p = `clip((t-${at.toFixed(3)})/${SLIDE},0,1)`; + return `${p}*${p}*(3-2*${p})`; + }; + const y = [ + String(geo.rowsTop), + ...Array.from({ length: geo.rows }, (_, r) => `${geo.rowHeight}*${ease(ledgerRevealAt(r))}`), + ].join("+"); + + await execFileP( + FFMPEG, + [ + "-nostdin", "-v", "error", "-y", + "-f", "lavfi", "-t", dur, "-i", `color=c=${pal.bg}:s=${width}x${height}:r=${render.fps}`, + "-loop", "1", "-framerate", String(render.fps), "-t", dur, "-i", geo.path, + "-loop", "1", "-framerate", String(render.fps), "-t", dur, "-i", curtain, + "-f", "lavfi", "-t", dur, + "-i", `anullsrc=channel_layout=stereo:sample_rate=${render.audioRate}`, + "-filter_complex", [ + `[0:v][1:v]overlay=x=0:y=0:shortest=1[a]`, + `[a][2:v]overlay=x=0:y='${y}':shortest=1,fps=${render.fps},setsar=1[v]`, + ].join(";"), + "-map", "[v]", "-map", "3:a", + ...encodeArgs(render), + "-shortest", + seg, + ], + { maxBuffer: 1 << 24 }, + ); + return seg; +} + +async function buildChartSegment(card, render, outDir, ledger) { + const chart = await renderChartCard(card, render, ledger, outDir); + const seg = path.join(outDir, "segments", `${card.id}.mp4`); + const pal = render.palette; + const { width, height } = render; + const VW = cardWidth(card, render); + const dur = String(card.seconds); + + // The wipe CANNOT be `crop=w='<ramp>'` — crop's w is config-time and `t` is + // undefined there ("Error when evaluating the expression"). So: overlay the + // finished chart, then slide an opaque pal.bg rectangle rightwards off it. + // The card ground is flat pal.bg, so this is an exact in-place wipe with no + // per-pixel filter, and it draws the plot in like a plotter. + // `hold: true` -- the closing chart is a HOLD, not a reveal. + // + // The wipe existed because this card was the first and only time the viewer + // saw the numbers plotted. With the chart band drawing live under the whole + // cut, wiping it in again would re-tell a story the viewer has just watched + // happen. So the card opens on the finished plot and the seconds go to + // reading the final gap and its flags instead. + if (card.hold) { + await execFileP( + FFMPEG, + [ + "-nostdin", "-v", "error", "-y", + "-f", "lavfi", "-t", dur, "-i", `color=c=${pal.bg}:s=${width}x${height}:r=${render.fps}`, + "-loop", "1", "-framerate", String(render.fps), "-t", dur, "-i", chart.path, + "-f", "lavfi", "-t", dur, + "-i", `anullsrc=channel_layout=stereo:sample_rate=${render.audioRate}`, + "-filter_complex", + `[0:v][1:v]overlay=x=0:y=0:shortest=1,fps=${render.fps},setsar=1[v]`, + "-map", "[v]", "-map", "2:a", + ...encodeArgs(render), + "-shortest", + seg, + ], + { maxBuffer: 1 << 24 }, + ); + return seg; + } + + const wipeW = VW - chart.plotX; + const wipe = path.join(outDir, "cards", `${card.id}.wipe.png`); + await execFileP("magick", [ + "-size", `${wipeW}x${Math.round(chart.plotH)}`, `xc:${pal.bg}`, wipe, + ]); + + const wipeStart = card.wipeStart ?? 0.8; + const wipeDur = card.wipeSeconds ?? Math.max(1, card.seconds - wipeStart - 3.0); + + await execFileP( + FFMPEG, + [ + "-nostdin", "-v", "error", "-y", + "-f", "lavfi", "-t", dur, "-i", `color=c=${pal.bg}:s=${width}x${height}:r=${render.fps}`, + "-loop", "1", "-framerate", String(render.fps), "-t", dur, "-i", chart.path, + "-loop", "1", "-framerate", String(render.fps), "-t", dur, "-i", wipe, + "-f", "lavfi", "-t", dur, + "-i", `anullsrc=channel_layout=stereo:sample_rate=${render.audioRate}`, + "-filter_complex", [ + `[0:v][1:v]overlay=x=0:y=0:shortest=1[a]`, + `[a][2:v]overlay=x='${chart.plotX}+${wipeW}*clip((t-${wipeStart})/${wipeDur.toFixed(3)},0,1)'` + + `:y=${Math.round(chart.plotY)}:shortest=1,fps=${render.fps},setsar=1[v]`, + ].join(";"), + "-map", "[v]", "-map", "3:a", + ...encodeArgs(render), + "-shortest", + seg, + ], + { maxBuffer: 1 << 24 }, + ); + return seg; +} + +// Crossfade every segment into the next. This is a full re-encode of the +// timeline — the concat demuxer can only stream-copy hard cuts — so --no-xfade +// stays available for quick iteration. +async function concatWithXfade(segments, render, outPath, railPlan, chrome = null) { + const D = render.transition ?? 0.5; + const durs = []; + for (const s of segments) durs.push(await probeDuration(s, render.fps)); + + const inputs = segments.flatMap((s) => ["-i", s]); + const parts = []; + let vlab = "[0:v]"; + let alab = "[0:a]"; + let acc = durs[0]; + + for (let i = 1; i < segments.length; i += 1) { + const off = acc - D; + parts.push(`${vlab}[${i}:v]xfade=transition=fade:duration=${D}:offset=${off.toFixed(3)}[v${i}]`); + parts.push(`${alab}[${i}:a]acrossfade=d=${D}:c1=tri:c2=tri[a${i}]`); + vlab = `[v${i}]`; + alab = `[a${i}]`; + acc = acc + durs[i] - D; + } + + // The rail attaches to the LAST xfade node, so it runs after every dissolve + // and sees an absolute, continuous `t`. One encode, not two. + // + // When the chrome is rendered rather than drawn, the PNG regions go on FIRST + // and the rail chain reads their output. Not the other way round: the rail + // chain ends in `format=yuv420p`, and overlaying an alpha sequence onto + // yuv420p is the fringing trap the rail already documents, one layer later. + const chromeIn = chrome ?? null; + const railIn = chromeIn ? chromeIn.outLabel : vlab; + const rc = railPlan + ? railFilterChain( + render.rail, railPlan.assets, railPlan.times, render, + railIn, segments.length, railPlan.total + 2, + { hideAt: railPlan.hideAt }, + ) + : null; + const railInputs = rc ? rc.inputs.filter((a) => a === "-i").length : 0; + const hf = chromeIn + ? chromeOverlayChain(render, chromeIn.regions, vlab, segments.length + railInputs, { + outLabel: chromeIn.outLabel, + final: !rc, + }) + : null; + if (hf) parts.push(hf.chain); + if (rc) parts.push(rc.chain); + + const tail = rc ? rc.outLabel : hf ? hf.outLabel : vlab; + + await execFileP( + FFMPEG, + [ + "-nostdin", "-v", "error", "-y", + ...inputs, + ...(rc ? rc.inputs : []), + ...(hf ? hf.inputs : []), + "-filter_complex", parts.join(";"), + "-map", tail, "-map", alab, + ...encodeArgs(render), + outPath, + ], + { maxBuffer: 1 << 26 }, + ); +} + +/** + * Run the rail chain over an already-concatenated file. + * + * Two callers need this. `--rail-only` iterates on the rail in seconds instead + * of re-running the whole concat; and `--no-xfade` has no choice, because + * concatHardCut is `-c copy` and a stream-copy mux cannot host a filtergraph + * at all. + */ +async function applyRail(inPath, outPath, render, railPlan, preview) { + const rc = railFilterChain( + render.rail, railPlan.assets, railPlan.times, render, + preview ? "[base]" : "[0:v]", 1, railPlan.total + 2, + { hideAt: railPlan.hideAt }, + ); + const parts = []; + if (preview) { + // -ss restarts `t` near zero, which would put every absolute-time ramp in + // the wrong place — the rail would look broken while being correct. Shift + // the timestamps back to where the expressions think they are, then rebase + // them so the preview file still starts at 0. + parts.push(`[0:v]setpts=PTS+${preview.start.toFixed(3)}/TB[base]`); + } + parts.push(rc.chain); + const tail = preview ? "[vshift]" : rc.outLabel; + if (preview) parts.push(`${rc.outLabel}setpts=PTS-STARTPTS[vshift]`); + + await execFileP( + FFMPEG, + [ + "-nostdin", "-v", "error", "-y", + ...(preview ? ["-ss", String(preview.start), "-t", String(preview.dur)] : []), + "-i", inPath, + ...rc.inputs, + "-filter_complex", parts.join(";"), + "-map", tail, "-map", "0:a", + ...encodeArgsVideoOnly(render), + outPath, + ], + { maxBuffer: 1 << 26 }, + ); +} + +// ---- chapter markers ----------------------------------------------------- +// A compilation like this is a reference document as much as a video: the report +// cites moments, and a viewer wants to jump to them. Every clip therefore becomes +// a chapter. Offsets are derived exactly the way concatWithXfade derives its xfade +// offsets, so they stay correct for both crossfaded and hard-cut timelines. +// +// ffmetadata is a line-based format where =, ;, # and \ are structural, so a +// title carrying any of them has to be escaped or the file silently mis-parses. +const ffmetaEscape = (s) => String(s).replace(/([=;#\\])/g, "\\$1").replace(/\n/g, " "); + +export async function segmentOffsets(segments, D, fps) { + const durs = []; + for (const s of segments) durs.push(await probeDuration(s, fps)); + const starts = []; + let acc = 0; + for (let i = 0; i < durs.length; i += 1) { + starts.push(acc); + acc += durs[i] - (i < durs.length - 1 ? D : 0); + } + return { starts, total: acc }; +} + +async function chapterTitle(entry, index, provenance) { + if (entry.chapter) return entry.chapter; + if (entry.type !== "clip") return entry.title ?? entry.heading ?? `Card ${index + 1}`; + try { + const meta = await videoMeta(entry.video, entry.channel ?? provenance.channelSlug, { siteChannel: entry.siteChannel, siteVideo: entry.siteVideo }); + const d = String(meta.uploadDate ?? ""); + const date = /^\d{8}$/.test(d) ? `${d.slice(0, 4)}-${d.slice(4, 6)}-${d.slice(6, 8)}` : d; + const title = String(meta.title ?? entry.video); + return `${date} — ${title.length > 60 ? `${title.slice(0, 57)}…` : title}`.trim(); + } catch { + return `${index + 1}. ${entry.video}`; + } +} + +async function muxChapters(finalPath, entries, segments, D, outDir, provenance, fps) { + if (segments.length < 2) return; + const { starts, total } = await segmentOffsets(segments, D, fps); + const lines = [";FFMETADATA1", ""]; + for (let i = 0; i < entries.length; i += 1) { + // Land just PAST the crossfade, so the marker opens on the incoming clip + // rather than on the outgoing one mid-dissolve. + const start = i === 0 ? 0 : starts[i] + D; + const end = i === entries.length - 1 ? total : starts[i + 1] + D; + lines.push( + "[CHAPTER]", + "TIMEBASE=1/1000", + `START=${Math.round(start * 1000)}`, + `END=${Math.round(end * 1000)}`, + `title=${ffmetaEscape(await chapterTitle(entries[i], i, provenance))}`, + "", + ); + } + const metaPath = path.join(outDir, "chapters.ffmeta"); + await writeFile(metaPath, lines.join("\n"), "utf8"); + + // Stream copy — adding chapters must never re-encode the finished timeline. + const tmp = finalPath.replace(/\.mp4$/, ".chapters.mp4"); + await execFileP( + FFMPEG, + ["-nostdin", "-v", "error", "-y", "-i", finalPath, "-i", metaPath, + "-map", "0", "-map_metadata", "0", "-map_chapters", "1", "-c", "copy", tmp], + { maxBuffer: 1 << 24 }, + ); + await rename(tmp, finalPath); + EMIT("chapters", { n: entries.length, file: path.basename(metaPath) }); +} + +async function concatHardCut(segments, outDir, outPath) { + const listPath = path.join(outDir, "concat.txt"); + await writeFile(listPath, segments.map((s) => `file '${s}'`).join("\n") + "\n", "utf8"); + await execFileP( + FFMPEG, + ["-nostdin", "-v", "error", "-y", "-f", "concat", "-safe", "0", + // The concat demuxer stitches per-file timestamps; without generated PTS a + // stream copy can hand the next stage a discontinuous timeline, which the + // rail's absolute-time expressions would then read off by that much. + "-fflags", "+genpts", + "-i", listPath, "-c", "copy", outPath], + { maxBuffer: 1 << 24 }, + ); +} + +// A hard-cut concat and a crossfaded one are different lengths, so a cached +// prerail from one is a wrong base for the other. Keeping them in separate files +// means the mode can be switched without a stale-cache trap -- and without the +// length assertion below having to be the thing that explains it. +const prerailPath = (outDir, slug, D) => + path.join(outDir, `${slug}.prerail${D === 0 ? "-hardcut" : ""}.mp4`); + +// The finished timeline must be exactly as long as segmentOffsets says. Anything +// else means a filter changed the length behind our backs. +async function assertConcatLength(file, expected, fps, what) { + const got = await probeDuration(file, fps); + if (Math.abs(got - expected) > 1.5 / fps) { + throw new Error( + `${what}: duration ${got.toFixed(3)}s but the timeline is ${expected.toFixed(3)}s ` + + `(${((got - expected) * fps).toFixed(1)} frames out)` + + (/prerail/.test(what) ? " — delete it and let this rebuild it" : ""), + ); + } +} + +/** + * Build a manifest into a video. + * + * Exported so umtool's driver runs the SAME code the CLI does. It is still + * SPAWNED rather than imported by the app: a 40-minute chain of yt-dlp and + * ffmpeg inside a request handler has no cancellation story, and a runaway + * grandchild would outlive the request that started it. + */ +export async function buildVideo({ manifestPath, opts = {}, out, only, fetchOnly } = {}) { + const variant = opts.variant ?? "sourced"; + const whole = JSON.parse(await readFile(manifestPath, "utf8")); + const manifest = selectVariant(whole, variant); + const { render, provenance } = manifest; + + // The manifest already records which archive it was built against, so a clone + // with no corpus needs no extra configuration to read cue windows. + CUES = createCueSource({ + siteOrigin: opts.siteOrigin ?? process.env.SITE_ORIGIN ?? siteOriginFromManifest(whole), + resolveSiteIds: opts.resolveSiteIds === true, + prefer: opts.cueSource ?? "auto", + log: (m) => EMIT("log", { message: m }), + }); + const outRoot = out ?? path.join(path.dirname(path.resolve(manifestPath)), "out"); + const dirs = variantPaths(outRoot, manifest.slug, variant); + const outDir = dirs.dir; + + await mkdir(dirs.rawDir, { recursive: true }); + for (const d of ["cards", "segments", "qr"]) { + await mkdir(path.join(outDir, d), { recursive: true }); + } + + // Fetch one clip's window and stop. This is what the clip bench's "fetch 20s + // more" runs, so a bench fetch and a build fetch can never disagree about + // naming, format selection, the VP9 trap or the Rumble HLS retry. + // Reads the WHOLE manifest, not the variant's view of it: a clip bench fetch + // is about a moment in the corpus, and which cut happens to carry it is + // beside the point. + if (fetchOnly) { + let entry = whole.timeline.find((e) => e.id === fetchOnly); + // `!== "clip"`, not `=== "card"`. The timeline's vocabulary is OPEN -- one + // real manifest carries `scroll` and `chart` entries -- and the card-only + // check sent `undefined` into the fetcher for either of those. + if (entry && entry.type !== "clip") { + throw new Error(`${fetchOnly} is a ${entry.type ?? "non-clip"} entry, not a clip`); + } + if (!entry) { + // A LEDGER CLAIM. Adjudicating one means listening around the moment, and + // most of the ledger is cited by no clip at all -- so the claim page asks + // for a window the timeline has no entry for. It is fetched through this + // same path so the file lands in clips-raw under the build's own naming, + // inherits the format pin and the Rumble HLS retry, and is REUSED by a + // later build rather than fetched a second time. + const claim = (whole.ledger ?? []).find((e) => e.id === fetchOnly); + if (!claim) throw new Error(`no timeline entry or ledger claim with id ${fetchOnly}`); + if (!claim.video) throw new Error(`ledger claim ${fetchOnly} has no \`video\` to fetch`); + const at = Number(claim.cite); + if (!Number.isFinite(at)) throw new Error(`ledger claim ${fetchOnly} has no \`cite\` second`); + // A claim is a MOMENT, not a window: the pad is the whole point, so the + // entry is a hair either side of the cite and --pad does the rest. + entry = { + id: claim.id, + video: claim.video, + channel: claim.channel ?? null, + start: Math.max(0, at - 1), + end: at + 1, + }; + } + const meta = await videoMeta(entry.video, entry.channel ?? provenance.channelSlug, { siteChannel: entry.siteChannel, siteVideo: entry.siteVideo }); + const r = await fetchClip(entry, meta, render, dirs.rawDir, opts); + EMIT("done", { out: r.path, fetchStart: r.fetchStart, cached: r.cached }); + return { out: r.path, failures: [] }; + } + + // Footer chrome is shared by every clip, so build it once up front. + const hyper = render.chromeEngine === "hyperframes"; + const chrome = hyper + ? reservedFooter(render) + : await renderFooterAssets(render, manifest.timelineNodes, outDir); + + const entries = manifest.timeline.filter((e) => !only || e.id === only); + if (only && !entries.length) throw new Error(`no timeline entry with id ${only}`); + + // Read once, at the ROOT: a source's state is a fact about the manifest, not + // about a variant. A stacked ledger card says why each claim is text rather + // than footage, and this is where that answer comes from. + const availability = new Map( + ( + await readFile(path.join(dirs.root, "availability.json"), "utf8").then( + (j) => JSON.parse(j).sources ?? [], + () => [], + ) + ).flatMap((src) => (src.claims ?? []).map((id) => [id, src.state])), + ); + const segments = []; + const failures = []; + + const D = opts.noXfade || (render.transition ?? 0.5) === 0 ? 0 : render.transition ?? 0.5; + + // Retro-fit chapters onto an already-built file without re-encoding it. The + // per-clip segments are still on disk, which is all the offsets need. + if (opts.chaptersOnly) { + const finalPath = dirs.final; + const segs = entries.map((e) => path.join(outDir, "segments", `${e.id}.mp4`)); + for (const seg of segs) { + if (!(await exists(seg))) + throw new Error(`--chapters-only needs ${seg}, which is missing — run a full build first`); + } + await muxChapters(finalPath, entries, segs, D, outDir, provenance, render.fps); + return { out: finalPath, failures: [] }; + } + + // Re-run the rail over a cached concat instead of rebuilding the timeline. + // The rail is the part that gets iterated on; the 40-minute concat is not. + if (opts.railOnly) { + if (!render.rail) throw new Error("--rail-only needs render.rail in the manifest"); + const finalPath = dirs.final; + const prerail = prerailPath(outDir, manifest.slug, D); + const segs = entries.map((e) => path.join(outDir, "segments", `${e.id}.mp4`)); + for (const seg of segs) { + if (!(await exists(seg))) + throw new Error(`--rail-only needs ${seg}, which is missing — run a full build first`); + } + if (!(await exists(prerail))) { + EMIT("concat", { mode: D === 0 ? "hardcut" : "xfade", n: segs.length }); + if (D === 0) await concatHardCut(segs, outDir, prerail); + else await concatWithXfade(segs, render, prerail, null); + } + const railPlan = await buildRailPlan(manifest, render, entries, segs, D, outDir); + await assertConcatLength(prerail, railPlan.total, render.fps, + `cached ${path.basename(prerail)}`); + const out = opts.preview + ? path.join(outDir, `${manifest.slug}.preview.mp4`) + : finalPath; + await applyRail(prerail, out, render, railPlan, opts.preview ?? null); + if (!opts.preview) { + await assertConcatLength(out, railPlan.total, render.fps, "rail build"); + // applyRail re-encodes, so the chapters muxed onto the previous final are + // gone. Put them back, or --rail-only quietly ships a chapterless cut. + if (!opts.noChapters) { + await muxChapters(out, entries, segs, D, outDir, provenance, render.fps); + } + } + EMIT("done", { out, failures: [] }); + return { out, failures: [] }; + } + + EMIT("start", { title: manifest.title, entries: entries.length, out: outDir }); + for (let i = 0; i < entries.length; i += 1) { + const entry = entries[i]; + try { + if (entry.type === "card") { + EMIT("card", { id: entry.id, i, n: entries.length }); + segments.push(await buildCardSegment(entry, render, outDir, manifest.timelineNodes)); + } else if (entry.type === "scroll" || entry.type === "chart" || entry.type === "ledger") { + EMIT("card", { id: entry.id, i, n: entries.length }); + if (!manifest.ledger?.length) + throw new Error(`${entry.id} is type:${entry.type} but the manifest has no ledger[]`); + segments.push( + entry.type === "scroll" + ? await buildScrollSegment(entry, render, outDir, manifest.ledger) + : entry.type === "chart" + ? await buildChartSegment(entry, render, outDir, manifest.ledger) + : await buildLedgerSegment(entry, render, outDir, manifest.ledger, availability), + ); + } else { + const meta = await videoMeta(entry.video, entry.channel ?? provenance.channelSlug, { siteChannel: entry.siteChannel, siteVideo: entry.siteVideo }); + EMIT("clip", { + id: entry.id, i, n: entries.length, video: entry.video, + section: entry.section, sectionEnter: !!entry.sectionEnter, + }); + segments.push( + await buildClipSegment( + entry, meta, render, dirs, opts, chrome, manifest.timelineNodes, provenance, + ), + ); + } + EMIT("segment", { id: entry.id, path: segments[segments.length - 1] }); + } catch (err) { + // Without --continue-on-error a dead source at entry 14 of 19 throws away + // the thirteen fetches already paid for. With it, everything buildable is + // built and the run reports what was not. + if (!opts.continueOnError) throw err; + const message = err?.message ?? String(err); + failures.push({ id: entry.id, message }); + EMIT("entry-failed", { id: entry.id, message }); + } + } + + if (only) { + EMIT("done", { out: segments[0], failures }); + return { out: segments[0], failures }; + } + + // A timeline that silently lost a clip is a worse outcome than no file at all: + // the finished video would look complete and be missing a citation. So the + // segments are kept (they cost the fetches) and the concat is refused. + if (failures.length) { + EMIT("note", { + message: `refusing to concat: ${failures.length} of ${entries.length} entries failed ` + + `(${failures.map((f) => f.id).join(", ")})`, + }); + return { out: null, failures }; + } + + const final = dirs.final; + const railPlan = opts.noRail ? null : await buildRailPlan(manifest, render, entries, segments, D, outDir); + + // The rendered chrome, if this manifest asks for it. Absent, `chromePlan` is + // null and every line below behaves exactly as it did -- which is the claim + // the MD5 check tests. + let chromePlan = null; + if (hyper) { + const regions = chromeRegions(render, outDir); + for (const r of regions) { + if (!(await exists(path.join(r.frames, "frame_000001.png")))) { + throw new Error( + `render.chromeEngine is "hyperframes" but ${r.name} has no frames at ${r.frames}. ` + + `Run compose-chrome.mjs --region ${r.name} --render first.`, + ); + } + } + chromePlan = { regions, outLabel: "[hfout]" }; + EMIT("note", { message: `chrome: ${regions.map((r) => `${r.name} ${r.width}x${r.height}`).join(", ")} as png-sequence` }); + } + + // `transition: 0` is a real editorial choice, not just a speed knob: hard cuts + // hit harder on a compilation whose point is repetition. Honouring it here keeps + // the manifest the source of truth, so a rebuild does not silently re-add fades. + EMIT("concat", { mode: D === 0 ? "hardcut" : "xfade", n: segments.length }); + if (D === 0) { + if (chromePlan) { + throw new Error( + 'render.chromeEngine "hyperframes" needs a filtergraph, and `transition: 0` concatenates with ' + + "-c copy, which cannot host one. Give the manifest a transition, or drop chromeEngine.", + ); + } + // concatHardCut is `-c copy`, which cannot host a filtergraph, so the rail + // has to be a second pass here whether we like it or not. + const prerail = prerailPath(outDir, manifest.slug, D); + await concatHardCut(segments, outDir, railPlan ? prerail : final); + if (railPlan) { + await assertConcatLength(prerail, railPlan.total, render.fps, "hard-cut concat"); + await applyRail(prerail, final, render, railPlan, null); + } + } else { + await concatWithXfade(segments, render, final, railPlan, chromePlan); + } + + // Length is the canary for the two ways a rail input can go wrong: a file + // LONGER than the timeline means a strip outran the main (a missing + // shortest=1), and a hang means an unbounded -loop 1. + if (railPlan) await assertConcatLength(final, railPlan.total, render.fps, "rail build"); + + if (!opts.noChapters) await muxChapters(final, entries, segments, D, outDir, provenance, render.fps); + + const { stdout } = await execFileP(FFPROBE, [ + "-v", "error", "-show_entries", "format=duration,size", + "-of", "default=noprint_wrappers=1", final, + ]); + const probe = Object.fromEntries( + stdout.trim().split("\n").map((l) => l.split("=")), + ); + EMIT("done", { out: final, duration: Number(probe.duration), size: Number(probe.size) }); + return { out: final, failures }; +} + +async function main() { + const argv = process.argv.slice(2); + const manifestPath = argv.find((a) => !a.startsWith("--")); + if (!manifestPath) { + console.error( + "usage: build-video.mjs <manifest.json> [--out <dir>] [--variant sourced|full]\n" + + " [--only <id>] [--fetch-only <id>]\n" + + " [--pad <s>] [--skip-fetch] [--no-xfade] [--no-chapters] [--chapters-only]\n" + + " [--progress ndjson] [--continue-on-error] [--no-reuse]\n" + + " [--no-rail] [--rail-only] [--preview <start> <dur>]\n" + + " [--site-origin <url>] [--resolve-site-ids] [--cue-source auto|local|http]", + ); + process.exit(2); + } + const flag = (n) => { + const i = argv.indexOf(n); + return i >= 0 ? argv[i + 1] : undefined; + }; + setProgressMode(flag("--progress") ?? "human"); + + const padArg = flag("--pad"); + const opts = { + variant: flag("--variant") ?? "sourced", + skipFetch: argv.includes("--skip-fetch"), + continueOnError: argv.includes("--continue-on-error"), + noXfade: argv.includes("--no-xfade"), + noChapters: argv.includes("--no-chapters"), + chaptersOnly: argv.includes("--chapters-only"), + noReuse: argv.includes("--no-reuse"), + noRail: argv.includes("--no-rail"), + railOnly: argv.includes("--rail-only"), + pad: padArg === undefined ? undefined : Number(padArg), + siteOrigin: flag("--site-origin"), + resolveSiteIds: argv.includes("--resolve-site-ids"), + cueSource: flag("--cue-source"), + }; + const pv = argv.indexOf("--preview"); + if (pv >= 0) { + opts.preview = { start: Number(argv[pv + 1]), dur: Number(argv[pv + 2]) }; + opts.railOnly = true; + if (!Number.isFinite(opts.preview.start) || !Number.isFinite(opts.preview.dur)) { + console.error("--preview takes <start> <dur> in seconds"); + process.exit(2); + } + } + + const { failures } = await buildVideo({ + manifestPath, + opts, + out: flag("--out"), + only: flag("--only"), + fetchOnly: flag("--fetch-only"), + }); + // Non-zero on a partial run, so a caller that ignores the events still learns + // the build did not produce what was asked for. + if (failures.length) process.exit(1); +} + +if (import.meta.url === `file://${process.argv[1]}`) { + main().catch((err) => { + EMIT("error", { message: err?.message ?? String(err) }); + console.error(err.message ?? err); + process.exit(1); + }); +} diff --git a/scripts/report-to-video/check-availability.mjs b/scripts/report-to-video/check-availability.mjs @@ -0,0 +1,182 @@ +#!/usr/bin/env node +// check-availability.mjs — is every source this manifest cites still fetchable? +// +// This is the one fact about a report video that goes stale in BOTH directions +// and that nothing on disk records. A source can be deleted between writing the +// manifest and building it (so a 40-minute build dies at clip 14 having paid for +// thirteen fetches), and a source can come back (so a manifest annotated "gone" +// stays wrong). Neither is visible until a build runs. +// +// So it runs first, it runs cheap, and it writes down when it ran. `--simulate` +// resolves formats without downloading a byte: a few seconds for a whole +// manifest against twenty-odd minutes for the build it protects. +// +// On the CLI: +// node scripts/report-to-video/check-availability.mjs <manifest.json> [--json] +// +// Options: +// --json Print the report as JSON instead of a table +// --out <dir> Output root (default: manifest dir + /out) +// --allow-missing Exit 0 even when a source is gone (report only) +// --max-age <days> Reuse a recorded verdict younger than this (default: 0) + +import { execFile } from "node:child_process"; +import { promisify } from "node:util"; +import { mkdir, readFile, writeFile } from "node:fs/promises"; +import path from "node:path"; + +import { DEFAULT_CHANNELS_DIR } from "./cues.mjs"; + +const execFileP = promisify(execFile); + +const YTDLP = process.env.YTDLP_BIN ?? "yt-dlp"; +// Resolved relative to the repo (see cues.mjs) rather than an absolute path in +// one machine's home directory, which every other clone would miss. +const CHANNELS_DIR = DEFAULT_CHANNELS_DIR; + +// yt-dlp says why in prose, and the distinction matters editorially: a private +// or removed video needs the clip converting to a quote card, while a network +// blip needs a retry. Anything unrecognised stays `maybe_missing` rather than +// being called deleted -- claiming a source is gone when it is not is the more +// expensive mistake, because it invites deleting a citation. +function classify(stderr) { + const t = String(stderr ?? ""); + if (/Private video|private/i.test(t)) return "private"; + if (/removed by the uploader|has been removed|no longer available|Video unavailable|does not exist|410/i.test(t)) + return "deleted"; + if (/age.?restrict|Sign in to confirm|confirm your age/i.test(t)) return "restricted"; + if (/members-only|join this channel/i.test(t)) return "members-only"; + if (/geo|not available in your country/i.test(t)) return "geo-blocked"; + return "maybe_missing"; +} + +async function cueMeta(videoId, channelSlug) { + const p = path.join(CHANNELS_DIR, channelSlug, "data", videoId, "transcript.cues.json"); + const d = JSON.parse(await readFile(p, "utf8")); + return { title: d.title, webpageUrl: d.webpageUrl, duration: d.duration }; +} + +export async function checkAvailability(manifestPath, { outDir, maxAgeDays = 0 } = {}) { + const manifest = JSON.parse(await readFile(manifestPath, "utf8")); + const slug = manifest.provenance?.channelSlug; + const dir = outDir ?? path.join(path.dirname(path.resolve(manifestPath)), "out"); + const file = path.join(dir, "availability.json"); + + // A clip may name its own channel: the same streamer's VODs are mirrored + // across more than one archive, and the same id under a different slug is a + // different file. So the unit of work is (channel, video), never video alone. + const wanted = new Map(); + const want = (channel, video) => { + const key = `${channel}/${video}`; + if (!wanted.has(key)) wanted.set(key, { key, channel, video, clips: [], claims: [] }); + return wanted.get(key); + }; + for (const e of manifest.timeline ?? []) { + if (e.type !== "clip") continue; + want(e.channel ?? slug, e.video).clips.push(e.id); + } + // LEDGER SOURCES TOO, not just the clipped ones. A cut that stacks its + // unclipped claims onto cards has to say WHY each one is a line of text + // rather than footage, and "the upload is gone" and "we did not cut it" are + // different sentences. Guessing which is which is how a live source ends up + // labelled deleted on screen. + for (const c of manifest.ledger ?? []) { + if (!c.video) continue; + want(c.channel ?? slug, c.video).claims.push(c.id); + } + + const prev = await readFile(file, "utf8").then( + (s) => JSON.parse(s), + () => ({ sources: [] }), + ); + const prevBy = new Map((prev.sources ?? []).map((s) => [s.key, s])); + const freshMs = maxAgeDays * 86400_000; + + const sources = []; + for (const w of wanted.values()) { + const was = prevBy.get(w.key); + if (freshMs > 0 && was?.checkedAt && Date.now() - Date.parse(was.checkedAt) < freshMs) { + sources.push({ ...was, clips: w.clips, claims: w.claims, reused: true }); + continue; + } + + let meta; + try { + meta = await cueMeta(w.video, w.channel); + } catch { + // No cue file is a DIFFERENT failure from a dead source, and it is the one + // the Rumble two-ids trap produces: the manifest names the MCP video id + // while the cues live under the URL slug. Build would die here too, so it + // is reported here rather than discovered twenty minutes in. + sources.push({ + ...w, ok: false, state: "no-cues", checkedAt: new Date().toISOString(), + error: `no transcript.cues.json under ${w.channel}/data/${w.video}`, + }); + continue; + } + + try { + await execFileP( + YTDLP, + ["--ignore-config", "--no-playlist", "--simulate", "--quiet", "--no-warnings", "--", meta.webpageUrl], + { maxBuffer: 1 << 24 }, + ); + sources.push({ + ...w, ok: true, state: "ok", title: meta.title, url: meta.webpageUrl, + checkedAt: new Date().toISOString(), error: null, + }); + } catch (err) { + const stderr = err?.stderr ?? err?.message ?? ""; + sources.push({ + ...w, ok: false, state: classify(stderr), title: meta.title, url: meta.webpageUrl, + checkedAt: new Date().toISOString(), error: String(stderr).trim().split("\n").slice(-3).join(" "), + }); + } + } + + const report = { manifest: path.resolve(manifestPath), checkedAt: new Date().toISOString(), sources }; + await mkdir(dir, { recursive: true }); + await writeFile(file, JSON.stringify(report, null, 2) + "\n", "utf8"); + return { ...report, file }; +} + +async function main() { + const argv = process.argv.slice(2); + const manifestPath = argv.find((a) => !a.startsWith("--")); + if (!manifestPath) { + console.error("usage: check-availability.mjs <manifest.json> [--json] [--out <dir>] [--allow-missing]"); + process.exit(2); + } + const flag = (n) => { + const i = argv.indexOf(n); + return i >= 0 ? argv[i + 1] : undefined; + }; + const report = await checkAvailability(manifestPath, { + outDir: flag("--out"), + maxAgeDays: Number(flag("--max-age") ?? 0), + }); + + if (argv.includes("--json")) { + console.log(JSON.stringify(report, null, 2)); + } else { + for (const s of report.sources) { + const mark = s.ok ? "ok " : "GONE"; + console.log( + `${mark} ${s.key.padEnd(40)} ${String(s.state).padEnd(14)} ` + + `${s.clips.length} clip(s)${s.reused ? " (cached)" : ""}`, + ); + if (!s.ok && s.error) console.log(` ${s.error}`); + } + const bad = report.sources.filter((s) => !s.ok).length; + console.log(`\n${report.sources.length} source(s), ${bad} unavailable -> ${report.file}`); + } + + if (report.sources.some((s) => !s.ok) && !argv.includes("--allow-missing")) process.exit(1); +} + +if (import.meta.url === `file://${process.argv[1]}`) { + main().catch((err) => { + console.error(err.message ?? err); + process.exit(1); + }); +} diff --git a/scripts/report-to-video/compose-chrome.mjs b/scripts/report-to-video/compose-chrome.mjs @@ -0,0 +1,599 @@ +#!/usr/bin/env node +// Build the chrome regions as HyperFrames compositions and render them to +// lossless PNG sequences the ffmpeg pass overlays. +// +// --------------------------------------------------------------------------- +// Why the chrome is a browser and the picture is not +// --------------------------------------------------------------------------- +// HyperFrames pre-extracts source video to JPEG q95 before compositing, which is +// unacceptable when the picture IS the cited evidence -- the whole argument of +// the cut is that you are watching the man say it. So FOOTAGE NEVER ENTERS +// CHROME. Each chrome region is its own small composition over a transparent +// background, rendered to RGBA PNG, and composited by the existing ffmpeg pass. +// That is ~37% of full-frame pixels rather than 100%. +// +// --------------------------------------------------------------------------- +// Why the sweep is one clip-path and not a dash offset per series +// --------------------------------------------------------------------------- +// The obvious build animates every path's stroke-dashoffset. It looks right for +// the strokes and wrong for everything else: the gap band between the two +// totals is a filled polygon with no stroke to offset, so it appears at full +// width the moment it fades in and the chart is already showing you an answer +// the playhead has not reached. One clip rect over the whole plot makes the +// reveal a property of the SWEEP rather than of each mark, and nothing can get +// ahead of it. +// +// --------------------------------------------------------------------------- +// Why the playhead is driven by out/schedule.json and not by dates +// --------------------------------------------------------------------------- +// The band has to move every frame and be in the right place when a rail row +// lands. Only the build knows when that is -- it depends on segment durations +// and the crossfade -- so the build writes the schedule and this reads it. +// Recomputing it here would be a second implementation of segmentOffsets() and +// would drift the first time the transition changed. +import { copyFile, mkdir, readFile, writeFile } from "node:fs/promises"; +import { execFile } from "node:child_process"; +import { promisify } from "node:util"; +import path from "node:path"; + +import { ledgerTotals, dateKey } from "./ledger-totals.mjs"; +import { selectVariant } from "./build-video.mjs"; + +const run = promisify(execFile); + +const esc = (s) => + String(s ?? "").replace(/&/g, "&amp;").replace(/</g, "&lt;").replace(/>/g, "&gt;"); + +const num = (v) => (Number.isInteger(v) ? String(v) : v.toFixed(1).replace(/\.0$/, "")); + +/** Days since epoch, so the x scale is linear in real time and not in claims. */ +const dayOf = (d) => Date.parse(`${dateKey(d)}T00:00:00Z`) / 86400000; + +// --------------------------------------------------------------------------- +// The chart band. +// --------------------------------------------------------------------------- + +const DEFAULT_CHART = { + height: 200, + yMax: 30, + from: "2020-01-01", + to: "2026-12-31", + statedColor: "#C55F9C", + impliedColor: "#EDF0EC", + impliedWidth: 4.5, + // NOT the palette's amber. #E8A33F sits at ΔE 12.4 from the coffee series' + // #D2732F at NORMAL vision -- below the 15 floor, so a flag badge beside a + // coffee mark was hard to tell from the coffee mark. #E0E24A adds no new + // worst pair on either the CVD or the normal-vision all-pairs check: the two + // worst pairs are identical with and without it. + flagColor: "#E0E24A", + gapFill: "#C55F9C", + gapOpacity: 0.1, + series: [ + { scope: "media", color: "#22AB83", dash: null }, + { scope: "coffee", color: "#D2732F", dash: "7 4" }, + { scope: "publica", color: "#6A82DC", dash: "2 4" }, + ], +}; + +const SCOPE_LABEL = { media: "The Quartering", coffee: "Coffee Brand", publica: "The Publica" }; + +/** The predicates that get a glyph on the mark. See the note at the call site. */ +const BADGED = new Set([ + "contradicts_component", + "same_day_conflict", + "self_negating", + "not_his_number", + // The same people, staff one month and contractors the next. It earns a + // badge for the same reason self_negating does: the mark is drawn at a + // height the claim's own words undercut. + "status_flip", +]); + +/** + * A step path. The figure he gave HOLDS until he gives another one, so the + * segment between two claims is flat and the change is vertical. A smooth line + * would draw a fortnight of intermediate headcounts nobody ever claimed. + */ +function stepPath(points, X, Y) { + if (!points.length) return ""; + const d = [`M${X(points[0].date).toFixed(1)},${Y(points[0].value).toFixed(1)}`]; + for (let i = 1; i < points.length; i += 1) { + d.push(`L${X(points[i].date).toFixed(1)},${Y(points[i - 1].value).toFixed(1)}`); + d.push(`L${X(points[i].date).toFixed(1)},${Y(points[i].value).toFixed(1)}`); + } + return d.join(" "); +} + +/** The same walk, but returning the polyline vertices — the gap band needs them. */ +function stepVerts(points, X, Y) { + const out = []; + points.forEach((p, i) => { + if (i > 0) out.push([X(p.date), Y(points[i - 1].value)]); + out.push([X(p.date), Y(p.value)]); + }); + return out; +} + +export function chartBandHtml(manifest, totals, schedule, opts = {}) { + const fonts = opts.fonts ?? null; + const render = manifest.render; + const cfg = { ...DEFAULT_CHART, ...(render.chart ?? {}) }; + const W = opts.width ?? render.width - (render.rail?.width ?? 0); + const H = cfg.height; + const pal = render.palette; + const DUR = opts.duration ?? schedule.total; + + const L = 92, R = 1120, T = 26, B = 150; + const AXIS = 158; + const d0 = dayOf(cfg.from), d1 = dayOf(cfg.to); + const X = (date) => L + ((dayOf(date) - d0) / (d1 - d0)) * (R - L); + // Headroom. The implied total peaks at exactly cfg.yMax in this corpus, and a + // series drawn along the top gridline reads as clipped rather than as a peak. + const peak = Math.max( + cfg.yMax, + ...(totals.series.implied ?? []).map((p) => p.value), + ...(totals.series.stated ?? []).map((p) => p.value), + ); + const yTop = peak > cfg.yMax - 1 ? Math.ceil(peak * 1.1) : cfg.yMax; + const Y = (v) => B - (Math.max(0, Math.min(yTop, v)) / yTop) * (B - T); + + const atOf = new Map(schedule.claims.map((c) => [c.id, c.at])); + const steps = totals.steps.filter((s) => atOf.has(s.id)); + + // ---- the series --------------------------------------------------------- + const seriesSvg = cfg.series + .map((s) => { + const pts = totals.series[s.scope] ?? []; + if (!pts.length) return ""; + return ( + `<path d="${stepPath(pts, X, Y)}" fill="none" stroke="${s.color}" stroke-width="2.2" ` + + `stroke-linejoin="round"${s.dash ? ` stroke-dasharray="${s.dash}"` : ""}/>` + ); + }) + .join(""); + + const statedPts = totals.series.stated; + const impliedPts = totals.series.implied; + + // ---- the gap between the two totals ------------------------------------- + // Only where BOTH are defined: before his first total there is nothing to be + // a gap from, and a band anchored to zero would read as a claim of its own. + let gapSvg = ""; + const firstStated = statedPts[0] ? dayOf(statedPts[0].date) : Infinity; + const firstImplied = impliedPts[0] ? dayOf(impliedPts[0].date) : Infinity; + const gapFrom = Math.max(firstStated, firstImplied); + if (Number.isFinite(gapFrom)) { + const up = stepVerts(impliedPts, X, Y).filter((p) => p[0] >= X(cfg.from) && true); + const dn = stepVerts(statedPts, X, Y); + const clipX = L + ((gapFrom - d0) / (d1 - d0)) * (R - L); + const poly = + `M${up.map((p) => `${p[0].toFixed(1)},${p[1].toFixed(1)}`).join(" L")} ` + + `L${R},${up.length ? up[up.length - 1][1].toFixed(1) : Y(0)} ` + + `L${R},${dn.length ? dn[dn.length - 1][1].toFixed(1) : Y(0)} ` + + `L${[...dn].reverse().map((p) => `${p[0].toFixed(1)},${p[1].toFixed(1)}`).join(" L")} Z`; + gapSvg = + `<clipPath id="gapstart"><rect x="${clipX.toFixed(1)}" y="0" width="${(R - clipX + 2).toFixed(1)}" height="${H}"/></clipPath>` + + `<g clip-path="url(#gapstart)">` + + `<path id="gapband" d="${poly}" fill="${cfg.gapFill}" opacity="${cfg.gapOpacity}"/>`; + + // Where the total is BELOW one of its own parts the gap is not a gap, it is + // an impossibility — so it is hatched rather than tinted. Same geometry, a + // different claim about what it means. + const bad = steps.filter((s) => s.flags.some((f) => f.rule === "contradicts_component")); + const bands = bad.map((s) => { + const x = X(s.date); + const next = statedPts.find((p) => dayOf(p.date) > dayOf(s.date)); + const x2 = next ? X(next.date) : R; + return `<rect x="${x.toFixed(1)}" y="0" width="${Math.max(2, x2 - x).toFixed(1)}" height="${H}"/>`; + }); + if (bands.length) { + gapSvg += + `<clipPath id="impossible">${bands.join("")}</clipPath>` + + `<g clip-path="url(#impossible)"><g clip-path="url(#gapclip)">` + + `<path d="${poly}" fill="url(#hatch)"/></g></g>` + + `<clipPath id="gapclip"><path d="${poly}"/></clipPath>`; + } + gapSvg += `</g>`; + } + + // ---- marks -------------------------------------------------------------- + // Two orthogonal encodings, not four shapes. FILL is evidence: filled means a + // clip plays behind it, hollow means the claim is counted but not quoted. + // BADGE is coherence. A qualitative claim has no y position at all. + const colourOf = (s) => + s.scope === "all" ? cfg.statedColor : (cfg.series.find((x) => x.scope === s.scope)?.color ?? pal.muted); + + const marks = []; + const ticks = []; + const badges = []; + for (const s of steps) { + const x = X(s.date); + const c = colourOf(s); + const live = !!s.entryId; + if (s.qualitative || s.value == null) { + ticks.push( + `<line class="qt" id="qt-${s.id}" x1="${x.toFixed(1)}" y1="${AXIS + 3}" x2="${x.toFixed(1)}" y2="${AXIS + 12}" ` + + `stroke="${c}" stroke-width="2" opacity="0"/>`, + ); + continue; + } + const y = Y(s.value); + marks.push( + live + ? `<circle class="mk" id="mk-${s.id}" cx="${x.toFixed(1)}" cy="${y.toFixed(1)}" r="4" fill="${c}" opacity="0"/>` + : `<circle class="mk" id="mk-${s.id}" cx="${x.toFixed(1)}" cy="${y.toFixed(1)}" r="4" fill="none" ` + + `stroke="${c}" stroke-width="1.8" stroke-dasharray="2 2" opacity="0"/>`, + ); + // Not every predicate earns a badge. `population_mismatch` already rides + // under the stated number as its qualifier ("salaried"), and + // `adjudicator` notes are for the inbox, not the screen. Badging all five + // put a triangle on half the marks, which is the same as badging none. + if (s.flags.some((f) => BADGED.has(f.rule))) { + badges.push( + `<path class="fg" id="fg-${s.id}" d="M${(x + 5).toFixed(1)},${(y - 6).toFixed(1)} l0,-10 l8,3 l-8,3 z" ` + + `fill="${cfg.flagColor}" stroke="${cfg.flagColor}" stroke-width="1.2" opacity="0"/>`, + ); + } + } + + // ---- direct end labels -------------------------------------------------- + // Direct end labels, pushed apart. + // + // Four of the five series end within a couple of people of each other, so + // their labels land on top of one another and the band becomes unreadable + // exactly where it is making its point. Same fix the closing card already + // uses: spread them, then elbow a leader back to the value each belongs to. + // Direct labels are the secondary encoding that lets the palette be legible + // at all under CVD, so an unreadable stack defeats the point of having them. + const wanted = [ + { pts: impliedPts, colour: cfg.impliedColor, text: "implied", weight: true }, + { pts: statedPts, colour: cfg.statedColor, text: "stated", weight: true }, + ...cfg.series.map((sr) => ({ + pts: totals.series[sr.scope] ?? [], + colour: sr.color, + text: SCOPE_LABEL[sr.scope] ?? sr.scope, + weight: false, + })), + ].filter((l) => l.pts.length); + + const LBLH = 15; + const placed = wanted + .map((l) => ({ ...l, lineY: Y(l.pts[l.pts.length - 1].value) })) + .sort((a, b) => a.lineY - b.lineY) + .map((l) => ({ ...l, y: l.lineY })); + for (let i = 1; i < placed.length; i += 1) { + placed[i].y = Math.max(placed[i].y, placed[i - 1].y + LBLH); + } + const over = placed.length ? placed[placed.length - 1].y - (B + 4) : 0; + if (over > 0) for (const l of placed) l.y -= over; + + const endLabels = placed + .map((l) => { + const elbow = + Math.abs(l.y - l.lineY) > 1.5 + ? `<path d="M${R} ${l.lineY.toFixed(1)} L${R + 5} ${l.lineY.toFixed(1)} ` + + `L${R + 5} ${l.y.toFixed(1)} L${R + 9} ${l.y.toFixed(1)}" fill="none" ` + + `stroke="${l.colour}" stroke-width="1" opacity="0.75"/>` + : ""; + return ( + elbow + + `<text x="${R + 13}" y="${(l.y + 3.5).toFixed(1)}" font-size="11" fill="${l.colour}"` + + `${l.weight ? ' font-weight="700"' : ""}>${esc(l.text)}</text>` + ); + }) + .join(""); + + // ---- the year axis ------------------------------------------------------ + const y0 = Number(cfg.from.slice(0, 4)), y1 = Number(cfg.to.slice(0, 4)); + const years = []; + for (let y = y0; y <= y1; y += 1) { + const x = X(`${y}-01-01`); + if (x < L - 1 || x > R + 1) continue; + years.push( + `<line x1="${x.toFixed(1)}" y1="${AXIS}" x2="${x.toFixed(1)}" y2="${AXIS + 4}" stroke="${pal.muted}" stroke-width="1"/>` + + `<text x="${x.toFixed(1)}" y="${AXIS + 22}" font-size="12" fill="${pal.muted}" text-anchor="middle">${y}</text>`, + ); + } + + const gridStep = yTop > 34 ? 10 : yTop > 12 ? 10 : 5; + const gridVals = []; + for (let v = 0; v <= yTop; v += gridStep) gridVals.push(v); + const grid = gridVals + .filter((v) => v <= yTop) + .map( + (v) => + `<line x1="${L}" y1="${Y(v).toFixed(1)}" x2="${R}" y2="${Y(v).toFixed(1)}" stroke="${render.rail?.rule ?? "#2A322F"}" stroke-width="1"/>` + + `<text x="${L - 10}" y="${(Y(v) + 4).toFixed(1)}" font-size="12" fill="${pal.muted}" text-anchor="end">${v}</text>`, + ) + .join(""); + + // ---- the sweep ---------------------------------------------------------- + // A piecewise-linear map from finished-video seconds to chart x, through the + // schedule's own (claim time, claim date) pairs. That is what makes the + // playhead track the CURRENT MOMENT rather than crawling at a constant rate: + // where the cut lingers, the playhead lingers. + const keys = steps.map((s) => ({ t: atOf.get(s.id), x: X(s.date) })).sort((a, b) => a.t - b.t); + const sweep = [{ t: 0, x: L }, ...keys, { t: DUR, x: R }]; + + // ---- the readout -------------------------------------------------------- + const RX = 1215; + const rollTweens = []; + let prevStated = null, prevImplied = null; + for (const s of steps) { + const t = atOf.get(s.id); + if (s.implied !== prevImplied) { + rollTweens.push({ t, k: "imp", v: s.implied, d: s.impliedDelta }); + prevImplied = s.implied; + } + if (s.stated !== prevStated) { + rollTweens.push({ t, k: "sta", v: s.stated, d: s.statedDelta, pop: s.population }); + prevStated = s.stated; + } + } + const flagCues = steps + .map((s) => { + const f = s.flags.find((x) => BADGED.has(x.rule)); + return f ? { t: atOf.get(s.id), text: f.text } : null; + }) + .filter(Boolean); + + const data = { sweep, marks: steps.map((s) => ({ id: s.id, t: atOf.get(s.id) })), rollTweens, flagCues, dur: DUR }; + + return `<!doctype html> +<html lang="en"> + <head> + <meta charset="UTF-8" /> + <meta name="viewport" content="width=${W}, height=${H}" /> + <script src="https://cdn.jsdelivr.net/npm/gsap@3.14.2/dist/gsap.min.js"></script> + <style> + /* The REAL files, copied in beside the composition. A bare local() + resolves in a desktop browser and FAILS in the render browser, which + then silently falls back and shifts every metric in the band. */ + @font-face { font-family: 'Band'; font-weight: 400; font-style: normal; + src: url('${fonts?.regular ?? ""}') format('truetype'); } + @font-face { font-family: 'Band'; font-weight: 700; font-style: normal; + src: url('${fonts?.bold ?? ""}') format('truetype'); } + * { margin: 0; padding: 0; box-sizing: border-box; } + html, body { margin: 0; width: ${W}px; height: ${H}px; overflow: hidden; background: transparent; } + body { font-family: 'Band', sans-serif; } + #root { position: relative; width: ${W}px; height: ${H}px; overflow: hidden; } + text { font-family: 'Band', sans-serif; } + .cap { position: absolute; font-size: 11px; letter-spacing: .06em; color: ${pal.muted}; } + .roll { position: absolute; font-size: 30px; font-weight: 700; color: ${pal.fg}; + font-variant-numeric: tabular-nums; } + .chip { position: absolute; font-size: 12px; font-weight: 700; padding: 1px 6px; border-radius: 3px; + opacity: 0; } + .qual { position: absolute; font-size: 11px; color: ${pal.muted}; } + /* Width is DERIVED, not fixed. The band is as wide as the frame less the + rail, so widening the rail narrows it — and a fixed 275px box that fit + at 1500 runs off the frame at 1420, clipping the plain-words reason + mid-sentence. Which is the one thing a predicate exists to say. */ + .why { position: absolute; font-size: 11px; color: ${cfg.flagColor}; width: ${W - RX - 14}px; line-height: 1.35; opacity: 0; } + </style> + </head> + <body> + <div id="root" data-composition-id="band" data-start="0" data-duration="${DUR.toFixed(2)}" + data-width="${W}" data-height="${H}"> + <div id="band" class="clip" data-start="0" data-duration="${DUR.toFixed(2)}" data-track-index="1" + style="position:absolute; inset:0;"> + <svg width="${W}" height="${H}" viewBox="0 0 ${W} ${H}"> + <defs> + <pattern id="hatch" width="7" height="7" patternUnits="userSpaceOnUse" patternTransform="rotate(45)"> + <line x1="0" y1="0" x2="0" y2="7" stroke="${cfg.flagColor}" stroke-width="1.6" opacity="0.55"/> + </pattern> + <clipPath id="sweep"><rect id="sweeprect" x="0" y="0" width="${L}" height="${H}"/></clipPath> + </defs> + + ${grid} + ${years.join("")} + <line x1="${L}" y1="${AXIS}" x2="${R}" y2="${AXIS}" stroke="${pal.muted}" stroke-width="1.4"/> + + <!-- everything that is DRAWN is inside the sweep, so nothing can get + ahead of the playhead -- including the filled gap band, which has + no stroke to offset and would otherwise appear whole. --> + <g clip-path="url(#sweep)"> + ${gapSvg} + <path id="s-implied" d="${stepPath(impliedPts, X, Y)}" fill="none" stroke="${cfg.impliedColor}" + stroke-width="${cfg.impliedWidth}" stroke-linejoin="round"/> + <path id="s-stated" d="${stepPath(statedPts, X, Y)}" fill="none" stroke="${cfg.statedColor}" + stroke-width="2.6" stroke-linejoin="round"/> + ${seriesSvg} + </g> + + ${ticks.join("")} + ${marks.join("")} + ${badges.join("")} + ${endLabels} + + <!-- The playhead wears the palette amber, NOT the flag colour. They + are different jobs -- one is chrome that says "here", the other is + data that says "this figure cannot be true" -- and a viewer who + has learned that the yellow triangle means trouble must not read + the same yellow sweeping across the plot every frame. --> + <line id="head" x1="${L}" y1="14" x2="${L}" y2="${AXIS + 14}" stroke="${pal.amber}" stroke-width="1.6"/> + </svg> + + <div class="cap" style="left:${RX}px; top:12px;">IMPLIED · OUR SUM</div> + <div class="roll" id="r-imp" style="left:${RX}px; top:26px;">—</div> + <div class="chip" id="c-imp" style="left:${RX}px; top:64px; background:${pal.accent}; color:${pal.bg};">+0</div> + + <div class="cap" style="left:${RX + 118}px; top:12px;">STATED</div> + <div class="roll" id="r-sta" style="left:${RX + 118}px; top:26px; color:${cfg.statedColor};">—</div> + <div class="chip" id="c-sta" style="left:${RX + 118}px; top:64px; background:${cfg.statedColor}; color:${pal.bg};">+0</div> + <div class="qual" id="q-sta" style="left:${RX + 118}px; top:88px;"></div> + + <div class="cap" style="left:${RX + 218}px; top:12px;">GAP</div> + <div class="roll" id="r-gap" style="left:${RX + 218}px; top:26px; color:${pal.muted};">—</div> + + <div class="why" id="why" style="left:${RX}px; top:112px;"></div> + <div class="cap" style="left:${RX}px; top:${H - 22}px; width:280px;">our sum of his per-company claims</div> + </div> + </div> + + <script> + const DATA = ${JSON.stringify(data)}; + window.__timelines = window.__timelines || {}; + const tl = gsap.timeline({ paused: true }); + + // The sweep. One tween per schedule leg, so the playhead moves at the rate + // the CUT moves rather than at a constant rate across the axis. + const rect = document.getElementById("sweeprect"); + const head = document.getElementById("head"); + for (let i = 1; i < DATA.sweep.length; i += 1) { + const a = DATA.sweep[i - 1], b = DATA.sweep[i]; + const d = Math.max(1 / 60, b.t - a.t); + tl.fromTo(rect, { attr: { width: a.x } }, { attr: { width: b.x }, duration: d, ease: "none" }, a.t); + tl.fromTo(head, { attr: { x1: a.x, x2: a.x } }, + { attr: { x1: b.x, x2: b.x }, duration: d, ease: "none" }, a.t); + } + + // Marks land as the playhead reaches them. + for (const m of DATA.marks) { + const dot = document.getElementById("mk-" + m.id); + if (dot) tl.fromTo(dot, { opacity: 0, scale: 0, transformOrigin: "center" }, + { opacity: 1, scale: 1, duration: 0.3, ease: "back.out(2)" }, m.t); + const tick = document.getElementById("qt-" + m.id); + if (tick) tl.fromTo(tick, { opacity: 0 }, { opacity: 0.9, duration: 0.25 }, m.t); + const flag = document.getElementById("fg-" + m.id); + if (flag) tl.fromTo(flag, { opacity: 0, scale: 0.4, transformOrigin: "center" }, + { opacity: 1, scale: 1, duration: 0.3, ease: "back.out(2)" }, m.t + 0.12); + } + + // The readout. Only the number that CHANGED rolls; the other holds, and a + // value:null claim moves neither. + const st = { imp: null, sta: null }; + const el = (id) => document.getElementById(id); + const paint = () => { + el("r-imp").textContent = st.imp == null ? "—" : String(Math.round(st.imp)); + el("r-sta").textContent = st.sta == null ? "—" : String(Math.round(st.sta)); + el("r-gap").textContent = + st.imp == null || st.sta == null ? "—" : String(Math.round(st.imp) - Math.round(st.sta)); + }; + paint(); + // seen is BUILD-time bookkeeping and st is RUNTIME state, and they must + // not be the same object. + // + // They were, and the readout showed its own conclusion before drawing a + // term of it: the loop below walked st to the FINAL value while wiring + // the tweens, so the first paint() any tween triggered rendered the other + // series' closing number. Measured: the sourced cut showed "STATED 5" for + // its first 53 seconds, and the full cut showed "IMPLIED 23" five seconds + // in, beside an empty plot. That is precisely the "showing an answer the + // playhead has not reached" this band's sweep exists to prevent. + const seen = { imp: null, sta: null }; + for (const r of DATA.rollTweens) { + const key = r.k; + if (seen[key] === null && r.v !== null) { + // First appearance: set rather than roll, because rolling up from a + // number that was never on screen invents a history. + tl.call(() => { st[key] = r.v; paint(); }, [], r.t); + } else { + const box = { v: seen[key] ?? 0 }; + tl.to(box, { + v: r.v ?? 0, duration: 0.45, ease: "power2.out", + onUpdate: () => { st[key] = box.v; paint(); }, + }, r.t); + } + seen[key] = r.v; + if (r.d != null) { + const chip = el(key === "imp" ? "c-imp" : "c-sta"); + const sign = r.d > 0 ? "▲ +" : "▼ "; + tl.call((c, s) => { c.textContent = s; }, [chip, sign + r.d], r.t); + tl.fromTo(chip, { opacity: 0, y: 5 }, { opacity: 1, y: 0, duration: 0.22 }, r.t); + tl.to(chip, { opacity: 0, duration: 0.3 }, r.t + 1.6); + } + if (key === "sta" && r.pop) { + tl.call((e, s) => { e.textContent = s; }, [el("q-sta"), r.pop], r.t); + } + } + + // The plain-words reason, on screen only while its claim is current. + for (const f of DATA.flagCues) { + tl.call((e, s) => { e.textContent = s; }, [el("why"), "⚑ " + f.text], f.t); + tl.fromTo(el("why"), { opacity: 0 }, { opacity: 1, duration: 0.3 }, f.t); + tl.to(el("why"), { opacity: 0, duration: 0.35 }, f.t + 3.2); + } + + window.__timelines["band"] = tl; + </script> + </body> +</html> +`; +} + +// --------------------------------------------------------------------------- +// CLI +// --------------------------------------------------------------------------- + +const HF_JSON = JSON.stringify( + { $schema: "https://hyperframes.heygen.com/schema/hyperframes.json", paths: { blocks: "compositions", assets: "assets" } }, + null, + 2, +); + +export async function composeChrome({ manifestPath, outDir, region = "chart", duration = null, doRender = false, fps = null, variant = "sourced" }) { + // The variant's view, and its own out directory. The band plots the ledger + // the cut carries; handed the whole manifest it would draw marks for claims + // this cut never makes and put the playhead schedule out by that many. + const manifest = selectVariant(JSON.parse(await readFile(manifestPath, "utf8")), variant); + const base = outDir ?? path.join(path.dirname(path.resolve(manifestPath)), "out", variant); + const schedule = JSON.parse(await readFile(path.join(base, "schedule.json"), "utf8")); + const totals = ledgerTotals(manifest.ledger); + + const projDir = path.join(base, "chrome", region); + await mkdir(path.join(projDir, "assets"), { recursive: true }); + if (region !== "chart") throw new Error(`unknown chrome region: ${region}`); + + // The SAME faces the ffmpeg cards use, copied in beside the composition. + // Chrome will not resolve a bare local() in the render browser, and the + // failure is silent: it falls back and every metric in the band shifts. + const fonts = {}; + for (const [slot, src] of [["regular", manifest.render.fontRegular], ["bold", manifest.render.fontBold]]) { + if (!src) continue; + const name = `${slot}${path.extname(src) || ".ttf"}`; + await copyFile(src, path.join(projDir, "assets", name)).catch(() => {}); + fonts[slot] = `assets/${name}`; + } + + const html = chartBandHtml(manifest, totals, schedule, { + fonts, + ...(duration ? { duration } : {}), + }); + await writeFile(path.join(projDir, "index.html"), html, "utf8"); + await writeFile(path.join(projDir, "hyperframes.json"), HF_JSON + "\n", "utf8"); + + if (!doRender) return { projDir, frames: null }; + + const frames = path.join(base, "chrome", `${region}-frames`); + await run("npx", [ + "--yes", "hyperframes@latest", "render", + "--format", "png-sequence", "--quality", "high", + "--fps", String(fps ?? manifest.render.fps), + "--output", frames, projDir, + ], { maxBuffer: 1 << 26 }); + return { projDir, frames }; +} + +if (import.meta.url === `file://${process.argv[1]}`) { + const argv = process.argv.slice(2); + const flag = (n) => { const i = argv.indexOf(n); return i < 0 ? null : argv[i + 1]; }; + const VALUED = new Set(["--out", "--region", "--duration", "--variant"]); + const manifestPath = argv.find((a, i) => !a.startsWith("--") && !VALUED.has(argv[i - 1])); + if (!manifestPath) { + console.error( + "usage: compose-chrome.mjs <manifest.json> [--region chart] [--variant sourced|full]\n" + + " [--duration <s>] [--out <dir>] [--render]", + ); + process.exit(2); + } + const r = await composeChrome({ + manifestPath, + outDir: flag("--out"), + region: flag("--region") ?? "chart", + variant: flag("--variant") ?? "sourced", + duration: flag("--duration") ? Number(flag("--duration")) : null, + doRender: argv.includes("--render"), + }); + console.log(r.frames ? `frames -> ${r.frames}` : `project -> ${r.projDir}`); +} diff --git a/scripts/report-to-video/cues.mjs b/scripts/report-to-video/cues.mjs @@ -0,0 +1,270 @@ +// Where a clip's caption cues come from. +// +// A clip window is widened from a cue span to a whole sentence, which needs cue +// END times. Nothing else in the pipeline carries them: a report citation is a +// single start second, and the MCP `Snippet` type has no `end` field. So this is +// the one place that answers "what are the real cue boundaries for this video". +// +// TWO SOURCES, SAME SHAPE. A local corpus stores each video as +// `<CHANNELS_DIR>/<slug>/data/<id>/transcript.cues.json`, and a *published* +// archive serves the same record inside a paginated shard. The two carry the +// same fields — `{ slug, id, channelSlug, title, uploadDate, duration, channel, +// description, platform, webpageUrl, cues: [{start, end, text}] }` — so one +// resolver serves both `loadCues` and `videoMeta`, and a caller cannot tell +// which it got beyond the `from` marker. +// +// That parity is what makes a corpus optional. Clone the repo, point a manifest +// at a public instance, and the video pipeline can cut clips without mirroring a +// single channel: the cue windows come over HTTP, and the media itself was +// always a network fetch (`yt-dlp --download-sections`). +// +// The shard walk is the contract published at `/corpus.json` under `shardScheme`: +// 1. GET <origin>/corpus.json -> channels[].manifests.transcripts +// 2. GET that manifest -> { pageCount, slugToPage: { <id>: N } } +// 3. GET page-<NNNN>.json (N zero-padded to 4) -> array of records +// 4. take the record whose `id` matches +// +// Local wins when present: it is faster, works offline, and is the operator's own +// data. HTTP is the fallback, not a preference. +// +// THE TWO SOURCES CAN DISAGREE, AND IT IS NOT ROUNDING. A published archive is a +// snapshot; a live corpus keeps moving. Re-synced platform captions, an +// auto-caption replacement or a re-transcription all rewrite a video's cues in +// place, and the archive keeps the text it was built from until it is rebuilt. +// Measured on this corpus (local 2026-08-13 against a 2026-08-07 publish): of +// four videos checked, three were byte-identical and one had 65 of its 84 cue +// texts changed with timings shifted by up to **2.24 s** — enough to cut a clip +// in the wrong place. +// +// So `prefer` is a real decision, not a micro-optimisation: +// "auto" (default) local when present, else HTTP. Right for an operator. +// "local" never fall back. Fail loudly instead of silently cutting from +// different cues than the ones a window was authored against. +// "http" always the archive. Right when you want the windows to match what a +// reader following the citation will actually see, and the only +// option that is reproducible on a machine with no corpus. +// Whatever answers, the returned record carries `from` so a caller can record it. + +import { readFile, writeFile, mkdir } from "node:fs/promises"; +import path from "node:path"; +import os from "node:os"; +import { createHash } from "node:crypto"; +import { fileURLToPath } from "node:url"; + +// This file lives at <repo>/scripts/report-to-video/, so the corpus a plain +// checkout would have is two levels up. Previously this defaulted to an absolute +// path inside the original author's home directory, which meant every other +// clone silently looked in a directory that does not exist. +const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", ".."); + +export const DEFAULT_CHANNELS_DIR = + process.env.CHANNELS_DIR ?? path.join(REPO_ROOT, "transcripts", "channels"); + +const DEFAULT_CACHE_DIR = + process.env.REPORT_CACHE_DIR ?? + path.join(os.homedir(), ".cache", "archilyzer-report-to-video"); + +// A shard page is capped at 8 MB and holds ~100 videos, so refetching one per +// clip — across two separate processes, resolve-windows then build-video — is +// the difference between usable and painful. Cached by URL on disk; archives are +// rebuilt rarely and a stale page only matters if the cues themselves changed. +function cacheKey(url) { + return createHash("sha1").update(url).digest("hex") + ".json"; +} + +export function pageFileName(pageNumber) { + return `page-${String(pageNumber).padStart(4, "0")}.json`; +} + +export function pageUrlFrom(manifestUrl, pageNumber) { + const u = new URL(manifestUrl); + u.pathname = u.pathname.replace(/[^/]+$/, pageFileName(pageNumber)); + return u.toString(); +} + +// The origin of the archive a manifest was built against. Every manifest already +// records this — `siteOrigin` explicitly, and `corpus` as `remote:<url>` or +// `local:<path>` — so the common case needs no configuration at all. +export function siteOriginFromManifest(manifest) { + const p = manifest?.provenance ?? {}; + if (typeof p.siteOrigin === "string" && p.siteOrigin.trim()) { + return p.siteOrigin.replace(/\/+$/, ""); + } + if (typeof p.corpus === "string" && p.corpus.startsWith("remote:")) { + return p.corpus.slice("remote:".length).replace(/\/+$/, ""); + } + if (typeof p.shareLink === "string" && /^https?:/.test(p.shareLink)) { + try { + return new URL(p.shareLink).origin; + } catch { + /* fall through */ + } + } + return null; +} + +export class CueLookupError extends Error { + constructor(message, { channelSlug, videoId, tried }) { + super(message); + this.name = "CueLookupError"; + this.channelSlug = channelSlug; + this.videoId = videoId; + this.tried = tried; + } +} + +export function createCueSource({ + channelsDir = DEFAULT_CHANNELS_DIR, + siteOrigin = null, + cacheDir = DEFAULT_CACHE_DIR, + fetchImpl = globalThis.fetch, + log = () => {}, + // "auto" | "local" | "http" — see the note on divergence above. + prefer = "auto", + // Opt-in, because it is expensive: see resolveSiteId below. + resolveSiteIds = false, +} = {}) { + const mem = new Map(); + + async function getJson(url) { + if (mem.has(url)) return mem.get(url); + const disk = cacheDir ? path.join(cacheDir, cacheKey(url)) : null; + if (disk) { + try { + const cached = JSON.parse(await readFile(disk, "utf8")); + mem.set(url, cached); + return cached; + } catch { + /* cold cache */ + } + } + log(`fetch ${url}`); + const res = await fetchImpl(url); + if (!res.ok) throw new Error(`GET ${url} -> ${res.status}`); + const json = await res.json(); + mem.set(url, json); + if (disk) { + try { + await mkdir(path.dirname(disk), { recursive: true }); + await writeFile(disk, JSON.stringify(json)); + } catch { + // A cache we cannot write is a slow run, not a failed one. + } + } + return json; + } + + async function channelEntry(origin, channelSlug) { + const corpus = await getJson(`${origin}/corpus.json`); + const found = (corpus.channels ?? []).find((c) => c.slug === channelSlug); + if (!found) { + throw new CueLookupError( + `channel "${channelSlug}" is not in ${origin}/corpus.json`, + { channelSlug, videoId: null, tried: [`${origin}/corpus.json`] }, + ); + } + return found; + } + + // Map a LOCAL video id onto the id the site serves, by scanning the channel's + // pages for a record whose `webpageUrl` contains it. + // + // WHY THIS EXISTS: a Rumble video has two ids. The site (and the MCP) key it by + // the EMBED id; the local cue directory is named for the URL SLUG. A manifest + // hand-authored against local cue dirs therefore carries slugs that are absent + // from the published `slugToPage` — every Rumble clip misses. + // + // WHY IT IS OPT-IN: it downloads a channel's shards until it hits a match, and + // a shard is up to 8 MB. That is a reasonable price to pay knowingly and a + // terrible one to pay silently, so the direct lookup fails with instructions + // instead and this runs only when asked. + async function resolveSiteId(origin, entry, wanted) { + const manifest = await getJson(entry.manifests.transcripts); + log(`resolving "${wanted}" by scanning ${manifest.pageCount} shard(s) of ${entry.slug}`); + for (let n = 0; n < manifest.pageCount; n += 1) { + const page = await getJson(pageUrlFrom(entry.manifests.transcripts, n)); + const hit = page.find( + (r) => r.id === wanted || r.slug === wanted || String(r.webpageUrl ?? "").includes(wanted), + ); + if (hit) return hit; + } + return null; + } + + async function fromHttp(channelSlug, videoId, hints) { + const origin = hints.siteOrigin ?? siteOrigin; + if (!origin) { + throw new CueLookupError( + `no local cues for ${channelSlug}/${videoId} and no archive origin to fetch them from ` + + `(set provenance.siteOrigin in the manifest, or pass --site-origin / SITE_ORIGIN)`, + { channelSlug, videoId, tried: ["local"] }, + ); + } + const siteChannel = hints.siteChannel ?? channelSlug; + const siteVideo = hints.siteVideo ?? videoId; + const entry = await channelEntry(origin, siteChannel); + const manifest = await getJson(entry.manifests.transcripts); + const pageNumber = manifest.slugToPage?.[siteVideo]; + + if (pageNumber === undefined) { + if (resolveSiteIds) { + const hit = await resolveSiteId(origin, entry, siteVideo); + if (hit) return { ...hit, from: "http" }; + } + throw new CueLookupError( + `"${siteVideo}" is not in ${siteChannel}'s published slugToPage on ${origin}.\n` + + ` If this is a Rumble clip, the archive is keyed by the EMBED id while a local cue\n` + + ` directory is named for the URL SLUG — they differ. Either add "siteVideo" (and\n` + + ` "siteChannel" if it also differs) to this clip in the manifest, or re-run with\n` + + ` --resolve-site-ids to find it by scanning the channel's shards (slow: downloads\n` + + ` up to 8 MB per shard until it matches).\n` + + ` Note that a clip's citeUrl is NOT usable here — it may deliberately cite a\n` + + ` different recording (a mirror that reads better), whose clock is not the same.`, + { channelSlug, videoId, tried: [entry.manifests.transcripts] }, + ); + } + + const page = await getJson(pageUrlFrom(entry.manifests.transcripts, pageNumber)); + const record = page.find((r) => r.id === siteVideo || r.slug === siteVideo); + if (!record) { + throw new CueLookupError( + `${siteChannel}/${siteVideo} is on shard ${pageNumber} per the manifest, but no record ` + + `there has that id — the published archive is inconsistent`, + { channelSlug, videoId, tried: [pageUrlFrom(entry.manifests.transcripts, pageNumber)] }, + ); + } + return { ...record, from: "http" }; + } + + async function fromLocal(channelSlug, videoId) { + const p = path.join(channelsDir, channelSlug, "data", videoId, "transcript.cues.json"); + const parsed = JSON.parse(await readFile(p, "utf8")); + return { ...parsed, from: "local" }; + } + + return { + channelsDir, + /** + * The full record for one video: cues plus the metadata build-video needs. + * `hints` may carry `siteChannel` / `siteVideo` (when the published archive + * keys this recording differently) and `siteOrigin` (per-manifest override). + */ + prefer, + async load(channelSlug, videoId, hints = {}) { + if (prefer === "http") return await fromHttp(channelSlug, videoId, hints); + try { + return await fromLocal(channelSlug, videoId); + } catch (err) { + if (err?.code !== "ENOENT" && err?.code !== "ENOTDIR") throw err; + if (prefer === "local") { + throw new CueLookupError( + `no local cues for ${channelSlug}/${videoId} under ${channelsDir}, and ` + + `--cue-source local forbids falling back to the archive`, + { channelSlug, videoId, tried: [channelsDir] }, + ); + } + return await fromHttp(channelSlug, videoId, hints); + } + }, + }; +} diff --git a/scripts/report-to-video/cues.test.mjs b/scripts/report-to-video/cues.test.mjs @@ -0,0 +1,297 @@ +// Tests for cues.mjs — the local-or-published cue resolver. +// +// The fetches are stubbed against a miniature of the real published shape, so +// these run offline and in CI. One separate, opt-in test hits a live archive to +// prove the miniature has not drifted from reality; see LIVE below. +// +// Run with: pnpm test:scripts +import assert from "node:assert/strict"; +import test from "node:test"; +import { mkdtemp, mkdir, writeFile, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +import { + createCueSource, + pageFileName, + pageUrlFrom, + siteOriginFromManifest, +} from "./cues.mjs"; + +const ORIGIN = "https://example.pages.dev"; + +// A published record. Deliberately the SAME field set a local transcript.cues.json +// carries — that parity is the whole reason one resolver can serve both. +const RECORD = { + slug: "chan/vid1", + id: "vid1", + channelSlug: "chan", + title: "A Title", + uploadDate: "20260101", + duration: 1200, + webpageUrl: "https://rumble.com/localslug-a-title.html", + cues: [ + { start: 0, end: 2.5, text: "first line" }, + { start: 2.5, end: 5, text: "second line." }, + ], +}; + +function stubFetch(routes, seen = []) { + return async (url) => { + seen.push(url); + if (!(url in routes)) return { ok: false, status: 404 }; + return { ok: true, status: 200, json: async () => routes[url] }; + }; +} + +const ROUTES = { + [`${ORIGIN}/corpus.json`]: { + channels: [ + { slug: "chan", manifests: { transcripts: `${ORIGIN}/transcripts/chan/manifest.json` } }, + ], + }, + [`${ORIGIN}/transcripts/chan/manifest.json`]: { + pageCount: 2, + slugToPage: { vid1: 0, other: 1 }, + }, + [`${ORIGIN}/transcripts/chan/page-0000.json`]: [RECORD], + [`${ORIGIN}/transcripts/chan/page-0001.json`]: [ + { ...RECORD, id: "other", slug: "chan/other", webpageUrl: "https://rumble.com/v94hyv-x.html" }, + ], +}; + +// cacheDir:null keeps every test off the real disk cache, so one test cannot +// poison another (or a developer's home directory). +function source(extra = {}) { + return createCueSource({ + channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"), + siteOrigin: ORIGIN, + cacheDir: null, + fetchImpl: stubFetch(ROUTES), + ...extra, + }); +} + +// --- the shard walk --------------------------------------------------------- + +test("page numbers are zero-padded to four digits", () => { + assert.equal(pageFileName(0), "page-0000.json"); + assert.equal(pageFileName(7), "page-0007.json"); + assert.equal(pageFileName(1234), "page-1234.json"); + // Not a cosmetic detail: page-0.json is a 404 on a real archive. + assert.notEqual(pageFileName(0), "page-0.json"); +}); + +test("a page URL replaces only the manifest's last segment", () => { + assert.equal( + pageUrlFrom("https://x.dev/transcripts/some-chan/manifest.json", 3), + "https://x.dev/transcripts/some-chan/page-0003.json", + ); +}); + +test("resolves a video over HTTP with cues intact", async () => { + const got = await source().load("chan", "vid1"); + assert.equal(got.from, "http"); + assert.equal(got.title, "A Title"); + assert.equal(got.duration, 1200); + assert.equal(got.cues.length, 2); + assert.deepEqual(got.cues[0], { start: 0, end: 2.5, text: "first line" }); + // The end times are the entire point: they are what widens a clip to a + // whole sentence, and nothing else in the pipeline carries them. + assert.ok(got.cues.every((c) => typeof c.end === "number")); +}); + +test("fetches each URL once, however many videos are read", async () => { + const seen = []; + const src = createCueSource({ + channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"), + siteOrigin: ORIGIN, + cacheDir: null, + fetchImpl: stubFetch(ROUTES, seen), + }); + await src.load("chan", "vid1"); + await src.load("chan", "vid1"); + await src.load("chan", "other"); + // corpus + manifest once each, then one shard per distinct page. + assert.deepEqual(seen, [ + `${ORIGIN}/corpus.json`, + `${ORIGIN}/transcripts/chan/manifest.json`, + `${ORIGIN}/transcripts/chan/page-0000.json`, + `${ORIGIN}/transcripts/chan/page-0001.json`, + ]); +}); + +// --- local wins ------------------------------------------------------------- + +test("a local corpus is preferred over the network", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "cues-local-")); + try { + const vdir = path.join(dir, "chan", "data", "vid1"); + await mkdir(vdir, { recursive: true }); + await writeFile( + path.join(vdir, "transcript.cues.json"), + JSON.stringify({ ...RECORD, title: "LOCAL COPY" }), + ); + const seen = []; + const src = createCueSource({ + channelsDir: dir, + siteOrigin: ORIGIN, + cacheDir: null, + fetchImpl: stubFetch(ROUTES, seen), + }); + const got = await src.load("chan", "vid1"); + assert.equal(got.from, "local"); + assert.equal(got.title, "LOCAL COPY"); + assert.deepEqual(seen, [], "must not touch the network when local data exists"); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +// --- the Rumble two-id trap ------------------------------------------------- + +test("a published-id miss fails loudly, naming the two-id trap", async () => { + // A Rumble video has two ids: the archive keys it by the EMBED id, while a + // local cue directory is named for the URL SLUG. A manifest authored against + // local dirs therefore carries an id the archive has never heard of. + await assert.rejects( + () => source().load("chan", "localslug"), + (err) => { + assert.equal(err.name, "CueLookupError"); + assert.match(err.message, /EMBED id/); + assert.match(err.message, /siteVideo/); + assert.match(err.message, /--resolve-site-ids/); + // It must also warn off the tempting wrong fix. + assert.match(err.message, /citeUrl is NOT usable/); + return true; + }, + ); +}); + +test("an explicit siteVideo hint resolves the mismatch", async () => { + const got = await source().load("chan", "localslug", { siteVideo: "vid1" }); + assert.equal(got.id, "vid1"); + assert.equal(got.cues.length, 2); +}); + +test("--resolve-site-ids finds the record by scanning shards", async () => { + // `other`'s webpageUrl embeds v94hyv, mirroring how a Rumble URL carries the + // slug while the archive is keyed by the embed id. + const got = await source({ resolveSiteIds: true }).load("chan", "v94hyv"); + assert.equal(got.id, "other"); + assert.equal(got.from, "http"); +}); + +test("scanning is off by default, because a shard is up to 8 MB", async () => { + await assert.rejects(() => source().load("chan", "v94hyv"), { name: "CueLookupError" }); +}); + +// --- prefer: the two sources can genuinely disagree -------------------------- + +test("prefer:http ignores a local copy entirely", async () => { + // Not a micro-optimisation. A published archive is a snapshot and a corpus + // keeps moving: measured on the real corpus, one video of four had 65 of its + // 84 cue texts rewritten and timings shifted by up to 2.24s between a + // 2026-08-07 publish and the local copy six days later. Which source answered + // decides where a clip gets cut. + const dir = await mkdtemp(path.join(tmpdir(), "cues-prefer-")); + try { + const vdir = path.join(dir, "chan", "data", "vid1"); + await mkdir(vdir, { recursive: true }); + await writeFile( + path.join(vdir, "transcript.cues.json"), + JSON.stringify({ ...RECORD, title: "LOCAL COPY" }), + ); + const src = createCueSource({ + channelsDir: dir, + siteOrigin: ORIGIN, + cacheDir: null, + fetchImpl: stubFetch(ROUTES), + prefer: "http", + }); + const got = await src.load("chan", "vid1"); + assert.equal(got.from, "http"); + assert.equal(got.title, "A Title", "must be the archive's copy, not the local one"); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("prefer:local refuses to fall back rather than cut from other cues", async () => { + const src = createCueSource({ + channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"), + siteOrigin: ORIGIN, + cacheDir: null, + fetchImpl: stubFetch(ROUTES), + prefer: "local", + }); + await assert.rejects(() => src.load("chan", "vid1"), (err) => { + assert.equal(err.name, "CueLookupError"); + assert.match(err.message, /forbids falling back/); + return true; + }); +}); + +// --- origin discovery ------------------------------------------------------- + +test("the archive origin comes from the manifest, in priority order", () => { + assert.equal( + siteOriginFromManifest({ provenance: { siteOrigin: "https://a.dev/" } }), + "https://a.dev", + "trailing slash trimmed", + ); + assert.equal( + siteOriginFromManifest({ provenance: { corpus: "remote:https://b.dev" } }), + "https://b.dev", + ); + assert.equal( + siteOriginFromManifest({ provenance: { shareLink: "https://c.dev/?v=x&t=1" } }), + "https://c.dev", + ); + // A local-corpus manifest names no remote origin, and must not invent one. + assert.equal(siteOriginFromManifest({ provenance: { corpus: "local:/srv/x" } }), null); + assert.equal(siteOriginFromManifest({}), null); +}); + +test("no origin and no local copy is a clear error, not a crash", async () => { + const src = createCueSource({ + channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"), + siteOrigin: null, + cacheDir: null, + fetchImpl: stubFetch({}), + }); + await assert.rejects(() => src.load("chan", "vid1"), (err) => { + assert.equal(err.name, "CueLookupError"); + assert.match(err.message, /no archive origin/); + return true; + }); +}); + +test("a channel absent from corpus.json is reported as such", async () => { + await assert.rejects(() => source().load("nosuch", "vid1"), (err) => { + assert.match(err.message, /not in .*corpus\.json/); + return true; + }); +}); + +// --- reality check ---------------------------------------------------------- + +// Opt-in: `LIVE=1 pnpm test:scripts`. The stubs above encode assumptions about a +// published archive's shape; this is the only thing that can catch them going +// stale. Skipped by default so the suite stays offline and deterministic. +test("LIVE: a real archive still matches the shape these stubs assume", { + skip: process.env.LIVE === "1" ? false : "set LIVE=1 to hit the network", +}, async () => { + const src = createCueSource({ + channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"), + siteOrigin: "https://jeralyzer.pages.dev", + cacheDir: null, + }); + const got = await src.load("chrissie-mayr", "2Pn_rMrHmEs"); + assert.equal(got.from, "http"); + assert.ok(got.cues.length > 0); + assert.ok(got.cues.every((c) => typeof c.start === "number" && typeof c.end === "number")); + for (const field of ["title", "uploadDate", "duration", "webpageUrl"]) { + assert.ok(got[field] !== undefined, `published record should carry ${field}`); + } +}); diff --git a/scripts/report-to-video/ledger-totals.mjs b/scripts/report-to-video/ledger-totals.mjs @@ -0,0 +1,540 @@ +// The ledger, adjudicated, turned into two running totals and five coherence +// flags. +// +// Plain ESM with no I/O and no rendering: `umtool`'s decisions reducer imports +// it from a Next server component, `compose-chrome.mjs` imports it from a CLI, +// and `node --test` runs it with a literal array. Every consumer gets the same +// arithmetic, which is the only way the chart band and the inbox can agree. +// +// --------------------------------------------------------------------------- +// Why an adjudication is required before any of this runs +// --------------------------------------------------------------------------- +// Both totals used to rest on `ledger[].company`, an undocumented interpretation +// that quietly mixed four different things: +// +// * SCOPE AMBIGUITY -- "I have 10 employees, my coffee company employees, my +// editors" is all-companies or coffee-only depending on where the comma +// falls, and nothing recorded which reading was taken. +// * DERIVED VALUES -- 18 and 20 are OUR sums of his per-company claims. He +// never utters either. The old manifest called 18 "the only explicit sum in +// the corpus", which is exactly backwards. +// * POPULATION DRIFT -- "full-time salaried", "employees", "all basically +// contractors" and "all 1099 and not full-time" were compared as one series. +// * SYNTHETIC VALUES -- 10.5 for "about 10 people, 11 people" and 5.5 for +// "five or six" are midpoints we invented and then attributed to him. +// +// So the STATED series may contain only a figure he utters as a single number +// for a named scope. Our arithmetic lives in the IMPLIED series, which says on +// screen that it is ours. That weakens "his stated total swings wildly" a little +// and makes it survive scrutiny. + +export const SCOPES = ["media", "coffee", "publica", "all"]; +/** The three that sum. `all` is a claim ABOUT the sum, never a term in it. */ +export const COMPANY_SCOPES = ["media", "coffee", "publica"]; +export const POPULATIONS = ["employees", "full-time", "salaried", "contractor", "1099", "people"]; +export const VALUE_KINDS = ["uttered", "derived", "synthetic"]; +export const SCOPE_CONFIDENCE = ["clear", "read", "unresolved"]; + +/** The six fields an entry must carry before it may feed a total. */ +export const ADJUDICATION_FIELDS = [ + "scope", + "scopeBasis", + "scopeConfidence", + "population", + "valueKind", + "flags", +]; + +const isStr = (v) => typeof v === "string" && v.trim().length > 0; + +/** + * Which of the six an entry is missing or has wrong. Empty array == adjudicated. + * + * Returned rather than thrown because this is what the inbox renders: one + * `claim-unadjudicated` row per entry, naming the fields still outstanding. + */ +export function adjudicationGaps(entry) { + const gaps = []; + if (!SCOPES.includes(entry?.scope)) gaps.push("scope"); + // The phrase that settles it. An adjudication with no basis is an opinion, + // and the whole point of the exercise was to stop shipping those. + if (!isStr(entry?.scopeBasis)) gaps.push("scopeBasis"); + if (!SCOPE_CONFIDENCE.includes(entry?.scopeConfidence)) gaps.push("scopeConfidence"); + if (!POPULATIONS.includes(entry?.population)) gaps.push("population"); + if (!VALUE_KINDS.includes(entry?.valueKind)) gaps.push("valueKind"); + if (!Array.isArray(entry?.flags)) gaps.push("flags"); + return gaps; +} + +export const isAdjudicated = (entry) => adjudicationGaps(entry).length === 0; + +/** Entries that carry a number nobody has ruled on yet. */ +export const unadjudicatedOf = (ledger) => + (ledger ?? []).filter((e) => !isAdjudicated(e)); + +export class UnadjudicatedLedger extends Error { + constructor(entries) { + const ids = entries.map((e) => e?.id ?? "?"); + super( + `${ids.length} ledger entr${ids.length === 1 ? "y is" : "ies are"} unadjudicated ` + + `(${ids.slice(0, 6).join(", ")}${ids.length > 6 ? ", …" : ""}). ` + + "Both totals lie if you act on an unadjudicated ledger — work the inbox first.", + ); + this.name = "UnadjudicatedLedger"; + this.entries = entries; + this.ids = ids; + } +} + +// --------------------------------------------------------------------------- +// Dates. Two entries are month-only ("2024-02", "2025-06") because that is all +// the source supports; both are qualitative, so they order the rail without +// touching a total. Sorting pads them to the first of the month, which puts them +// before any dated claim in the same month rather than guessing a day. +// --------------------------------------------------------------------------- +export function dateKey(d) { + const s = String(d ?? ""); + const m = /^(\d{4})(?:-(\d{2}))?(?:-(\d{2}))?$/.exec(s); + if (!m) return "9999-99-99"; + return `${m[1]}-${m[2] ?? "01"}-${m[3] ?? "01"}`; +} + +/** Chronological, ties broken by ledger id so the order is total and stable. */ +export const chronological = (ledger) => + [...(ledger ?? [])].sort((a, b) => { + const d = dateKey(a.date).localeCompare(dateKey(b.date)); + return d !== 0 ? d : String(a.id).localeCompare(String(b.id)); + }); + +// --------------------------------------------------------------------------- +// The five predicates. +// +// "Doesn't make sense" is COMPUTED, never asserted. Each one is named, each one +// renders as plain words on screen, and each one is a decision the human takes +// rather than a fact the video states. +// +// Deliberately NOT a rule: a large rise or fall between claims. Fluctuation is +// the SUBJECT of the video, not a defect, and flagging it would be putting a +// thumb on the scale. +// --------------------------------------------------------------------------- + +const SCOPE_LABEL = { + media: "The Quartering", + coffee: "Coffee Brand Coffee", + publica: "The Publica", + all: "all companies", +}; + +/** Populations that describe someone who is not, on his own account, staff. */ +const UNDERCUTTING = new Set(["contractor", "1099"]); + +export const PREDICATES = [ + "contradicts_component", + "same_day_conflict", + "self_negating", + "population_mismatch", + "not_his_number", + "status_flip", +]; + +// --------------------------------------------------------------------------- +// Employment status, as two camps. +// +// `employees` and `people` are in NEITHER. They are what he says when he is not +// making a claim about status at all, and reading them as one side or the other +// would manufacture reversals out of a change of vocabulary. +// --------------------------------------------------------------------------- +const STAFF = new Set(["full-time", "salaried"]); +const CONTRACT = new Set(["contractor", "1099"]); +const campOf = (p) => (STAFF.has(p) ? "staff" : CONTRACT.has(p) ? "contract" : null); +const STATUS_WORD = { + "full-time": "full time", + salaried: "salaried", + contractor: "contractors", + 1099: "1099, not full-time", +}; + +/** + * Whether two claims are about overlapping people. + * + * An `all` claim is about every payroll he has, so it is comparable with any + * company; two different companies are not comparable with each other. Without + * that asymmetry the corpus's clearest reversal -- December 2024's ten at the + * channel are "all basically contractors", May 2025's ten or eleven are "full + * time" -- is invisible, because one is scoped to the channel and the other to + * everything. + */ +const comparableScopes = (a, b) => a === b || a === "all" || b === "all"; + +/** + * A stated total below his own most recent claim for ONE company. + * + * The canonical case: 2023-04-21 "how will I pay my six employees?" against + * 2023-04-19 "already… almost 10 staff members" at The Publica, two days + * earlier. Six cannot contain ten. + */ +function contradictsComponent(step, state) { + if (step.scope !== "all" || step.value == null) return null; + const worst = COMPANY_SCOPES.map((c) => state[c]) + .filter((s) => s && s.value != null && s.value > step.value) + .sort((a, b) => b.value - a.value)[0]; + if (!worst) return null; + return { + rule: "contradicts_component", + text: + `he says ${fmt(step.value)} total; he said ${fmt(worst.value)} at ` + + `${SCOPE_LABEL[worst.scope]} ${gapWords(worst.date, step.date)}`, + against: worst.id, + }; +} + +/** Two claims, same scope, same date, different values. */ +function sameDayConflict(step, byScopeDate) { + if (step.value == null) return null; + const peers = (byScopeDate.get(`${step.scope}|${step.date}`) ?? []).filter( + (o) => o.id !== step.id && o.value != null && o.value !== step.value, + ); + if (!peers.length) return null; + const other = peers[0]; + return { + rule: "same_day_conflict", + text: + step.scope === "all" + ? `two different totals the same day — ${fmt(step.value)} and ${fmt(other.value)}` + : `two different ${SCOPE_LABEL[step.scope]} counts the same day — ` + + `${fmt(step.value)} and ${fmt(other.value)}`, + against: other.id, + }; +} + +/** + * The quote's own qualifier undercuts the count: he names a number of + * "employees" and then says in the same breath that they are not employees. + * + * Derived from the adjudicated `population`, not from a regex over the quote — + * whether "basically contractors" undercuts "10 employees" is exactly the + * reading a human is signing off on. + */ +function selfNegating(step) { + if (!UNDERCUTTING.has(step.population)) return null; + if (step.value == null) return null; + return { + rule: "self_negating", + text: `${fmt(step.value)} ${step.population === "1099" ? "— “all 1099 and not full-time”" : "— “all basically contractors”"}`, + }; +} + +/** A different denominator from the rest of its own series. */ +function populationMismatch(step, baseline) { + const base = baseline.get(step.scope) ?? "employees"; + if (step.population === base) return null; + return { + rule: "population_mismatch", + text: `${step.population}, not ${base}`, + baseline: base, + }; +} + +/** + * The same people, described as staff and then as contractors, or the reverse. + * + * Only against the MOST RECENT comparable claim that carries a camp at all -- + * not against every earlier one. Fire on every pair and a single 2022 "all 1099 + * and not full-time" flags each of the next seven claims in turn, which reads + * as seven findings when it is one. Bounded this way, the predicate fires at + * the TRANSITIONS, which is what a flip-flop is. + */ +function statusFlip(step, priorCamps) { + const camp = campOf(step.population); + if (!camp) return null; + let prev = null; + for (let i = priorCamps.length - 1; i >= 0; i -= 1) { + if (comparableScopes(priorCamps[i].scope, step.scope)) { prev = priorCamps[i]; break; } + } + if (!prev || prev.camp === camp) return null; + const when = gapWords(prev.date, step.date); + const same = + prev.value != null && step.value != null && prev.value === step.value + ? `the same ${fmt(step.value)} were` + : "they were"; + return { + rule: "status_flip", + text: `${when} ${same} ${STATUS_WORD[prev.population]}, now ${STATUS_WORD[step.population]}`, + against: prev.id, + }; +} + +/** Our arithmetic or our midpoint, wearing his voice. */ +function notHisNumber(step) { + if (step.valueKind === "uttered") return null; + return { + rule: "not_his_number", + text: + step.valueKind === "derived" + ? "our sum, not his figure" + : "our midpoint, not his figure", + }; +} + +const fmt = (n) => (Number.isInteger(n) ? String(n) : String(n)); + +function gapWords(from, to) { + const a = new Date(`${dateKey(from)}T00:00:00Z`).getTime(); + const b = new Date(`${dateKey(to)}T00:00:00Z`).getTime(); + const days = Math.round((b - a) / 86400000); + if (!Number.isFinite(days)) return "earlier"; + if (days <= 0) return "the same day"; + if (days === 1) return "a day earlier"; + if (days < 31) return `${days} days earlier`; + const months = Math.round(days / 30.44); + if (months < 24) return `${months} month${months === 1 ? "" : "s"} earlier`; + return `${Math.round(days / 365.25)} years earlier`; +} + +// --------------------------------------------------------------------------- +// The walk. +// --------------------------------------------------------------------------- + +/** + * Running per-company state, both totals, per-step deltas and every fired + * predicate — one pass, in date order. + * + * @param {Array<object>} ledger + * @param {{ strict?: boolean }} [opts] strict:false computes over whatever is + * adjudicated so the inbox can show its own coherence rows while the rest of + * the ledger is still being worked. Any renderer must use strict:true. + */ +export function ledgerTotals(ledger, { strict = true } = {}) { + const all = chronological(ledger); + const pending = all.filter((e) => !isAdjudicated(e)); + if (strict && pending.length) throw new UnadjudicatedLedger(pending); + + // Only adjudicated entries take part. An `unresolved` scope is a legitimate + // outcome and MUST NOT silently feed a total, so it is carried on the step + // (the rail still shows the row) and skipped by both series. + const usable = all.filter(isAdjudicated); + + // The baseline population per scope: the most common one in that scope's own + // series, ties going to `employees`. Computed over uttered values only, so a + // pile of derived rows cannot move the baseline the real claims are judged by. + const baseline = new Map(); + for (const scope of SCOPES) { + const counts = new Map(); + for (const e of usable) { + if (e.scope !== scope || e.valueKind !== "uttered") continue; + counts.set(e.population, (counts.get(e.population) ?? 0) + 1); + } + let best = "employees"; + let bestN = counts.get("employees") ?? 0; + for (const [p, n] of counts) if (n > bestN) [best, bestN] = [p, n]; + baseline.set(scope, best); + } + + const byScopeDate = new Map(); + for (const e of usable) { + if (e.scopeConfidence === "unresolved") continue; + const k = `${e.scope}|${e.date}`; + if (!byScopeDate.has(k)) byScopeDate.set(k, []); + byScopeDate.get(k).push(e); + } + + /** Most recent usable claim per company scope, as we walk. */ + const state = { media: null, coffee: null, publica: null }; + /** Every claim so far that says what its people ARE, in order. */ + const priorCamps = []; + let stated = null; + let implied = null; + + const steps = []; + const series = { media: [], coffee: [], publica: [], stated: [], implied: [] }; + + for (const e of usable) { + const counts = e.value != null; + const usableHere = counts && e.scopeConfidence !== "unresolved"; + + const flags = []; + // `not_his_number` and `population_mismatch` judge the entry alone; + // `contradicts_component` and `same_day_conflict` judge it against the walk. + if (usableHere) { + const f1 = contradictsComponent({ ...e }, state); + if (f1) flags.push(f1); + const f2 = sameDayConflict({ ...e }, byScopeDate); + if (f2) flags.push(f2); + const f3 = selfNegating(e); + if (f3) flags.push(f3); + const f5 = notHisNumber(e); + if (f5) flags.push(f5); + } + // OUTSIDE the `usableHere` gate, deliberately. "All my workers are contract + // workers" carries no figure at all, and it is the single clearest status + // claim in the corpus — gating this on a number would drop exactly the + // rows the predicate exists to read. + if (e.scopeConfidence !== "unresolved") { + const f6 = statusFlip(e, priorCamps); + if (f6) flags.push(f6); + } + // Recorded AFTER the predicate reads it, so a claim never flips against + // itself, and only for entries whose scope is settled -- an unresolved + // scope cannot say whose status reversed. + if (e.scopeConfidence !== "unresolved" && campOf(e.population)) { + priorCamps.push({ + id: e.id, scope: e.scope, date: e.date, value: e.value ?? null, + population: e.population, camp: campOf(e.population), + }); + } + const f4 = populationMismatch(e, baseline); + if (f4) flags.push(f4); + // Anything the adjudicator wrote by hand rides alongside the computed ones. + for (const raw of e.flags ?? []) { + if (isStr(raw)) flags.push({ rule: "adjudicator", text: raw }); + } + + const beforeStated = stated; + const beforeImplied = implied; + + if (usableHere && COMPANY_SCOPES.includes(e.scope)) { + state[e.scope] = { id: e.id, scope: e.scope, value: e.value, date: e.date }; + series[e.scope].push({ id: e.id, date: e.date, value: e.value }); + } + + // The IMPLIED total is our sum of his most recent per-company claims. It is + // recomputed after every company step, and it is defined as soon as ONE + // company has a number — a sum of one is still our sum. + const basis = {}; + let sum = null; + for (const c of COMPANY_SCOPES) { + basis[c] = state[c] ? { ...state[c] } : null; + if (state[c]) sum = (sum ?? 0) + state[c].value; + } + implied = sum; + + // The STATED total moves only on an `all`-scope claim he actually utters. + if (usableHere && e.scope === "all" && e.valueKind === "uttered") { + stated = e.value; + series.stated.push({ id: e.id, date: e.date, value: e.value }); + } + + if (implied !== beforeImplied) { + series.implied.push({ id: e.id, date: e.date, value: implied }); + } + + const movedStated = stated !== beforeStated; + const movedImplied = implied !== beforeImplied; + + steps.push({ + id: e.id, + date: e.date, + scope: e.scope, + scopeConfidence: e.scopeConfidence, + scopeBasis: e.scopeBasis, + population: e.population, + valueKind: e.valueKind, + value: e.value ?? null, + display: e.display ?? null, + label: e.label ?? null, + quote: e.quote ?? null, + src: e.src ?? null, + roles: Array.isArray(e.roles) && e.roles.length ? e.roles : null, + entryId: e.entryId ?? null, + // No y position, ever. "several", "very few" and "+1" are a tick below the + // time axis and a rail row; forcing them onto the value axis would be + // inventing a number, which is the thing this whole module exists to stop. + qualitative: !counts, + stated, + implied, + statedDelta: movedStated && beforeStated != null ? stated - beforeStated : null, + impliedDelta: movedImplied && beforeImplied != null ? implied - beforeImplied : null, + gap: stated != null && implied != null ? implied - stated : null, + impliedBasis: basis, + // Which number rolled. A `value: null` claim moves neither, and the + // readout must hold both rather than animate a change that did not happen. + moved: movedStated && movedImplied ? "both" : movedStated ? "stated" : movedImplied ? "implied" : null, + flags, + }); + } + + return { + steps, + series, + baseline: Object.fromEntries(baseline), + unresolved: usable.filter((e) => e.scopeConfidence === "unresolved").map((e) => e.id), + unadjudicated: pending.map((e) => e.id), + final: { + stated, + implied, + gap: stated != null && implied != null ? implied - stated : null, + }, + }; +} + +// --------------------------------------------------------------------------- +// The roster +// --------------------------------------------------------------------------- +// Every time he enumerates WHO works for him it is two video editors and a +// graphics designer -- in July 2023, September 2023, October 2023, December +// 2023 and December 2024. The totals attached to that roster are three, then +// four, then ten. So the roster is the control: it is the thing that does not +// move while the numbers above it do, and drawing it is what makes that +// visible without the video having to assert it. +// +// `roles` is OPTIONAL and is not one of the six. A claim with no roster is not +// unadjudicated -- most claims are a number and nothing else, and gating on a +// field that only five entries can ever carry would block the inbox forever. + +/** One enumerated role. `verbatim` is his words; the rest is our reading. */ +export const ROLE_FIELDS = ["role", "count", "verbatim"]; + +/** Is this a usable roster? Returns the reasons it is not. */ +export function rolesGaps(roles) { + if (roles === undefined) return []; + if (!Array.isArray(roles) || !roles.length) return ["roles must be a non-empty array"]; + const bad = []; + roles.forEach((r, i) => { + if (!isStr(r?.role)) bad.push(`roles[${i}].role`); + if (!Number.isFinite(r?.count) || r.count < 0) bad.push(`roles[${i}].count`); + if (!isStr(r?.verbatim)) bad.push(`roles[${i}].verbatim`); + }); + return bad; +} + +/** + * The roster he last enumerated at or before `date`, or null. + * + * One implementation, because the rail draws it under the tally and a chapter + * card states it in words, and the two disagreeing would be the video arguing + * with itself on screen. + */ +export function rosterAt(ledger, date) { + const key = dateKey(date); + let best = null; + for (const e of chronological(ledger)) { + if (!Array.isArray(e.roles) || !e.roles.length) continue; + if (dateKey(e.date) > key) break; + best = e; + } + return best ? { id: best.id, date: best.date, roles: best.roles } : null; +} + +/** + * `2 editors · 1 designer` — the roster in as few words as it can be said in. + * + * The LAST word of the role is the label, because that is the part that carries + * the meaning ("video editor" → editors, "graphics designer" → designers) and + * the rail column has room for nothing else. + */ +export function rosterLine(roles) { + if (!Array.isArray(roles) || !roles.length) return null; + return roles + .map((r) => { + const head = String(r.role).trim().split(/\s+/).pop(); + return `${r.count} ${head}${r.count === 1 ? "" : "s"}`; + }) + .join(" · "); +} + +/** Every fired predicate, flattened — what the `claim-incoherent` rows are. */ +export function coherenceFlags(ledger, opts) { + return ledgerTotals(ledger, opts).steps.flatMap((s) => + s.flags.map((f) => ({ id: s.id, date: s.date, scope: s.scope, ...f })), + ); +} diff --git a/scripts/report-to-video/ledger-totals.test.mjs b/scripts/report-to-video/ledger-totals.test.mjs @@ -0,0 +1,705 @@ +// Tests for ledger-totals.mjs. +// +// The fixture is a hand-built miniature of the real quartering-employee-count +// ledger: the same shapes, the same two named landmines, and nothing else. Run +// with: pnpm test:scripts +import assert from "node:assert/strict"; +import test from "node:test"; + +import { + ADJUDICATION_FIELDS, + PREDICATES, + UnadjudicatedLedger, + adjudicationGaps, + chronological, + coherenceFlags, + dateKey, + isAdjudicated, + ledgerTotals, + rolesGaps, + rosterAt, + rosterLine, +} from "./ledger-totals.mjs"; +import { scheduleClaims, selectVariant } from "./build-video.mjs"; +import { ledgerRevealAt, ledgerSeconds } from "./render-cards.mjs"; + +/** An adjudicated entry, with the six fields defaulted to the boring answer. */ +const claim = (o) => ({ + scope: "media", + scopeBasis: "names the channel", + scopeConfidence: "clear", + population: "employees", + valueKind: "uttered", + flags: [], + ...o, +}); + +// --------------------------------------------------------------------------- +// The gate +// --------------------------------------------------------------------------- + +test("adjudicationGaps names every missing field", () => { + assert.deepEqual(adjudicationGaps({ id: "x" }).sort(), [...ADJUDICATION_FIELDS].sort()); + assert.deepEqual(adjudicationGaps(claim({ id: "x" })), []); + assert.equal(isAdjudicated(claim({ id: "x" })), true); +}); + +test("an empty scopeBasis is not an adjudication", () => { + // An adjudication with no basis is an opinion, and the point of the exercise + // was to stop shipping those. + assert.deepEqual(adjudicationGaps(claim({ id: "x", scopeBasis: " " })), ["scopeBasis"]); +}); + +test("each of the six is checked against its own vocabulary", () => { + assert.deepEqual(adjudicationGaps(claim({ scope: "everything" })), ["scope"]); + assert.deepEqual(adjudicationGaps(claim({ population: "staff" })), ["population"]); + assert.deepEqual(adjudicationGaps(claim({ valueKind: "guessed" })), ["valueKind"]); + assert.deepEqual(adjudicationGaps(claim({ scopeConfidence: "maybe" })), ["scopeConfidence"]); + assert.deepEqual(adjudicationGaps(claim({ flags: "none" })), ["flags"]); +}); + +test("ledgerTotals refuses to run on an unadjudicated ledger", () => { + const led = [claim({ id: "a", date: "2022-01-01", value: 4 }), { id: "b", date: "2022-02-01", value: 5 }]; + assert.throws(() => ledgerTotals(led), UnadjudicatedLedger); + assert.throws(() => ledgerTotals(led), /b/); + // …but computes over what it has when the inbox asks. + const soft = ledgerTotals(led, { strict: false }); + assert.deepEqual(soft.unadjudicated, ["b"]); + assert.equal(soft.steps.length, 1); +}); + +// --------------------------------------------------------------------------- +// Ordering +// --------------------------------------------------------------------------- + +test("month-only dates sort to the first of the month", () => { + assert.equal(dateKey("2024-02"), "2024-02-01"); + assert.equal(dateKey("2024-02-15"), "2024-02-15"); + assert.equal(dateKey("garbage"), "9999-99-99"); + const ids = chronological([ + { id: "b", date: "2024-02-15" }, + { id: "a", date: "2024-02" }, + ]).map((e) => e.id); + assert.deepEqual(ids, ["a", "b"]); +}); + +test("ties break on id, so the order is total and stable", () => { + const ids = chronological([ + { id: "z", date: "2024-01-01" }, + { id: "a", date: "2024-01-01" }, + ]).map((e) => e.id); + assert.deepEqual(ids, ["a", "z"]); +}); + +// --------------------------------------------------------------------------- +// The two series +// --------------------------------------------------------------------------- + +const SERIES_FIXTURE = [ + claim({ id: "m1", date: "2022-11-02", scope: "media", value: 5 }), + claim({ id: "c1", date: "2023-04-19", scope: "coffee", value: 5, scopeBasis: "names coffee" }), + claim({ id: "p1", date: "2023-04-19", scope: "publica", value: 10, scopeBasis: "names the publica" }), + claim({ id: "e0", date: "2022-08-30", scope: "all", value: 4, scopeBasis: "no company named" }), +]; + +test("the implied total is our sum of his most recent per-company claims", () => { + const t = ledgerTotals(SERIES_FIXTURE); + const at = (id) => t.steps.find((s) => s.id === id); + // 2022-08-30 is the FIRST entry: no company has spoken, so there is nothing + // to sum. Our sum of no claims is not zero, it is undefined. + assert.equal(at("e0").implied, null); + assert.equal(at("e0").gap, null); + // 2022-11-02: media alone. A sum of one is still our sum. + assert.equal(at("m1").implied, 5); + // 2023-04-19: media 5 + coffee 5 + publica 10. + assert.equal(at("p1").implied, 20); + assert.equal(t.final.implied, 20); +}); + +test("the stated total is the last figure he utters for the whole payroll", () => { + const t = ledgerTotals(SERIES_FIXTURE); + assert.equal(t.final.stated, 4); + // The exact landmark the plan calls for: implied 20 against stated 4. + assert.equal(t.steps.find((s) => s.id === "p1").gap, 16); + assert.deepEqual( + t.series.stated.map((p) => [p.id, p.value]), + [["e0", 4]], + ); +}); + +test("a derived sum never enters the stated series", () => { + // This is the correction the whole re-cut turns on: 18 and 20 are OUR + // arithmetic. He never utters either, so neither may be plotted as his claim. + const led = [ + claim({ id: "c", date: "2024-11-23", scope: "coffee", value: 10, scopeBasis: "names coffee" }), + claim({ id: "m", date: "2024-11-23", scope: "media", value: 8 }), + claim({ + id: "e8", + date: "2024-11-23", + scope: "all", + value: 18, + valueKind: "derived", + scopeBasis: "10 + 8, summed by us", + }), + ]; + const t = ledgerTotals(led); + assert.deepEqual(t.series.stated, []); + assert.equal(t.final.stated, null); + assert.equal(t.final.implied, 18); + const flags = t.steps.find((s) => s.id === "e8").flags.map((f) => f.rule); + assert.ok(flags.includes("not_his_number")); +}); + +test("a synthetic midpoint never enters the stated series either", () => { + const t = ledgerTotals([ + claim({ + id: "e11", + date: "2025-05-30", + scope: "all", + value: 10.5, + valueKind: "synthetic", + population: "people", + scopeBasis: "“about 10 people, 11 people full time”", + }), + ]); + assert.equal(t.final.stated, null); + const f = t.steps[0].flags.find((x) => x.rule === "not_his_number"); + assert.equal(f.text, "our midpoint, not his figure"); +}); + +test("an unresolved scope feeds neither total", () => { + const t = ledgerTotals([ + claim({ id: "m1", date: "2022-01-01", scope: "media", value: 5 }), + claim({ + id: "e?", + date: "2024-08-14", + scope: "all", + value: 10, + scopeConfidence: "unresolved", + scopeBasis: "“I have 10 employees, my coffee company employees…” — comma decides it", + }), + ]); + assert.equal(t.final.stated, null, "an unresolved claim must not become the stated total"); + assert.equal(t.final.implied, 5); + assert.deepEqual(t.unresolved, ["e?"]); +}); + +test("a qualitative claim has no y position and moves neither number", () => { + const t = ledgerTotals([ + claim({ id: "m1", date: "2022-01-01", scope: "media", value: 5 }), + claim({ id: "m12", date: "2024-02", scope: "media", value: null, scopeBasis: "hires a publisher" }), + ]); + const q = t.steps.find((s) => s.id === "m12"); + assert.equal(q.qualitative, true); + assert.equal(q.value, null); + assert.equal(q.moved, null, "a value:null claim must move neither number"); + assert.equal(q.implied, 5); +}); + +test("deltas report the step, and only on the number that moved", () => { + const t = ledgerTotals([ + claim({ id: "m1", date: "2022-01-01", scope: "media", value: 5 }), + claim({ id: "m2", date: "2022-02-01", scope: "media", value: 3 }), + ]); + const s = t.steps.find((x) => x.id === "m2"); + assert.equal(s.impliedDelta, -2); + assert.equal(s.statedDelta, null); + assert.equal(s.moved, "implied"); +}); + +// --------------------------------------------------------------------------- +// The five predicates +// --------------------------------------------------------------------------- + +test("contradicts_component: stated 6 against The Publica's 10 two days earlier", () => { + // The case the plan names. Six cannot contain ten. + const t = ledgerTotals([ + claim({ id: "p01", date: "2023-04-19", scope: "publica", value: 10, scopeBasis: "names the publica" }), + claim({ id: "e04", date: "2023-04-21", scope: "all", value: 6, scopeBasis: "no company named" }), + ]); + const f = t.steps.find((s) => s.id === "e04").flags.find((x) => x.rule === "contradicts_component"); + assert.ok(f, "contradicts_component must fire"); + assert.equal(f.against, "p01"); + assert.equal(f.text, "he says 6 total; he said 10 at The Publica 2 days earlier"); +}); + +test("contradicts_component does not fire when the total covers every component", () => { + const t = ledgerTotals([ + claim({ id: "p", date: "2023-04-19", scope: "publica", value: 4, scopeBasis: "b" }), + claim({ id: "e", date: "2023-04-21", scope: "all", value: 9, scopeBasis: "b" }), + ]); + assert.equal( + t.steps.find((s) => s.id === "e").flags.filter((f) => f.rule === "contradicts_component").length, + 0, + ); +}); + +test("same_day_conflict: two different totals on 2024-12-18", () => { + // The other case the plan names: 10, then 10+10 as 20, same day, same scope. + const t = ledgerTotals([ + claim({ id: "e09", date: "2024-12-18", scope: "all", value: 10, scopeBasis: "no company named" }), + claim({ + id: "e10", + date: "2024-12-18", + scope: "all", + value: 20, + valueKind: "derived", + scopeBasis: "10 + coffee's 10, summed by us", + }), + ]); + for (const id of ["e09", "e10"]) { + const f = t.steps.find((s) => s.id === id).flags.find((x) => x.rule === "same_day_conflict"); + assert.ok(f, `same_day_conflict must fire on ${id}`); + assert.match(f.text, /two different totals the same day/); + } +}); + +test("same_day_conflict ignores two claims about different companies", () => { + const t = ledgerTotals([ + claim({ id: "m", date: "2023-08-09", scope: "media", value: 4 }), + claim({ id: "c", date: "2023-08-09", scope: "coffee", value: 10, scopeBasis: "names coffee" }), + ]); + assert.equal(coherenceFlags([]).length, 0); + for (const s of t.steps) { + assert.equal(s.flags.filter((f) => f.rule === "same_day_conflict").length, 0); + } +}); + +test("same_day_conflict ignores a repeated identical figure", () => { + const t = ledgerTotals([ + claim({ id: "a", date: "2023-08-09", scope: "coffee", value: 10, scopeBasis: "x" }), + claim({ id: "b", date: "2023-08-09", scope: "coffee", value: 10, scopeBasis: "x" }), + ]); + for (const s of t.steps) { + assert.equal(s.flags.filter((f) => f.rule === "same_day_conflict").length, 0); + } +}); + +test("self_negating: a count of employees who are not employees", () => { + const t = ledgerTotals([ + claim({ + id: "e09", + date: "2024-12-18", + scope: "all", + value: 10, + population: "contractor", + scopeBasis: "no company named", + }), + ]); + const f = t.steps[0].flags.find((x) => x.rule === "self_negating"); + assert.ok(f); + assert.match(f.text, /basically contractors/); +}); + +test("population_mismatch: salaried against an employees baseline", () => { + const led = [ + claim({ id: "a", date: "2022-01-01", scope: "all", value: 4, scopeBasis: "x" }), + claim({ id: "b", date: "2023-01-01", scope: "all", value: 6, scopeBasis: "x" }), + claim({ id: "c", date: "2024-09-10", scope: "all", value: 5, population: "salaried", scopeBasis: "x" }), + ]; + const t = ledgerTotals(led); + assert.equal(t.baseline.all, "employees"); + const f = t.steps.find((s) => s.id === "c").flags.find((x) => x.rule === "population_mismatch"); + assert.equal(f.text, "salaried, not employees"); + // …and the two that match the baseline stay clean. + for (const id of ["a", "b"]) { + assert.equal( + t.steps.find((s) => s.id === id).flags.filter((x) => x.rule === "population_mismatch").length, + 0, + ); + } +}); + +test("the baseline is per scope, not global", () => { + const t = ledgerTotals([ + claim({ id: "a", date: "2022-01-01", scope: "media", value: 4, population: "full-time" }), + claim({ id: "b", date: "2022-02-01", scope: "media", value: 5, population: "full-time" }), + claim({ id: "c", date: "2022-03-01", scope: "coffee", value: 6, scopeBasis: "x" }), + ]); + assert.equal(t.baseline.media, "full-time"); + assert.equal(t.baseline.coffee, "employees"); + assert.equal(t.steps.find((s) => s.id === "c").flags.length, 0); +}); + +test("a large swing is deliberately NOT a flag", () => { + // Fluctuation is the subject of the video, not a defect. Flagging it would be + // putting a thumb on the scale. + const t = ledgerTotals([ + claim({ id: "a", date: "2025-02-12", scope: "media", value: 10 }), + claim({ id: "b", date: "2025-06-20", scope: "media", value: 3 }), + ]); + assert.deepEqual(t.steps.find((s) => s.id === "b").flags, []); +}); + +test("an adjudicator's hand-written flag rides alongside the computed ones", () => { + const t = ledgerTotals([ + claim({ id: "a", date: "2025-02-12", scope: "media", value: 10, flags: ["reading someone else's tweet"] }), + ]); + assert.deepEqual(t.steps[0].flags, [{ rule: "adjudicator", text: "reading someone else's tweet" }]); +}); + +test("coherenceFlags flattens every fired predicate with its claim id", () => { + const flags = coherenceFlags([ + claim({ id: "p", date: "2023-04-19", scope: "publica", value: 10, scopeBasis: "x" }), + claim({ + id: "e", + date: "2023-04-21", + scope: "all", + value: 6, + population: "salaried", + valueKind: "derived", + scopeBasis: "x", + }), + ]); + const mine = flags.filter((f) => f.id === "e").map((f) => f.rule).sort(); + assert.deepEqual(mine, ["contradicts_component", "not_his_number", "population_mismatch"]); +}); + +// --------------------------------------------------------------------------- +// The editorial decisions this corpus actually turned on. +// +// These are shaped like the real ledger rather than reading it: the file lives +// outside the repo, and a test that loads it would fail for whoever does not +// have that report checked out. What they pin is the RULING, so a later change +// to the adjudication has to come past a red test rather than quietly restating +// the video's argument. +// --------------------------------------------------------------------------- + +test("December 2024: two tens on one day, and no stated total at all", () => { + // He says ten at the channel and ten at the coffee company. Our sum is 20 and + // he never utters it, so the implied line steps and the stated line does not + // move — which is the whole correction, in one day of the corpus. + const t = ledgerTotals([ + claim({ id: "m", date: "2024-11-27", scope: "media", value: 10 }), + claim({ id: "c13", date: "2024-12-18", scope: "coffee", value: 10, scopeBasis: "names coffee" }), + claim({ + id: "e09", date: "2024-12-18", scope: "media", value: 10, population: "contractor", + scopeBasis: "“plus Coffee Brand Coffee, which ALSO has 10” — coffee is outside the ten", + }), + claim({ + id: "e10", date: "2024-12-18", scope: "all", value: 20, valueKind: "derived", + scopeBasis: "10 + 10, summed by us", + }), + ]); + assert.equal(t.final.implied, 20); + assert.equal(t.final.stated, null, "no figure he utters covers the whole payroll that day"); + assert.deepEqual(t.series.stated, []); + // e09 and c13 are the SAME day but different scopes, so this is not a + // same-day conflict — it is two companies, which is the point. + const e09 = t.steps.find((s) => s.id === "e09"); + assert.equal(e09.flags.filter((f) => f.rule === "same_day_conflict").length, 0); + assert.ok(e09.flags.some((f) => f.rule === "self_negating")); +}); + +test("a stated total below a component fires even months later", () => { + // 2024-09-10: five salaried across four ventures, against ten at the coffee + // company eleven months earlier. The component claim is stale, not wrong, and + // a total that cannot contain it is still incoherent. + const t = ledgerTotals([ + claim({ id: "c09", date: "2023-09-27", scope: "coffee", value: 10, scopeBasis: "names coffee" }), + claim({ + id: "e07", date: "2024-09-10", scope: "all", value: 5, population: "salaried", + scopeBasis: "enumerates meme intros, thumbnails, tailgates and thepublica.com", + }), + ]); + const f = t.steps.find((s) => s.id === "e07").flags.find((x) => x.rule === "contradicts_component"); + assert.ok(f); + assert.equal(f.against, "c09"); + assert.match(f.text, /11 months earlier/); +}); + +test("the implied total carries a company forward until he speaks about it again", () => { + // The Publica is claimed once, in April 2023, and never again. Our sum keeps + // carrying that ten for years — which is honest arithmetic over his claims and + // exactly why the series must be labelled as ours. + const t = ledgerTotals([ + claim({ id: "p", date: "2023-04-19", scope: "publica", value: 10, scopeBasis: "launch video" }), + claim({ id: "m", date: "2025-06-20", scope: "media", value: 3, population: "people" }), + ]); + assert.equal(t.final.implied, 13); + assert.equal(t.series.publica.length, 1, "one Publica claim in the whole corpus"); +}); + +// --------------------------------------------------------------------------- +// status_flip — the sixth predicate +// --------------------------------------------------------------------------- + +test("status_flip: December's contractors are May's full-timers", () => { + // The corpus's clearest reversal, and the reason the predicate compares + // across scopes when one of them is `all`: the ten are the CHANNEL's in + // December and the WHOLE payroll's in May, and reading those as unrelated + // series is how the flip stayed invisible. + const t = ledgerTotals([ + claim({ + id: "e09", date: "2024-12-18", scope: "media", value: 10, population: "contractor", + scopeBasis: "“plus Coffee Brand Coffee, which ALSO has 10”", + }), + claim({ + id: "e11", date: "2025-05-30", scope: "all", value: 10.5, population: "full-time", + valueKind: "synthetic", scopeBasis: "“about 10 people, 11 people full time”", + }), + ]); + const f = t.steps.find((s) => s.id === "e11").flags.find((x) => x.rule === "status_flip"); + assert.ok(f, "status_flip must fire on the reversal"); + assert.equal(f.against, "e09"); + assert.equal(f.text, "5 months earlier they were contractors, now full time"); + // …and the December claim itself has nothing before it to reverse. + assert.equal( + t.steps.find((s) => s.id === "e09").flags.filter((x) => x.rule === "status_flip").length, + 0, + ); +}); + +test("status_flip names the figure when it is the same one", () => { + const t = ledgerTotals([ + claim({ id: "a", date: "2024-01-01", scope: "all", value: 10, population: "contractor", scopeBasis: "x" }), + claim({ id: "b", date: "2024-03-01", scope: "all", value: 10, population: "full-time", scopeBasis: "x" }), + ]); + const f = t.steps.find((s) => s.id === "b").flags.find((x) => x.rule === "status_flip"); + assert.equal(f.text, "2 months earlier the same 10 were contractors, now full time"); +}); + +test("status_flip fires on a claim carrying no figure at all", () => { + // "That's why all my workers are contract workers" is the single clearest + // status claim in the corpus and it names no number. Gating the predicate on + // a value would drop exactly the rows it exists to read. + const t = ledgerTotals([ + claim({ id: "e11", date: "2025-05-30", scope: "all", value: 10.5, population: "full-time", + valueKind: "synthetic", scopeBasis: "x" }), + claim({ id: "e12", date: "2025-10-29", scope: "all", value: null, population: "contractor", + scopeBasis: "“all my workers are contract workers”" }), + ]); + const f = t.steps.find((s) => s.id === "e12").flags.find((x) => x.rule === "status_flip"); + assert.ok(f); + assert.equal(f.against, "e11"); +}); + +test("status_flip only fires at the transition, not against every earlier claim", () => { + // One 2022 "all 1099" would otherwise flag each of the next four claims in + // turn, which reads as four findings when it is one. + const led = [ + claim({ id: "e01", date: "2022-03-16", scope: "all", value: null, population: "1099", scopeBasis: "x" }), + claim({ id: "a", date: "2023-01-01", scope: "all", value: 4, population: "full-time", scopeBasis: "x" }), + claim({ id: "b", date: "2023-06-01", scope: "all", value: 5, population: "full-time", scopeBasis: "x" }), + claim({ id: "c", date: "2023-09-01", scope: "all", value: 6, population: "salaried", scopeBasis: "x" }), + ]; + const fired = ledgerTotals(led).steps + .filter((s) => s.flags.some((f) => f.rule === "status_flip")) + .map((s) => s.id); + assert.deepEqual(fired, ["a"]); +}); + +test("status_flip does not read `employees` or `people` as a status at all", () => { + // He says "employees" when he is not making a claim about status. Treating + // it as one side or the other manufactures a reversal out of vocabulary. + const t = ledgerTotals([ + claim({ id: "a", date: "2024-01-01", scope: "media", value: 4, population: "full-time" }), + claim({ id: "b", date: "2024-06-01", scope: "media", value: 5, population: "employees" }), + claim({ id: "c", date: "2024-09-01", scope: "media", value: 6, population: "people" }), + ]); + for (const s of t.steps) { + assert.equal(s.flags.filter((f) => f.rule === "status_flip").length, 0); + } +}); + +test("status_flip does not compare two different companies", () => { + const t = ledgerTotals([ + claim({ id: "c", date: "2023-06-30", scope: "coffee", value: 6, population: "full-time", scopeBasis: "x" }), + claim({ id: "p", date: "2023-09-01", scope: "publica", value: 3, population: "contractor", scopeBasis: "x" }), + ]); + assert.equal( + t.steps.find((s) => s.id === "p").flags.filter((f) => f.rule === "status_flip").length, + 0, + ); +}); + +test("an unresolved scope cannot say whose status reversed", () => { + const t = ledgerTotals([ + claim({ id: "u", date: "2022-03-16", scope: "all", value: null, population: "salaried", + scopeConfidence: "unresolved", scopeBasis: "spoken on someone else's channel" }), + claim({ id: "v", date: "2022-08-30", scope: "all", value: 4, population: "contractor", scopeBasis: "x" }), + ]); + for (const s of t.steps) { + assert.equal(s.flags.filter((f) => f.rule === "status_flip").length, 0); + } +}); + +test("status_flip is one of the named predicates", () => { + assert.ok(PREDICATES.includes("status_flip")); + assert.equal(PREDICATES.length, 6); +}); + +// --------------------------------------------------------------------------- +// The roster +// --------------------------------------------------------------------------- + +const ROSTERED = [ + claim({ + id: "m07", date: "2023-10-17", scope: "media", value: 3, population: "full-time", + roles: [ + { role: "video editor", count: 2, verbatim: "two video editors" }, + { role: "graphics designer", count: 1, verbatim: "a graphics designer" }, + ], + }), + claim({ + id: "m08", date: "2023-12-04", scope: "media", value: 4, population: "full-time", + roles: [ + { role: "video editor", count: 2, verbatim: "two video editors" }, + { role: "graphic designer", count: 1, verbatim: "a graphic designer" }, + ], + }), +]; + +test("rosterAt: the same two editors and one designer, whatever the total says", () => { + // This is the whole finding. October's three and December's four are the SAME + // roster; the total moved and the people did not. + assert.equal(rosterAt(ROSTERED, "2023-10-17").id, "m07"); + assert.equal(rosterAt(ROSTERED, "2023-11-30").id, "m07"); + assert.equal(rosterAt(ROSTERED, "2023-12-04").id, "m08"); + assert.equal(rosterAt(ROSTERED, "2026-01-01").id, "m08"); + assert.equal(rosterAt(ROSTERED, "2023-01-01"), null, "nothing enumerated yet is not a roster"); + assert.equal(rosterLine(rosterAt(ROSTERED, "2023-10-17").roles), "2 editors · 1 designer"); + assert.equal(rosterLine(rosterAt(ROSTERED, "2023-12-04").roles), "2 editors · 1 designer"); +}); + +test("rosterAt ignores claims that enumerate nothing", () => { + const led = [...ROSTERED, claim({ id: "m19", date: "2025-06-20", scope: "media", value: 3, population: "people" })]; + assert.equal(rosterAt(led, "2025-12-01").id, "m08"); +}); + +test("roles are optional, and checked when present", () => { + assert.deepEqual(rolesGaps(undefined), []); + assert.deepEqual(adjudicationGaps(claim({ id: "x" })), [], "a claim with no roster is still adjudicated"); + assert.deepEqual(rolesGaps([{ role: "video editor", count: 2, verbatim: "two video editors" }]), []); + assert.deepEqual(rolesGaps([{ role: "", count: 2, verbatim: "x" }]), ["roles[0].role"]); + assert.deepEqual(rolesGaps([{ role: "a", count: null, verbatim: "" }]), ["roles[0].count", "roles[0].verbatim"]); + assert.deepEqual(rolesGaps([]), ["roles must be a non-empty array"]); +}); + +// --------------------------------------------------------------------------- +// Deleting the two derived rows +// --------------------------------------------------------------------------- + +test("removing a derived `all` row moves neither series", () => { + // 18 ("10+8") and 20 ("10+10") are our arithmetic over rows that are already + // in the ledger. Taking them out has to change nothing at all — which is the + // whole justification for taking them out. + const real = [ + claim({ id: "c12", date: "2024-11-23", scope: "coffee", value: 10, scopeBasis: "names coffee" }), + claim({ id: "m14", date: "2024-11-23", scope: "media", value: 8 }), + claim({ id: "e07", date: "2024-09-10", scope: "all", value: 5, population: "salaried", scopeBasis: "x" }), + ]; + const derived = claim({ + id: "e08", date: "2024-11-23", scope: "all", value: 18, valueKind: "derived", + scopeBasis: "10 + 8, summed by us", + }); + const withRow = ledgerTotals([...real, derived]); + const without = ledgerTotals(real); + assert.deepEqual(without.final, withRow.final); + assert.deepEqual( + without.series.implied.map((p) => p.value), + withRow.series.implied.map((p) => p.value), + ); + assert.deepEqual(without.series.stated, withRow.series.stated); +}); + +// --------------------------------------------------------------------------- +// The variant filter +// --------------------------------------------------------------------------- + +const VARIANT_MANIFEST = { + timeline: [ + { type: "card", id: "t00", style: "title", heading: "all 50", variants: { sourced: { heading: "the 20" } } }, + { type: "clip", id: "a01" }, + { type: "ledger", id: "L1", variant: "full" }, + ], + ledger: [ + { id: "m02", entryId: "a01" }, + { id: "m03", entryId: "L1" }, + { id: "m04" }, + ], +}; + +test("selectVariant keeps only the claims whose entry survives", () => { + const sourced = selectVariant(VARIANT_MANIFEST, "sourced"); + assert.deepEqual(sourced.timeline.map((e) => e.id), ["t00", "a01"]); + assert.deepEqual(sourced.ledger.map((c) => c.id), ["m02"]); + + const full = selectVariant(VARIANT_MANIFEST, "full"); + assert.deepEqual(full.timeline.map((e) => e.id), ["t00", "a01", "L1"]); + // m04 is pinned to nothing at all, so it is in neither cut — in `full` every + // claim earns an entry, which is what makes the same filter serve both. + assert.deepEqual(full.ledger.map((c) => c.id), ["m02", "m03"]); +}); + +test("selectVariant merges a card's per-variant copy and drops the override key", () => { + const t = selectVariant(VARIANT_MANIFEST, "sourced").timeline[0]; + assert.equal(t.heading, "the 20"); + assert.equal(t.variants, undefined); + assert.equal(selectVariant(VARIANT_MANIFEST, "full").timeline[0].heading, "all 50"); +}); + +test("selectVariant refuses a variant nobody defined", () => { + assert.throws(() => selectVariant(VARIANT_MANIFEST, "director's cut"), /unknown variant/); +}); + +test("selectVariant does not mutate the manifest it was given", () => { + const before = JSON.stringify(VARIANT_MANIFEST); + selectVariant(VARIANT_MANIFEST, "sourced"); + selectVariant(VARIANT_MANIFEST, "full"); + assert.equal(JSON.stringify(VARIANT_MANIFEST), before); +}); + +// --------------------------------------------------------------------------- +// Scheduling claims onto the timeline +// --------------------------------------------------------------------------- + +test("a claim on a stacked ledger card is pinned to its own row's reveal", () => { + // Pinning all three to the segment's mid-dissolve lands three rail rows on + // one frame AND breaks the pin-order guard, which needs strict monotonicity. + const entries = [ + { type: "card", id: "t00" }, + { type: "clip", id: "a01" }, + { type: "ledger", id: "L01", claims: ["m03", "c02", "c03"] }, + { type: "clip", id: "b01" }, + ]; + const starts = [0, 6, 16, 30]; + const ledger = [ + { id: "m02", entryId: "a01" }, + { id: "m03", entryId: "L01" }, + { id: "c02", entryId: "L01" }, + { id: "c03", entryId: "L01" }, + { id: "c01", entryId: "b01" }, + ]; + const at = scheduleClaims(ledger, entries, starts, 0.4, 60); + assert.equal(at[0], 6.2, "a clipped claim still lands mid-dissolve"); + assert.equal(at[1], 16 + ledgerRevealAt(0)); + assert.equal(at[2], 16 + ledgerRevealAt(1)); + assert.equal(at[3], 16 + ledgerRevealAt(2)); + assert.equal(at[4], 30.2); + for (let i = 1; i < at.length; i += 1) assert.ok(at[i] > at[i - 1], `pin ${i} must move forward`); +}); + +test("a card's own reveals all finish inside the card", () => { + // `seconds` is derived from the same clock the pins read; if it were authored + // by hand a short card would schedule rail rows past its own last frame. + for (const n of [1, 2, 3, 4]) { + assert.ok(ledgerRevealAt(n - 1) < ledgerSeconds(n), `${n} rows must all reveal`); + } + // Float arithmetic, so compare to the tolerance a frame actually has. + assert.ok(Math.abs(ledgerSeconds(4) - (2.2 + 1.3 * 4)) < 1e-9); +}); + +test("a pin that runs backwards is refused, not smoothed over", () => { + // The ledger and the timeline disagreeing about the order of events is a + // manifest bug, and the whole cut rests on the two agreeing. + const entries = [{ type: "clip", id: "a" }, { type: "clip", id: "b" }]; + assert.throws( + () => + scheduleClaims( + [{ id: "x", entryId: "b" }, { id: "y", entryId: "a" }], + entries, [0, 10], 0.4, 30, + ), + /not in the cut's order/, + ); +}); diff --git a/scripts/report-to-video/package.json b/scripts/report-to-video/package.json @@ -0,0 +1,25 @@ +{ + "name": "report-to-video", + "version": "0.1.0", + "private": true, + "type": "module", + "description": "Turn a cited sweep report into a narrated-by-text video.", + "bin": { + "report-build-video": "./build-video.mjs", + "report-resolve-windows": "./resolve-windows.mjs", + "report-check-availability": "./check-availability.mjs", + "report-verify-build": "./verify-build.mjs", + "report-compose-chrome": "./compose-chrome.mjs" + }, + "exports": { + "./build-video": "./build-video.mjs", + "./check-availability": "./check-availability.mjs", + "./compose-chrome": "./compose-chrome.mjs", + "./cues": "./cues.mjs", + "./ledger-totals": "./ledger-totals.mjs", + "./package.json": "./package.json", + "./render-cards": "./render-cards.mjs", + "./resolve-windows": "./resolve-windows.mjs", + "./verify-build": "./verify-build.mjs" + } +} diff --git a/scripts/report-to-video/render-cards.mjs b/scripts/report-to-video/render-cards.mjs @@ -0,0 +1,1530 @@ +#!/usr/bin/env node +// render-cards.mjs — turn a video manifest's `card` entries into PNG stills. +// +// One PNG per card, written to <outDir>/cards/<id>.png at the manifest's render +// resolution. Text is laid out by ImageMagick's Pango delegate rather than +// ffmpeg's drawtext: Pango wraps, kerns and takes inline markup, so a card is a +// single markup string instead of a stack of hand-positioned drawtext filters. +// +// Card styles (manifest `style` field): +// title — the opening card: big heading, subtitle, provenance footer +// chapter — an act break: small amber kicker over a large heading +// bullets — heading plus a list of caveats +// sources — closing attribution +// +// `status` is RETIRED. It existed to quote a claim whose source had gone, and +// the `ledger` entry type does that better: it says the same words, alongside +// the arithmetic the claim moves, and it says WHY there is no footage from a +// probe rather than from a hand-written kicker that nothing re-checks. +// +// In the app: not used. On the CLI: +// node scripts/report-to-video/render-cards.mjs <manifest.json> [--out <dir>] +// +// Options: +// --out <dir> Output root (default: the manifest's directory + /out) +// --only <id> Render just one card, by manifest id +// +// Requires: ImageMagick built with Pango (magick -list format | grep PANGO). + +import { execFile } from "node:child_process"; +import { promisify } from "node:util"; +import { mkdir, writeFile, readFile } from "node:fs/promises"; +import path from "node:path"; +import { dateKey, ledgerTotals, rosterLine } from "./ledger-totals.mjs"; + +const execFileP = promisify(execFile); + +const RSVG = process.env.RSVG_BIN ?? "rsvg-convert"; +const QRENCODE = process.env.QRENCODE_BIN ?? "qrencode"; + +// Every card is drawn at the manifest's full `width`, but once a rail column is +// configured the RIGHT `rail.width` pixels of the frame belong to it. Text, +// rules and timeline nodes therefore lay out inside the CONTENT width, while the +// canvas stays full-width — the card ground is flat `pal.bg`, which is exactly +// what the rail wants behind it. +export const railWidth = (render) => render.rail?.width ?? 0; +export const contentWidth = (render) => render.width - railWidth(render); + +/** + * How wide a card's content may be. + * + * `hideRail` cards slide the rail off over their own dissolve, so they get the + * WHOLE frame. Everything else lays out inside the content width and leaves the + * rail column as ground. + */ +export const cardWidth = (card, render) => + card?.hideRail ? render.width : contentWidth(render); + +// Pango markup is XML-ish, so anything we interpolate has to be escaped first. +// Curly quotes and the ellipsis pass through fine; only these five matter. +function esc(s) { + return String(s) + .replace(/&/g, "&amp;") + .replace(/</g, "&lt;") + .replace(/>/g, "&gt;") + .replace(/"/g, "&quot;") + .replace(/'/g, "&apos;"); +} + +function span(text, { size, color, weight, family = "Fira Sans" }) { + const attrs = [`font_family="${family}"`, `size="${Math.round(size * 1024)}"`]; + if (color) attrs.push(`foreground="${color}"`); + if (weight) attrs.push(`weight="${weight}"`); + return `<span ${attrs.join(" ")}>${text}</span>`; +} + +// Each style returns Pango markup for the whole card body. Blank lines are real +// newlines in the markup — Pango honours them, which is how vertical rhythm is +// set without positioning each run separately. +function markupFor(card, pal) { + const H = (t, size = 62) => + span(esc(t), { size, color: pal.fg, weight: "bold" }); + const KICKER = (t) => + span(esc(t.toUpperCase()), { size: 24, color: pal.amber, weight: "bold" }); + const SUB = (t, size = 30) => span(esc(t), { size, color: pal.muted }); + + switch (card.style) { + case "title": + return [ + span(esc(card.heading), { size: 82, color: pal.fg, weight: "bold" }), + "", + span(esc(card.sub), { size: 38, color: pal.accent }), + "", + "", + SUB(card.foot, 24), + ].join("\n"); + + case "chapter": + return [ + card.kicker ? KICKER(card.kicker) : null, + card.kicker ? "" : null, + H(card.heading), + card.sub ? "" : null, + card.sub ? SUB(card.sub, 32) : null, + ] + .filter((l) => l !== null) + .join("\n"); + + case "bullets": { + const items = (card.bullets ?? []).flatMap((b) => [ + `${span("— ", { size: 30, color: pal.accent, weight: "bold" })}${span( + esc(b), + { size: 30, color: pal.fg }, + )}`, + "", + ]); + return [H(card.heading, 52), "", ...items].join("\n"); + } + + case "sources": + return [ + H(card.heading, 52), + "", + card.sub ? SUB(card.sub, 32) : null, + card.foot ? "" : null, + card.foot + ? card.foot + .split("\n") + .map((l) => SUB(l, 24)) + .join("\n") + : null, + ] + .filter((l) => l !== null) + .join("\n"); + + default: + return H(card.heading ?? card.id); + } +} + +// A chapter break that just states a heading is dead air — it stops the video to +// say something the next clip is about to say anyway. This draws the whole +// project arc instead, with the current step lit and everything before it +// filled, so the pause carries information: where we are and how far is left. +// +// Nodes come from the manifest's top-level `timelineNodes`; the card names its +// position with `step` (0-based). +async function renderTimelineCard(card, render, nodes, outDir) { + const pal = render.palette; + const { width, height } = render; + const outPath = path.join(outDir, "cards", `${card.id}.png`); + const dir = path.join(outDir, "cards"); + + const VW = cardWidth(card, render); + const x0 = 260; + const x1 = VW - 260; + const axisY = Math.round(height * 0.56); + const gap = (x1 - x0) / (nodes.length - 1); + const xs = nodes.map((_, i) => Math.round(x0 + i * gap)); + const cur = card.step; + + const args = ["-size", `${width}x${height}`, `xc:${pal.bg}`, "-strokewidth", "4"]; + + // Track: filled up to the current node, dim beyond it. + args.push( + "-stroke", pal.muted, "-fill", "none", + "-draw", `line ${xs[0]},${axisY} ${xs[xs.length - 1]},${axisY}`, + ); + if (cur > 0) { + args.push("-stroke", pal.accent, "-draw", `line ${xs[0]},${axisY} ${xs[cur]},${axisY}`); + } + + // Nodes: past and present filled, future hollow. The current one is larger and + // amber so the eye lands on it without needing a label to say "you are here". + nodes.forEach((_, i) => { + const r = i === cur ? 19 : 11; + const color = i === cur ? pal.amber : i < cur ? pal.accent : pal.bg; + args.push( + "-stroke", i <= cur ? (i === cur ? pal.amber : pal.accent) : pal.muted, + "-fill", color, + "-draw", `circle ${xs[i]},${axisY} ${xs[i] + r},${axisY}`, + ); + }); + + args.push("-stroke", "none"); + + // Heading, centred over the whole card. + const headMarkup = [ + span(esc((card.kicker ?? nodes[cur].label).toUpperCase()), { + size: 26, color: pal.amber, weight: "bold", + }), + "", + span(esc(card.heading ?? nodes[cur].title), { size: 62, color: pal.fg, weight: "bold" }), + card.sub ? "" : null, + card.sub ? span(esc(card.sub), { size: 30, color: pal.muted }) : null, + ] + .filter((l) => l !== null) + .join("\n"); + + // Heading, left-aligned on the same margin the other card styles use. + const headPath = path.join(dir, `${card.id}.head.pango`); + await writeFile(headPath, headMarkup, "utf8"); + args.push( + "(", "-size", `${VW - 460}x`, "-background", "none", + "-define", `pango:width=${VW - 460}`, + `pango:@${headPath}`, ")", + "-gravity", "NorthWest", + "-geometry", `+${Math.round(VW * 0.09) + 58}+${Math.round(height * 0.19)}`, + "-composite", + ); + + // Per-node date labels, centred under their dot. ImageMagick's + // `pango:alignment` define does not actually centre the text inside the box, + // so measure each rendered label and place it by hand instead of trusting it. + for (let i = 0; i < nodes.length; i += 1) { + const isCur = i === cur; + const labMarkup = span(esc(nodes[i].label), { + size: isCur ? 24 : 21, + color: isCur ? pal.fg : i < cur ? pal.muted : "#5c5570", + weight: isCur ? "bold" : "normal", + }); + const labPath = path.join(dir, `${card.id}.n${i}.pango`); + const labPng = path.join(dir, `${card.id}.n${i}.png`); + await writeFile(labPath, labMarkup, "utf8"); + await execFileP("magick", ["-background", "none", `pango:@${labPath}`, labPng]); + const { stdout } = await execFileP("magick", ["identify", "-format", "%w", labPng]); + const w = Number(stdout.trim()); + args.push(labPng, "-geometry", `+${xs[i] - Math.round(w / 2)}+${axisY + 44}`, "-composite"); + } + + args.push(outPath); + await execFileP("magick", args, { maxBuffer: 1 << 24 }); + return outPath; +} + +// Footer chrome, drawn once and overlaid on every clip: a track, a dot per +// section and its label. The *progress* along it is not baked in here — the fill +// bar and the amber marker are drawn by ffmpeg at encode time so they can slide +// between sections instead of cutting. See buildClipSegment in build-video.mjs. +// +// Returns the geometry the encoder needs to place those moving parts. +export async function renderFooterAssets(render, nodes, outDir) { + const pal = render.palette; + const { width } = render; + const FH = render.footerHeight ?? 92; + const dir = path.join(outDir, "cards"); + + // A cut whose clips are not a progression through time has nothing for a + // timeline to say, and a footer drawn anyway is chrome that has not earned its + // place. No nodes (or an explicit zero height) means no footer at all — the + // caller letterboxes against the header alone. + if (!nodes?.length || FH === 0) { + return { + footer: null, marker: null, bar: null, trackLen: 0, + footerHeight: 0, trackY: 0, xs: [], x0: 0, markerRadius: 0, + }; + } + + // The band itself stays full-frame so it reads as one strip running under the + // rail; only the TRACK is pulled in to the content width. + const x0 = 200; + const x1 = contentWidth(render) - 200; + const trackY = 26; + const gap = (x1 - x0) / (nodes.length - 1); + const xs = nodes.map((_, i) => Math.round(x0 + i * gap)); + + const footer = path.join(dir, "_footer.png"); + const args = [ + "-size", `${width}x${FH}`, `xc:${pal.bg}`, + "-strokewidth", "3", + "-stroke", "#3a3450", "-fill", "none", + "-draw", `line ${xs[0]},${trackY} ${xs[xs.length - 1]},${trackY}`, + "-stroke", "none", + ]; + for (const x of xs) { + args.push("-fill", "#4a4363", "-draw", `circle ${x},${trackY} ${x + 6},${trackY}`); + } + + // Two lines per node: what happened, then when. Both are measured and placed + // by hand — ImageMagick's `pango:alignment` define does not actually centre + // text inside its box. + for (let i = 0; i < nodes.length; i += 1) { + const lines = [ + { text: nodes[i].label, size: 18, color: pal.fg, dy: 18 }, + { text: nodes[i].date, size: 16, color: pal.muted, dy: 42 }, + ]; + for (const [k, ln] of lines.entries()) { + const pPath = path.join(dir, `_footer.n${i}.l${k}.pango`); + const pPng = path.join(dir, `_footer.n${i}.l${k}.png`); + await writeFile(pPath, span(esc(ln.text), { size: ln.size, color: ln.color }), "utf8"); + await execFileP("magick", ["-background", "none", `pango:@${pPath}`, pPng]); + const { stdout } = await execFileP("magick", ["identify", "-format", "%w", pPng]); + args.push( + pPng, + "-geometry", `+${xs[i] - Math.round(Number(stdout.trim()) / 2)}+${trackY + ln.dy}`, + "-composite", + ); + } + } + args.push(footer); + await execFileP("magick", args, { maxBuffer: 1 << 24 }); + + // The fill bar, as a strip to be TRANSLATED under a fixed crop rather than a + // drawbox whose width depends on `t`. drawbox has no time variable — its `t` + // is the box thickness — so the width expression the encoder used to build + // never evaluated and the bar was always full. Left half accent, right half + // transparent: sliding the crop window left across it grows the accent run. + const trackLen = xs[xs.length - 1] - x0; + const bar = path.join(dir, "_bar.png"); + await execFileP("magick", [ + "-size", `${trackLen * 2}x3`, "xc:none", + "-fill", pal.accent, "-stroke", "none", + "-draw", `rectangle 0,0 ${trackLen - 1},2`, + bar, + ]); + + // The travelling marker. + const marker = path.join(dir, "_marker.png"); + const r = 11; + await execFileP("magick", [ + "-size", `${r * 2 + 2}x${r * 2 + 2}`, "xc:none", + "-fill", pal.amber, "-stroke", "none", + "-draw", `circle ${r + 1},${r + 1} ${r * 2 + 1},${r + 1}`, + marker, + ]); + + return { footer, marker, bar, trackLen, footerHeight: FH, trackY, xs, x0, markerRadius: r }; +} + +// =========================================================================== +// The claim rail +// =========================================================================== +// A vertical ledger down the right edge that appends one row per claim as the +// video runs, with a live per-company tally beside it. The dates and the numbers +// are SPOKEN in the clips and shown only in the header citation line, so a viewer +// can hear "nearly ten" three times without ever seeing that the three refer to +// three different companies. The rail is what makes that visible. +// +// Every asset here is a STRIP, not a per-state still: one tall PNG whose window +// ffmpeg slides with an animated `crop`. That is the whole trick — swapping +// stills can only cut, but a crop can ease, and one input per moving part keeps +// the filtergraph small enough that ffmpeg does not deadlock on chained +// `-loop 1` inputs. See railFilterChain in build-video.mjs for the ramps. +// +// Assets are authored as SVG and rasterized with rsvg-convert rather than drawn +// with ImageMagick primitives: the strips need right-aligned columns, hairlines +// and ~500 individually-placed text runs, and one rsvg call beats four hundred +// `magick` invocations. Fonts inside the SVG resolve through fontconfig, so the +// family name has to MATCH the Pango cards ("Fira Sans"), not the font FILE that +// render.fontRegular points at. + +const RAIL_FONT = "Fira Sans"; + +// A tiny SVG text run. Everything is placed absolutely — no flow, no wrapping. +function svgText(x, y, text, o = {}) { + const a = [ + `x="${x}"`, `y="${y}"`, + `font-family="${o.family ?? RAIL_FONT}"`, + `font-size="${o.size ?? 15}"`, + `fill="${o.color}"`, + ]; + if (o.weight) a.push(`font-weight="${o.weight}"`); + if (o.anchor) a.push(`text-anchor="${o.anchor}"`); + if (o.ls) a.push(`letter-spacing="${o.ls}"`); + if (o.opacity != null) a.push(`opacity="${o.opacity}"`); + return `<text ${a.join(" ")}>${esc(text)}</text>`; +} + +const svgDoc = (w, h, body) => + `<svg xmlns="http://www.w3.org/2000/svg" width="${w}" height="${h}" ` + + `viewBox="0 0 ${w} ${h}">${body}</svg>`; + +// rsvg-convert is deterministic about output size in a way ImageMagick's RSVG +// delegate is not (its -density is ignored for sizing in some builds), so the +// pixel dimensions ffmpeg's crop arithmetic depends on are guaranteed here. +async function rasterize(svg, svgPath, pngPath, w, h) { + await writeFile(svgPath, svg, "utf8"); + await execFileP(RSVG, ["-w", String(w), "-h", String(h), "-o", pngPath, svgPath], { + maxBuffer: 1 << 26, + }); + return pngPath; +} + +// Truncate to a pixel budget. Fira Sans at these sizes averages ~0.50em per +// character; a hair conservative is right, because an overflowing row would run +// under the value column rather than wrap. +function fit(text, size, maxPx) { + const max = Math.max(4, Math.floor(maxPx / (size * 0.5))); + const t = String(text ?? ""); + return t.length <= max ? t : `${t.slice(0, max - 1).trimEnd()}…`; +} + +/** + * Every pixel measurement the rail needs, derived once so the asset builder and + * the filtergraph builder cannot disagree about a single one of them. + * + * The log window is deliberately sized to a WHOLE number of rows and butted + * against the bottom of the rail column: that is what lets the curtain (below) + * park exactly one window-height down and end up outside the rail entirely. + */ +export function railGeometry(render, nClaims) { + const rail = render.rail ?? {}; + const RW = rail.width ?? 500; + const HH = render.headerHeight ?? 56; + // reservedFooterHeight, NOT render.footerHeight. The two differ by 100px the + // moment the chart band is on (the band takes the footer's ground and 100 + // more), and deriving the rail's height from the smaller one ran the column + // a hundred pixels past the band's own top edge -- about two rows of the log + // window, drawn below the line everything else letterboxes to. + const FH = reservedFooterHeight(render); + const ROWH = rail.rowHeight ?? 38; + const PAD = rail.pad ?? 22; + const TALLYROWH = rail.tallyRowHeight ?? 40; + const nTracks = (rail.tracks ?? []).length; + + const RHGT = render.height - HH - FH; + // The tally's header belongs to the CHROME, not to the rolling block. Put it + // in the block and every roll scrolls a duplicate copy of it up through the + // window, which reads as noise rather than as a counter changing. + const TALLYHEAD_REL = rail.tallyTop ?? 76; + const TALLYTOP_REL = TALLYHEAD_REL + 26; + const TALLYH = nTracks * TALLYROWH; + + // The rolling CELL: the only part of a tally row that ever changes. The + // swatch and the company name sit left of it and belong to the chrome, so + // that a number moving does not drag its own label up the screen with it. + const CELLW = rail.cellWidth ?? 170; + const CELLX = RW - PAD - CELLW; + + // The roster he last enumerated. One line, and the point of it is that it + // stands still while the numbers above it move. + const ROSTERH = rail.rosterHeight ?? 26; + const ROSTERTOP_REL = TALLYTOP_REL + TALLYH + 14; + const ROSTERX = PAD + (rail.rosterLabelWidth ?? 62); + const ROSTERW = RW - PAD - ROSTERX; + + // The provenance tile, parked at the foot of the column: one QR per clip, + // bordered so it reads as a link rather than as decoration. + const QRSIZE = rail.qrSize ?? 132; + const TILEW = RW - 2 * PAD; + const TILEH = QRSIZE + 22; + const TILETOP_REL = RHGT - (rail.qrBottom ?? 12) - TILEH; + + const LOGTOP_REL = ROSTERTOP_REL + ROSTERH + 34; + // The log window is a WHOLE number of rows and stops short of the tile, so + // the curtain parks exactly one window-height down and the QR overlay (which + // comes after it in the chain) is never painted over. + const K = Math.max(1, Math.floor((TILETOP_REL - 12 - LOGTOP_REL) / ROWH)); + const LOGH = K * ROWH; + + return { + RW, PAD, ROWH, TALLYROWH, K, LOGH, TALLYH, RHGT, + CELLW, CELLX, + ROSTERH, ROSTERTOP_REL, ROSTERTOP: HH + ROSTERTOP_REL, ROSTERX, ROSTERW, + QRSIZE, TILEW, TILEH, TILETOP_REL, TILETOP: HH + TILETOP_REL, + VW: render.width - RW, + RX: render.width - RW, + RTOP: HH, + LOGTOP_REL, LOGTOP: HH + LOGTOP_REL, + TALLYTOP_REL, TALLYTOP: HH + TALLYTOP_REL, TALLYHEAD_REL, + nClaims, + logStripH: Math.max(LOGH, nClaims * ROWH), + }; +} + +/** + * How much of the frame the chrome band owns at the bottom. + * + * ONE definition, because three different renderers need it and they were + * already disagreeing: the closing chart drew its footnotes into the bottom + * 100px and the ledger scroll sized its window to `height - header - 100`, both + * of which are wrong the moment the band takes 200. The symptom is a card that + * looks finished in isolation and has its last two lines sitting under the + * chart in the cut. + */ +export function reservedFooterHeight(render) { + return render.chromeEngine === "hyperframes" + ? (render.chart?.height ?? 200) + : (render.footerHeight ?? 100); +} + +/** + * Every roll each tally cell will perform, as rows of one shared strip. + * + * --------------------------------------------------------------------------- + * Why the strip's LAYOUT carries the direction + * --------------------------------------------------------------------------- + * The whole block used to slide as one slab: when the coffee company's number + * changed, all four rows moved. Text that has not changed must not move, so + * each track now gets its own cell and its own y expression. + * + * A crop window can only walk a strip, and it walks in whichever direction its + * y expression takes it. So the DIRECTION of a roll is decided when the rows + * are laid out, not when the ramp is written: + * + * rise rows [old, new] crop walks DOWN the strip, content moves UP + * fall rows [new, old] crop walks UP the strip, content moves DOWN + * + * Between two transitions the crop steps instantly to the next pair's starting + * row. That step is invisible because the row it leaves and the row it arrives + * at hold IDENTICAL content — which is the reason every pair repeats the value + * it starts from rather than sharing a row with its neighbour. + * + * The delta chip rides along on both rows of the pair, and therefore stays on + * screen until the next change. That is deliberate: it reads as "how this + * number last moved", and blanking it at the step would make the invisible + * reposition visible. + * + * A repeated identical figure still rolls, upward. He said it again on a new + * date, and the `as of` line underneath is what changed. + * + * @returns {{lanes: Array<object>, rows: number}} one lane per track plus the + * roster lane, each with `rows` (the cells to draw) and `steps` (per claim, + * `null` or `{a, b}` — the row the roll starts on and the row it ends on). + */ +export function tallyTracks(ledger, tracks, opts = {}) { + const rosterLineOf = opts.rosterLine ?? (() => null); + const lanes = tracks.map((tr) => ({ + key: tr.key, track: tr, kind: "tally", + rows: [{ empty: true, track: tr }], + steps: new Array(ledger.length).fill(null), + cur: null, + })); + const byKey = Object.fromEntries(lanes.map((l) => [l.key, l])); + + const roster = { + key: "__roster", kind: "roster", + rows: [{ empty: true }], + steps: new Array(ledger.length).fill(null), + cur: null, + }; + + /** Lay a transition down as a pair of rows and record where it starts/ends. */ + const transition = (lane, from, to, rise) => { + const p = lane.rows.length; + if (rise) { + lane.rows.push(from, to); + return { a: p, b: p + 1 }; + } + lane.rows.push(to, from); + return { a: p + 1, b: p }; + }; + + ledger.forEach((c, i) => { + // ---- the four company cells ---- + const lane = byKey[c.scope ?? c.company]; + // UTTERED only. The tally says "the latest figure he has given", and a sum + // we performed is not one — showing 18 here while a card beside it says he + // never said eighteen makes the video contradict itself on screen. + if (lane && c.value != null && (!c.valueKind || c.valueKind === "uttered")) { + const prev = lane.cur; + const next = { + track: lane.track, + display: c.display ?? String(c.value), + value: c.value, + date: c.date, + population: c.population ?? null, + delta: prev ? Number((c.value - prev.value).toFixed(2)) : null, + }; + const from = prev ? { ...prev, delta: prev.delta } : { empty: true, track: lane.track }; + lane.steps[i] = transition(lane, from, next, !prev || c.value >= prev.value); + lane.cur = next; + } + + // ---- the roster ---- + // Only when the LINE changes. He enumerates the same two editors and one + // designer in October and again in December; rolling the line to arrive at + // the words it already said would animate the one thing that held still. + const line = Array.isArray(c.roles) && c.roles.length ? rosterLineOf(c.roles) : null; + if (line && line !== roster.cur?.line) { + const next = { line, date: c.date }; + const from = roster.cur ? { ...roster.cur } : { empty: true }; + roster.steps[i] = transition(roster, from, next, true); + roster.cur = next; + } + }); + + const all = [...lanes, roster]; + return { lanes: all, rows: Math.max(...all.map((l) => l.rows.length)) }; +} + +/** + * The colour of the population word under a figure. Never a new hue -- the + * chip is `pal.muted` text, so the dataviz gate does not have to be re-run. + */ +const POP_WORD = { + employees: "employees", + "full-time": "full time", + salaried: "salaried", + contractor: "contractors", + 1099: "1099", + people: "people", +}; + +/** A small solid triangle, because a font may not carry ▲ and tofu is worse. */ +function svgTri(x, y, up, color) { + const d = up + ? `M${x},${y + 7} L${x + 4.5},${y} L${x + 9},${y + 7} z` + : `M${x},${y} L${x + 4.5},${y + 7} L${x + 9},${y} z`; + return `<path d="${d}" fill="${color}"/>`; +} + +/** One rolling tally cell, drawn into a CELLW x TALLYROWH box at (x, y). */ +function tallyCellSvg(cell, x, y, g, pal, h) { + const right = x + g.CELLW; + const out = [`<rect x="${x}" y="${y}" width="${g.CELLW}" height="${h}" fill="${pal.bg}"/>`]; + if (cell.empty) { + out.push( + svgText(right, y + 24, "—", { size: 22, color: pal.muted, weight: "bold", anchor: "end", opacity: 0.5 }), + svgText(right, y + 37, "not yet stated", { size: 10.5, color: pal.muted, opacity: 0.6, anchor: "end" }), + ); + return out.join(""); + } + const colour = cell.track?.color ?? pal.fg; + out.push( + svgText(right, y + 24, cell.display, { size: 22, color: colour, weight: "bold", anchor: "end" }), + ); + if (cell.delta != null && cell.delta !== 0) { + const up = cell.delta > 0; + out.push( + svgTri(x, y + 11, up, colour), + svgText(x + 14, y + 22, `${up ? "+" : "−"}${Math.abs(cell.delta)}`, { + size: 13, color: colour, weight: "bold", + }), + ); + } + const chip = cell.population ? ` · ${POP_WORD[cell.population] ?? cell.population}` : ""; + out.push( + svgText(right, y + 37, `as of ${cell.date}${chip}`, { + size: 10.5, color: pal.muted, anchor: "end", opacity: 0.9, + }), + ); + return out.join(""); +} + +/** One roster line, drawn into a ROSTERW x ROSTERH box. */ +function rosterCellSvg(cell, x, y, g, pal) { + const out = [`<rect x="${x}" y="${y}" width="${g.ROSTERW}" height="${g.ROSTERH}" fill="${pal.bg}"/>`]; + out.push( + cell.empty + ? svgText(x, y + 18, "not yet enumerated", { size: 12.5, color: pal.muted, opacity: 0.55 }) + : svgText(x, y + 18, fit(cell.line, 13, g.ROSTERW), { size: 13, color: pal.muted }), + ); + return out.join(""); +} + +// --------------------------------------------------------------------------- +// The provenance tile +// --------------------------------------------------------------------------- +// A compilation asks the viewer to take the edit on trust. The QR is the +// antidote: it resolves to this clip's exact start in the archive's own viewer, +// so anyone can pull up the surrounding hour and check the cut is fair. +// +// It used to float over the bottom-right of the PICTURE, which is the one place +// in the frame the cut promises never to draw on. In the rail's foot it is a +// bordered tile that reads as a link, and it becomes one more strip walked by a +// crop -- one tile per segment, stepped instantaneously at the mid-dissolve, +// exactly like the other four. +// +// Two rules learned the hard way: it must be FULLY OPAQUE (a translucent QR +// will not scan) and it must keep its quiet zone (the white border is part of +// the symbol, not decoration). +export function qrUrlFor(entry, provenance) { + if (entry.type === "clip") { + // A mirror's LOCAL slug is not the id the site serves, so an explicit + // per-clip citeUrl always wins over the derived one. + return ( + entry.citeUrl ?? + `${provenance.siteOrigin}/?v=${encodeURIComponent( + `${entry.channel ?? provenance.channelSlug}/${entry.video}`, + )}&t=${Math.floor(entry.cite ?? entry.start)}` + ); + } + // A card is not a moment, so it gets the search rather than a timestamp. + // + // NOT `provenance.shareLink`. That link carries all 23 channel filters and is + // ~1.4k characters — a version-40 symbol, 177 modules inside a 132 px tile, + // which is roughly 0.7 px per module and unscannable. `qrLink` is the same + // query without the channel list (~200 chars, 63 modules, verified scannable + // at this size); the site origin is the fallback. A code nobody can scan is + // worse than a short one. + return provenance.qrLink ?? provenance.siteOrigin ?? ""; +} + +async function qrTileStrip(entries, provenance, render, g, outDir) { + const pal = render.palette; + const dir = path.join(outDir, "cards"); + const qrDir = path.join(outDir, "qr"); + const q = render.qr ?? {}; + const QR = g.QRSIZE; + const TH = g.TILEH; + + // One PNG per DISTINCT url, then a tile per segment referencing it. + const seen = new Map(); + const urls = entries.map((e) => qrUrlFor(e, provenance)); + for (const url of urls) { + if (seen.has(url)) continue; + const png = path.join(qrDir, `q${seen.size.toString().padStart(2, "0")}.png`); + await execFileP(QRENCODE, [ + "-o", png, "-s", String(q.scale ?? 4), "-m", String(q.quiet ?? 3), + "-l", q.ecc ?? "M", url, + ]); + // Nearest-neighbour to an exact box: a resampled QR blurs its module edges + // and stops scanning, and the geometry has to be known before this runs. + const sized = path.join(qrDir, `q${seen.size.toString().padStart(2, "0")}.${QR}.png`); + await execFileP("magick", [png, "-filter", "point", "-resize", `${QR}x${QR}!`, sized]); + seen.set(url, sized); + } + + const stripH = entries.length * TH; + const body = [`<rect x="0" y="0" width="${g.TILEW}" height="${stripH}" fill="${pal.bg}"/>`]; + const images = []; + entries.forEach((e, i) => { + const y = i * TH; + const isClip = e.type === "clip"; + body.push( + `<rect x="0.5" y="${y + 0.5}" width="${g.TILEW - 1}" height="${TH - 1}" fill="${pal.bg}" ` + + `stroke="${pal.accent}" stroke-width="1"/>`, + svgText(14, y + 32, "SCAN → JERALYZER", { + size: 12.5, color: pal.accent, weight: "bold", ls: 1.3, + }), + svgText(14, y + 58, isClip ? "this exact moment," : "the sweep this is cut from,", { + size: 13.5, color: pal.fg, + }), + svgText(14, y + 78, isClip ? "in the archive's own viewer" : "every claim, searchable", { + size: 13.5, color: pal.fg, + }), + svgText(14, y + 106, "the archive outlives the platform", { + size: 11.5, color: pal.muted, opacity: 0.8, + }), + ); + images.push({ png: seen.get(urls[i]), x: g.TILEW - QR - 11, y: y + 11 }); + }); + + const svgPath = path.join(dir, "_rail_qr.svg"); + const basePng = path.join(dir, "_rail_qr.base.png"); + await rasterize(svgDoc(g.TILEW, stripH, body.join("")), svgPath, basePng, g.TILEW, stripH); + + // The codes are composited rather than inlined: an <image href> in the SVG + // would be resampled by rsvg, and a resampled QR does not scan. + const out = path.join(dir, "_rail_qr.png"); + const args = [basePng]; + for (const im of images) args.push(im.png, "-geometry", `+${im.x}+${im.y}`, "-composite"); + args.push(out); + await execFileP("magick", args, { maxBuffer: 1 << 26 }); + return { path: out, tileH: TH, urls }; +} + +/** + * Build the rail strips. Returns their paths plus the geometry, so the caller + * never re-derives a size the pixels already committed to. + * + * `entries` and `provenance` are needed for the QR strip only; without them the + * tile is skipped and the rail is the four strips it always was. + */ +export async function renderRailAssets(render, ledger, outDir, entries = null, provenance = null) { + const pal = render.palette; + const rail = render.rail; + const tracks = rail.tracks ?? []; + const byKey = Object.fromEntries(tracks.map((t) => [t.key, t])); + const g = railGeometry(render, ledger.length); + const dir = path.join(outDir, "cards"); + const rule = rail.rule ?? "#2A322F"; + const P = (n) => path.join(dir, n); + + // ---- chrome: opaque, full rail column, never gated ------------------- + // It runs for the whole video rather than being switched on with enable=, + // which removes four expressions and four off-by-one opportunities. + // + // The tally SWATCH and LABEL live here. They never change, and while they + // rode in the rolling block a coffee figure changing dragged "The Quartering" + // up the screen with it — text moving for a reason that was not about it. + const chromeBody = [ + `<rect x="0" y="0" width="${g.RW}" height="${g.RHGT}" fill="${pal.bg}"/>`, + `<rect x="0" y="0" width="1" height="${g.RHGT}" fill="${rule}"/>`, + svgText(g.PAD, 32, "THE CLAIM LEDGER", { size: 15, color: pal.amber, weight: "bold", ls: 1.6 }), + svgText(g.PAD, 55, "every count he has given, as he gave it", { size: 14, color: pal.muted }), + `<rect x="${g.PAD}" y="70" width="${g.RW - 2 * g.PAD}" height="1" fill="${rule}"/>`, + svgText(g.PAD, g.TALLYHEAD_REL + 16, "LATEST FIGURE HE HAS GIVEN", { + size: 12, color: pal.muted, weight: "bold", ls: 1.4, + }), + ...tracks.map((tr, j) => { + const ry = g.TALLYTOP_REL + j * g.TALLYROWH; + return ( + `<rect x="${g.PAD}" y="${ry + 13}" width="10" height="10" fill="${tr.color}"/>` + + svgText(g.PAD + 20, ry + 22, fit(tr.label, 14, g.CELLX - g.PAD - 24), { + size: 14, color: pal.fg, + }) + ); + }), + `<rect x="${g.PAD}" y="${g.ROSTERTOP_REL - 8}" width="${g.RW - 2 * g.PAD}" height="1" fill="${rule}"/>`, + svgText(g.PAD, g.ROSTERTOP_REL + 19, "ROSTER", { + size: 11, color: pal.muted, weight: "bold", ls: 1.4, + }), + `<rect x="${g.PAD}" y="${g.LOGTOP_REL - 26}" width="${g.RW - 2 * g.PAD}" height="1" fill="${rule}"/>`, + svgText(g.PAD, g.LOGTOP_REL - 8, "IN THE ORDER STATED", { + size: 12, color: pal.muted, weight: "bold", ls: 1.4, + }), + ].join(""); + const chrome = await rasterize( + svgDoc(g.RW, g.RHGT, chromeBody), P("_rail_chrome.svg"), P("_rail_chrome.png"), g.RW, g.RHGT, + ); + + // ---- log strip: every claim, stacked, no padding --------------------- + const valX = g.RW - g.PAD; + const textX = g.PAD + 20; + const textBudget = valX - textX - 74; + const rows = ledger.map((c, i) => { + const y = i * g.ROWH; + // `scope` is the ADJUDICATED answer and `company` the undocumented guess it + // replaced. Falls back so a manifest with no adjudication yet still renders. + const tr = byKey[c.scope ?? c.company]; + // In `sourced` every row has a clip behind it, so `live` is always true + // there; `full` keeps the distinction because a stacked ledger card is a + // weaker citation than footage and must not look like one. + const live = !!c.entryId && !c.unsourced; + const ink = live ? pal.fg : pal.muted; + const dotFill = live ? (tr?.color ?? pal.accent) : "none"; + return [ + `<rect x="0" y="${y}" width="${g.RW}" height="${g.ROWH}" fill="${pal.bg}"/>`, + `<circle cx="${g.PAD + 5}" cy="${y + 19}" r="4.5" fill="${dotFill}" ` + + `stroke="${tr?.color ?? pal.muted}" stroke-width="1.5" opacity="${live ? 1 : 0.55}"/>`, + svgText(textX, y + 17, c.date, { size: 13.5, color: pal.muted, opacity: live ? 1 : 0.7 }), + svgText(valX, y + 19, c.display ?? "—", { + size: 18, color: live ? (tr?.color ?? pal.fg) : pal.muted, + weight: "bold", anchor: "end", opacity: live ? 1 : 0.65, + }), + svgText(textX, y + 33, fit(c.label ?? c.quote ?? "", 13, textBudget + 74), { + size: 13, color: ink, opacity: live ? 1 : 0.6, + }), + `<rect x="${g.PAD}" y="${y + g.ROWH - 1}" width="${g.RW - 2 * g.PAD}" height="1" fill="${rule}"/>`, + ].join(""); + }).join(""); + const log = await rasterize( + svgDoc(g.RW, g.logStripH, rows), P("_rail_log.svg"), P("_rail_log.png"), g.RW, g.logStripH, + ); + + // ---- curtain --------------------------------------------------------- + // With ONE log strip and the window parked at the top while the list is still + // filling, rows i+1…K-1 would show claims the video has not made yet. This + // opaque rectangle rides just below the last revealed row and, once the list + // is full, parks exactly one window-height down — outside the window. It is + // pal.bg precisely so that parking there is invisible; the QR tile overlays + // AFTER it, which is what stops the parked curtain covering the code. + const curtain = P("_rail_curtain.png"); + await execFileP("magick", ["-size", `${g.RW}x${g.LOGH}`, `xc:${pal.bg}`, curtain]); + + // ---- highlight ------------------------------------------------------- + const hl = await rasterize( + svgDoc(g.RW, g.ROWH, [ + `<rect x="0" y="0" width="${g.RW}" height="${g.ROWH}" fill="${pal.amber}" opacity="0.10"/>`, + `<rect x="${g.PAD - 12}" y="4" width="3" height="${g.ROWH - 8}" fill="${pal.amber}"/>`, + ].join("")), + P("_rail_hl.svg"), P("_rail_hl.png"), g.RW, g.ROWH, + ); + + // ---- tally strip: one COLUMN per lane, side by side ------------------- + // Four cells and the roster in one PNG: five crops at different x out of one + // input, rather than five inputs. Every column is as tall as the tallest, so + // a single strip height serves them all. + const { lanes, rows: nRows } = tallyTracks(ledger, tracks, { rosterLine }); + const cellH = g.TALLYROWH; + const laneX = []; + let sx = 0; + for (const lane of lanes) { + const w = lane.kind === "roster" ? g.ROSTERW : g.CELLW; + laneX.push({ x: sx, w }); + sx += w; + } + const stripW = sx; + const stripH = nRows * cellH; + const strip = [`<rect x="0" y="0" width="${stripW}" height="${stripH}" fill="${pal.bg}"/>`]; + lanes.forEach((lane, li) => { + const { x } = laneX[li]; + lane.rows.forEach((cell, ri) => { + strip.push( + lane.kind === "roster" + ? rosterCellSvg(cell, x, ri * cellH, g, pal) + : tallyCellSvg(cell, x, ri * cellH, g, pal, cellH), + ); + }); + }); + const tally = await rasterize( + svgDoc(stripW, stripH, strip.join("")), P("_rail_tally.svg"), P("_rail_tally.png"), stripW, stripH, + ); + + // ---- the provenance tile --------------------------------------------- + const qr = + entries && provenance && render.qr !== false + ? await qrTileStrip(entries, provenance, render, g, outDir) + : null; + + return { + chrome, log, curtain, hl, tally, qr, + geom: g, + lanes: lanes.map((lane, li) => ({ + key: lane.key, kind: lane.kind, steps: lane.steps, + x: laneX[li].x, w: laneX[li].w, cellH, + })), + }; +} + +// =========================================================================== +// Stacked ledger cards +// =========================================================================== +// The `full` cut's answer to the 28 claims the sweep found and no clip covers. +// +// Dimmed rail rows were the old answer, and they were confusing: a row with no +// audio behind it slid past with nothing to say for itself, and 28 of them read +// as padding rather than as evidence. So each one gets SCREEN TIME instead -- +// its date, its scope, its quote, his figure, and what that figure does to our +// running sum. Consecutive unclipped claims share a card and reveal in +// sequence, which is why 28 claims cost 14 cards and about 67 seconds. +// +// The reveal is the rail curtain's device: an opaque `pal.bg` rectangle walking +// down the card. Nothing fades, nothing moves; rows simply stop being covered. + +/** The reveal clock. One definition, because `scheduleClaims` pins to it. */ +export const LEDGER_LEAD = 0.9; +export const LEDGER_STEP = 1.3; +export const LEDGER_TAIL = 2.6; +export const ledgerRevealAt = (r) => LEDGER_LEAD + LEDGER_STEP * r; +export const ledgerSeconds = (n) => LEDGER_LEAD + LEDGER_STEP * n + LEDGER_TAIL - LEDGER_STEP; + +/** + * Why this claim is a line of text and not footage. + * + * "The upload is gone" and "we did not cut it" are different sentences, and + * saying the first about a live source is the kind of error that makes a whole + * compilation untrustworthy. So the words come from a `yt-dlp --simulate` + * probe, recorded in out/availability.json with the date it ran. + */ +export const SOURCE_TAG = { + ok: "not clipped", + deleted: "source deleted", + private: "source private", + "members-only": "members only", + restricted: "age-restricted", + "geo-blocked": "geo-blocked", + "no-cues": "no archived transcript", + maybe_missing: "source unreachable", +}; + +/** Greedy wrap to a pixel budget, at most `maxLines`, last line elided. */ +function wrapPx(text, size, maxPx, maxLines) { + const perChar = size * 0.5; + const cols = Math.max(8, Math.floor(maxPx / perChar)); + const words = String(text ?? "").split(/\s+/).filter(Boolean); + const lines = []; + let line = ""; + for (const w of words) { + if (line && (line + " " + w).length > cols) { + lines.push(line); + line = w; + if (lines.length === maxLines) break; + } else { + line = line ? line + " " + w : w; + } + } + if (lines.length < maxLines && line) lines.push(line); + if (lines.length === maxLines) { + const used = lines.join(" ").split(/\s+/).length; + if (used < words.length) lines[maxLines - 1] = fit(lines[maxLines - 1] + " …", size, maxPx); + } + return lines; +} + +/** + * One card carrying a run of consecutive unclipped claims. + * + * Returns the geometry the encoder needs to walk the curtain: where the rows + * start and how tall each one is. + */ +export async function renderLedgerCard(card, render, ledger, outDir, avail = null) { + const pal = render.palette; + const tracks = render.rail?.tracks ?? []; + const byKey = Object.fromEntries(tracks.map((t) => [t.key, t])); + const VW = cardWidth(card, render); + const H = render.height; + const dir = path.join(outDir, "cards"); + const rule = render.rail?.rule ?? "#2A322F"; + const RESERVED = reservedFooterHeight(render); + + const ids = card.claims ?? []; + const rows = ids.map((id) => ledger.find((c) => c.id === id)).filter(Boolean); + if (rows.length !== ids.length) { + const missing = ids.filter((id) => !ledger.some((c) => c.id === id)); + throw new Error(`ledger card ${card.id} names claims that are not in the ledger: ${missing.join(", ")}`); + } + + // The arithmetic is READ, never recomputed: one implementation of the walk, + // or the card and the chart band can disagree about the same sum. + const steps = new Map(ledgerTotals(ledger).steps.map((st) => [st.id, st])); + + const M = 96; + const ARITHW = 320; + const arithX = VW - M - ARITHW; + const quoteW = arithX - 70 - M; + + const body = [`<rect x="0" y="0" width="${VW}" height="${H}" fill="${pal.bg}"/>`]; + body.push( + svgText(M, 80, (card.kicker ?? "found in the sweep, not clipped here").toUpperCase(), { + size: 22, color: pal.amber, weight: "bold", ls: 1.4, + }), + svgText(M, 132, card.heading ?? "What the sweep found and this cut cannot show you", { + size: 40, color: pal.fg, weight: "bold", + }), + svgText(M, 168, card.sub ?? "his own words, and what they do to our running sum", { + size: 20, color: pal.muted, + }), + `<rect x="${M}" y="${192}" width="${VW - 2 * M}" height="1" fill="${rule}"/>`, + ); + + // Rows are a fixed height and the BLOCK is centred in what is left of the + // frame. Stretching two rows to fill 640px puts a hand's width of nothing + // between them; packing them at the top leaves the same gap in one lump at + // the bottom. Centring is the only arrangement that reads as deliberate. + const top = 214; + const available = H - RESERVED - top - 24; + const ROWH = Math.min(180, Math.floor(available / rows.length)); + const rowsTop = top + Math.floor((available - ROWH * rows.length) / 2); + + rows.forEach((c, r) => { + const y = rowsTop + r * ROWH; + const tr = byKey[c.scope ?? c.company]; + const st = steps.get(c.id); + const state = avail?.get(c.id) ?? null; + const tag = SOURCE_TAG[state] ?? "not clipped"; + + body.push( + svgText(M, y + 32, c.date, { size: 20, color: pal.muted }), + // The scope, as a bordered pill in its own colour. Which payroll a number + // is about is the whole argument, so it is never left to the ink alone. + `<rect x="${M + 148}" y="${y + 12}" width="${Math.max(120, (tr?.label?.length ?? 8) * 7.6 + 22)}" ` + + `height="26" rx="13" fill="none" stroke="${tr?.color ?? pal.muted}" stroke-width="1.2"/>`, + svgText(M + 159, y + 30, tr?.label ?? c.scope ?? "", { size: 14.5, color: tr?.color ?? pal.muted }), + svgText(M + 148 + Math.max(120, (tr?.label?.length ?? 8) * 7.6 + 22) + 16, y + 30, tag, { + size: 14.5, color: pal.muted, opacity: 0.85, + }), + svgText(arithX - 70, y + 40, c.display ?? "—", { + size: 34, color: tr?.color ?? pal.fg, weight: "bold", anchor: "end", + }), + ); + wrapPx(`“${c.quote ?? c.label ?? ""}”`, 25, quoteW, 2).forEach((line, li) => { + body.push(svgText(M, y + 76 + li * 33, line, { size: 25, color: pal.fg })); + }); + + // ---- the arithmetic column ---- + // Which layer this claim moved, lit; the others held, dimmed. The point is + // that the total on the right is OURS and is made of his own figures. + body.push( + svgText(arithX, y + 24, "OUR RUNNING SUM", { + size: 11, color: pal.muted, weight: "bold", ls: 1.3, + }), + ); + const basis = st?.impliedBasis ?? {}; + const companies = tracks.filter((t) => t.key !== "all"); + // Before any company has given a figure, our sum is not zero — it is + // undefined, and four dashes in a column say that far less clearly than + // one sentence does. + if (!companies.some((t) => basis[t.key])) { + body.push( + svgText(arithX, y + 52, "no company figure yet,", { size: 14, color: pal.muted }), + svgText(arithX, y + 74, "so our sum is not defined", { size: 14, color: pal.muted }), + ); + if (r < rows.length - 1) { + body.push( + `<rect x="${M}" y="${y + ROWH - 1}" width="${VW - 2 * M}" height="1" fill="${rule}" opacity="0.6"/>`, + ); + } + return; + } + companies.forEach((t, k) => { + const b = basis[t.key]; + const moved = (c.scope ?? c.company) === t.key; + const ty = y + 48 + k * 24; + body.push( + svgText(arithX, ty, fit(t.shortLabel ?? t.label, 13, ARITHW - 90), { + size: 13, color: moved ? t.color : pal.muted, opacity: moved ? 1 : 0.55, + }), + svgText(arithX + ARITHW, ty, b ? String(b.value) : "—", { + size: 17, color: moved ? t.color : pal.muted, weight: "bold", anchor: "end", + opacity: moved ? 1 : 0.55, + }), + ); + }); + const sy = y + 48 + companies.length * 24; + body.push( + `<rect x="${arithX}" y="${sy + 6}" width="${ARITHW}" height="1" fill="${rule}"/>`, + svgText(arithX, sy + 30, "IMPLIED", { size: 13, color: pal.fg, weight: "bold", ls: 1.2 }), + svgText(arithX + ARITHW, sy + 32, st?.implied == null ? "—" : String(st.implied), { + size: 22, color: pal.fg, weight: "bold", anchor: "end", + }), + ); + if (st?.impliedDelta) { + const up = st.impliedDelta > 0; + body.push( + svgTri(arithX + 84, sy + 22, up, pal.amber), + svgText(arithX + 98, sy + 30, `${up ? "+" : "−"}${Math.abs(st.impliedDelta)}`, { + size: 14, color: pal.amber, weight: "bold", + }), + ); + } + + if (r < rows.length - 1) { + body.push( + `<rect x="${M}" y="${y + ROWH - 1}" width="${VW - 2 * M}" height="1" fill="${rule}" opacity="0.6"/>`, + ); + } + }); + + const outPath = path.join(dir, `${card.id}.png`); + await rasterize(svgDoc(VW, H, body.join("")), path.join(dir, `${card.id}.svg`), outPath, VW, H); + return { path: outPath, width: VW, rowsTop, rowHeight: ROWH, rows: rows.length }; +} + +// =========================================================================== +// End sequence: the ledger scroll and the step chart +// =========================================================================== + +/** + * The whole ledger as one tall PNG for an animated crop to walk. + * + * --------------------------------------------------------------------------- + * One chronological line, a column per company + * --------------------------------------------------------------------------- + * It used to group by company: four blocks, each date-sorted inside itself. The + * cut plays in ONE chronology, and grouping at the end re-tells it in an order + * the viewer has not just watched — and it hides the only thing worth seeing + * here, which is that the four payrolls were being described in the same weeks. + * + * So: one date-ordered list, and the company is read from COLUMN POSITION. That + * makes colour the secondary encoding rather than the only one, which is the + * same rule the chart already runs under. + * + * The rail hides for this card (`hideRail`), so it is drawn at the FULL frame + * width rather than the content width. + * + * Returns the CONTENT HEIGHT because the scroll expression is written against + * it — crop clamps its own y, so an off-by-a-few degrades into a static last + * frame rather than an error, but only if the caller knows the real number. + */ +export async function renderScrollCard(card, render, ledger, outDir) { + const pal = render.palette; + const tracks = render.rail?.tracks ?? []; + const VW = cardWidth(card, render); + const dir = path.join(outDir, "cards"); + const rule = render.rail?.rule ?? "#2A322F"; + + const M = 96; + const ROWH = 42; + const body = []; + + // Columns. The value columns are right-aligned on their own gridline, so a + // number's horizontal position IS its company even before the colour reads. + const COLW = 152; + const dateX = M; + const colX = tracks.map((_, i) => M + 168 + i * COLW); + const popX = M + 168 + tracks.length * COLW + 24; + const labelX = popX + 132; + const labelW = VW - M - labelX; + + let y = 66; + body.push(svgText(M, y, card.heading ?? "THE COMPLETE LEDGER", { + size: 30, color: pal.fg, weight: "bold", ls: 1.5, + })); + y += 32; + body.push(svgText(M, y, card.sub ?? `${ledger.length} dated claims, in the order he made them`, { + size: 19, color: pal.muted, + })); + y += 44; + + // The column heads, which are the legend. No separate key: a company name + // over its own column of figures is the shortest legend there is. + body.push(`<rect x="${M}" y="${y - 4}" width="${VW - 2 * M}" height="1" fill="${rule}"/>`); + body.push(svgText(dateX, y + 26, "DATE", { size: 13, color: pal.muted, weight: "bold", ls: 1.3 })); + // No swatch beside the head: the head is already IN the track's colour, and + // the column position is the primary encoding either way. A swatch would only + // land on top of the words, since a right-anchored run cannot be measured + // here to leave room for one. + tracks.forEach((tr, i) => { + body.push( + svgText(colX[i], y + 26, fit(tr.shortLabel ?? tr.label, 13, COLW - 12), { + size: 13, color: tr.color, weight: "bold", anchor: "end", ls: 0.6, + }), + ); + }); + body.push( + svgText(popX, y + 26, "AS WHAT", { size: 13, color: pal.muted, weight: "bold", ls: 1.3 }), + svgText(labelX, y + 26, "WHAT HE SAID", { size: 13, color: pal.muted, weight: "bold", ls: 1.3 }), + ); + y += 40; + body.push(`<rect x="${M}" y="${y}" width="${VW - 2 * M}" height="1" fill="${rule}"/>`); + y += 8; + + const byKey = Object.fromEntries(tracks.map((t, i) => [t.key, i])); + const rows = [...ledger].sort((a, b) => dateKey(a.date).localeCompare(dateKey(b.date))); + for (const c of rows) { + const live = !!c.entryId && !c.unsourced; + const i = byKey[c.scope ?? c.company]; + const tr = tracks[i]; + body.push( + svgText(dateX, y + 26, c.date, { size: 18, color: pal.muted, opacity: live ? 1 : 0.7 }), + ); + if (tr) { + body.push( + svgText(colX[i], y + 26, c.display ?? "—", { + size: 21, color: live ? tr.color : pal.muted, weight: "bold", anchor: "end", + opacity: live ? 1 : 0.6, + }), + ); + } + body.push( + svgText(popX, y + 26, POP_WORD[c.population] ?? c.population ?? "", { + size: 15, color: pal.muted, opacity: live ? 0.9 : 0.6, + }), + svgText(labelX, y + 26, fit(c.label ?? "", 18, labelW), { + size: 18, color: live ? pal.fg : pal.muted, opacity: live ? 1 : 0.6, + }), + `<rect x="${M}" y="${y + ROWH - 1}" width="${VW - 2 * M}" height="1" fill="${rule}" opacity="0.5"/>`, + ); + y += ROWH; + } + y += 60; + + const contentHeight = y; + const outPath = path.join(dir, `${card.id}.png`); + await rasterize( + svgDoc(VW, contentHeight, `<rect x="0" y="0" width="${VW}" height="${contentHeight}" fill="${pal.bg}"/>${body.join("")}`), + path.join(dir, `${card.id}.svg`), outPath, VW, contentHeight, + ); + return { path: outPath, contentHeight, width: VW }; +} + +/** + * The four-series step chart, over the claims flagged `plotted`. + * + * COLOUR IS NOT THE ONLY ENCODING here, and that is a hard requirement rather + * than a flourish: no four-colour categorical palette clears the data-viz + * all-pairs CVD gate (three slots is the documented ceiling), so each series + * also carries a distinct dash pattern and a direct end-of-line label. The four + * hues themselves are the published artifact's, re-validated against this + * video's darker ground (#0F1312) on the adjacent pairlist — the pairlist for + * line charts — where all five checks pass. + */ +export async function renderChartCard(card, render, ledger, outDir) { + const pal = render.palette; + const tracks = render.rail?.tracks ?? []; + const VW = cardWidth(card, render); + const H = render.height; + const dir = path.join(outDir, "cards"); + const rule = render.rail?.rule ?? "#2A322F"; + + // The series come from ledger-totals, not from the legacy `plotted` flag. + // `plotted` was set under the OLD reading, in which a sum we performed sat in + // the same series as a figure he uttered. Drawing from it now would put 18 and + // 20 back on his line, after the whole point of the adjudication was to take + // them off it. + let totals = null; + try { + totals = ledgerTotals(ledger); + } catch { + // An unadjudicated ledger still renders -- as the three company series only, + // because the two totals are exactly what it cannot be trusted about. + totals = null; + } + const pts = ledger.filter((c) => c.value != null && (c.scope ?? c.company) !== "all"); + const yr = (d) => { + const [Y, M2, D2] = d.split("-").map(Number); + return Y + (M2 - 1) / 12 + (D2 - 1) / 365; + }; + const X0 = yr("2020-01-01"), X1 = yr("2026-12-31"); + const YMAX = + Math.max(21, ...pts.map((p) => p.value), ...(totals?.series.implied ?? []).map((p) => p.value)) + 1; + + const RESERVED = reservedFooterHeight(render); + const box = { l: 150, r: 300, t: 190, b: 130 + RESERVED }; + const plotW = VW - box.l - box.r; + const plotH = H - box.t - box.b; + const px = (v) => box.l + ((v - X0) / (X1 - X0)) * plotW; + const py = (v) => H - box.b - (v / YMAX) * plotH; + + const body = [`<rect x="0" y="0" width="${VW}" height="${H}" fill="${pal.bg}"/>`]; + body.push( + svgText(box.l, 78, "WHAT HE SAID, AND WHAT IT ADDS UP TO", { + size: 34, color: pal.fg, weight: "bold", ls: 1.5, + }), + svgText(box.l, 112, "every figure he utters, against the company he was talking about", { + size: 20, color: pal.muted, + }), + svgText(box.l, 146, "the heavy line is ours — his own per-company claims, added up", { + size: 18, color: pal.amber, + }), + ); + + // grid + axes + for (let gv = 0; gv <= YMAX - 1; gv += 5) { + body.push( + `<rect x="${box.l}" y="${py(gv)}" width="${plotW}" height="1" fill="${rule}"/>`, + svgText(box.l - 16, py(gv) + 6, String(gv), { size: 17, color: pal.muted, anchor: "end" }), + ); + } + body.push(svgText(box.l - 16, py(YMAX - 1) - 22, "PEOPLE", { + size: 13, color: pal.muted, weight: "bold", anchor: "end", ls: 1.2, + })); + for (let Y = 2020; Y <= 2026; Y += 1) { + const x = px(yr(`${Y}-01-01`)); + body.push( + `<rect x="${x}" y="${box.t}" width="1" height="${py(0) - box.t}" fill="${rule}" opacity="0.7"/>`, + svgText(x, py(0) + 30, String(Y), { size: 17, color: pal.muted, anchor: "middle" }), + ); + } + body.push(`<rect x="${box.l}" y="${py(0)}" width="${plotW}" height="2" fill="${pal.muted}"/>`); + + // One step path per series, plus a dot per claim and a direct end label. + // + // FIVE series, not four. The three companies are his, drawn as before. The + // fourth is what he says the WHOLE payroll is -- only ever a figure he utters + // as one number. The fifth is what his own per-company claims add up to, and + // it is ours: a heavy neutral step, because an aggregate is not a categorical + // peer of the things it aggregates and must not consume a palette slot. + const DASH = ["", "12 6", "3 7", "18 5 4 5"]; + const labels = []; + const drawn = []; + tracks.forEach((tr, ti) => { + if (tr.key === "all") return; + drawn.push({ + tr, dash: DASH[ti % 4], width: 3.5, + pts: pts.filter((c) => (c.scope ?? c.company) === tr.key) + .slice().sort((a, b) => a.date.localeCompare(b.date)) + .map((c) => ({ date: c.date, value: c.value, display: c.display, hedged: c.hedged })), + }); + }); + if (totals) { + const allTrack = tracks.find((t) => t.key === "all"); + drawn.push({ + tr: { key: "stated", color: allTrack?.color ?? pal.accent, label: "stated total" }, + dash: "18 5 4 5", width: 3.5, dots: true, + pts: totals.series.stated.map((p) => ({ date: p.date, value: p.value, display: String(p.value) })), + }); + drawn.push({ + tr: { key: "implied", color: pal.fg, label: "implied — our sum" }, + dash: "", width: 6, dots: false, + pts: totals.series.implied.map((p) => ({ date: p.date, value: p.value, display: String(p.value) })), + }); + } + + for (const sr of drawn) { + const mine = sr.pts; + if (!mine.length) continue; + // A step, not a line: the figure he gave holds until he gives another one, + // so the segment between two claims must be flat and the change vertical. + let d = ""; + let prevY = null; + for (const [i, c] of mine.entries()) { + const x = px(yr(c.date)); + const yv = py(c.value); + d += i === 0 + ? `M ${x.toFixed(1)} ${yv.toFixed(1)}` + : ` L ${x.toFixed(1)} ${prevY.toFixed(1)} L ${x.toFixed(1)} ${yv.toFixed(1)}`; + prevY = yv; + } + const last = mine[mine.length - 1]; + const lastY = py(last.value); + d += ` L ${(box.l + plotW).toFixed(1)} ${lastY.toFixed(1)}`; + body.push( + `<path d="${d}" fill="none" stroke="${sr.tr.color}" stroke-width="${sr.width}" ` + + `stroke-linejoin="round"${sr.dash ? ` stroke-dasharray="${sr.dash}"` : ""}/>`, + ); + if (sr.dots !== false) { + for (const c of mine) { + body.push( + `<circle cx="${px(yr(c.date)).toFixed(1)}" cy="${py(c.value).toFixed(1)}" r="${c.hedged ? 5 : 6}" ` + + `fill="${c.hedged ? pal.bg : sr.tr.color}" stroke="${sr.tr.color}" stroke-width="2.5"/>`, + ); + } + } + labels.push({ tr: sr.tr, last, lineY: lastY, y: lastY }); + } + + // The closing hold annotates the gap it has just finished drawing. + if (card.hold && totals && totals.final.stated != null && totals.final.implied != null) { + const xR = box.l + plotW; + const yS = py(totals.final.stated); + const yI = py(totals.final.implied); + body.push( + `<rect x="${(xR - 190).toFixed(1)}" y="${Math.min(yI, yS).toFixed(1)}" width="170" ` + + `height="${Math.abs(yS - yI).toFixed(1)}" fill="${tracks.find((t) => t.key === "all")?.color ?? pal.accent}" opacity="0.12"/>`, + `<path d="M ${(xR - 105).toFixed(1)} ${yI.toFixed(1)} L ${(xR - 105).toFixed(1)} ${yS.toFixed(1)}" ` + + `stroke="${pal.amber}" stroke-width="2"/>`, + svgText(xR - 96, (yI + yS) / 2 - 4, `gap ${Math.round(totals.final.implied - totals.final.stated)}`, { + size: 22, color: pal.amber, weight: "bold", + }), + svgText(xR - 96, (yI + yS) / 2 + 22, "between his last total and our sum", { + size: 14, color: pal.muted, + }), + ); + } + + // Three of the four series end within a couple of people of each other, so + // their direct labels land on top of one another. Push them apart and elbow a + // leader line back to the value each one actually belongs to — direct labels + // are the secondary encoding that lets a four-colour palette be legible at + // all, so an unreadable stack would defeat the point of having them. + const LBLH = 48; + labels.sort((a, b) => a.y - b.y); + for (let i = 1; i < labels.length; i += 1) { + labels[i].y = Math.max(labels[i].y, labels[i - 1].y + LBLH); + } + const overshoot = labels.length ? labels[labels.length - 1].y - (H - box.b - 10) : 0; + if (overshoot > 0) for (const l of labels) l.y -= overshoot; + for (const l of labels) { + const lx = box.l + plotW; + if (Math.abs(l.y - l.lineY) > 2) { + body.push( + `<path d="M ${lx} ${l.lineY.toFixed(1)} L ${lx + 9} ${l.lineY.toFixed(1)} ` + + `L ${lx + 9} ${l.y.toFixed(1)} L ${lx + 14} ${l.y.toFixed(1)}" fill="none" ` + + `stroke="${l.tr.color}" stroke-width="1.5" opacity="0.75"/>`, + ); + } + body.push( + svgText(lx + 20, l.y + 2, l.tr.label, { size: 18, color: l.tr.color, weight: "bold" }), + svgText(lx + 20, l.y + 22, `last stated ${l.last.display}`, { size: 14, color: pal.muted }), + ); + } + + body.push( + svgText(box.l, H - RESERVED - 56, "hollow dot = a hedge word (“nearly ten”, “a handful”), not a figure", { + size: 16, color: pal.muted, + }), + svgText(box.l, H - RESERVED - 30, "each series is dashed as well as coloured — the shapes carry the reading on their own; " + + "sums and midpoints are ours and are never drawn as his", { + size: 16, color: pal.muted, + }), + ); + + const outPath = path.join(dir, `${card.id}.png`); + await rasterize(svgDoc(VW, H, body.join("")), path.join(dir, `${card.id}.svg`), outPath, VW, H); + return { + path: outPath, + plotX: box.l, plotY: box.t, plotW, plotH: py(0) - box.t + 2, + }; +} + +export async function renderCard(card, render, outDir, nodes) { + if (card.style === "timeline") { + if (!nodes?.length) throw new Error(`card ${card.id} is style:timeline but no timelineNodes given`); + return renderTimelineCard(card, render, nodes, outDir); + } + return renderPlainCard(card, render, outDir); +} + +async function renderPlainCard(card, render, outDir) { + const pal = render.palette; + const { width, height } = render; + const VW = cardWidth(card, render); + const textWidth = Math.round(VW * 0.74); + const outPath = path.join(outDir, "cards", `${card.id}.png`); + + // Pango reads its markup from a file to keep it clear of shell/argv quoting. + const markupPath = path.join(outDir, "cards", `${card.id}.pango`); + await writeFile(markupPath, markupFor(card, pal), "utf8"); + + // One magick invocation: solid ground, an accent rule down the left margin, + // then the Pango block composited over it. The rule is what keeps the cards + // recognisably one family across styles. + const barX = Math.round(VW * 0.09); + const barTop = Math.round(height * 0.28); + const barBottom = Math.round(height * 0.72); + + const args = [ + "-size", `${width}x${height}`, + `xc:${pal.bg}`, + "-fill", pal.accent, + "-draw", `rectangle ${barX},${barTop} ${barX + 6},${barBottom}`, + "(", + // `-size` is still set to the full frame from the canvas above, and the + // pango delegate honours it — leaving it alone renders the text into a + // 1920x1080 box, which pins the block to the top and wraps at the frame + // edge instead of the margin. Reset it to the text column, height auto. + "-size", `${textWidth}x`, + "-background", "none", + "-define", `pango:width=${textWidth}`, + "-define", "pango:alignment=left", + "-define", "pango:wrap=word", + `pango:@${markupPath}`, + ")", + "-gravity", "West", + "-geometry", `+${barX + 58}+0`, + "-composite", + outPath, + ]; + + await execFileP("magick", args, { maxBuffer: 1 << 24 }); + return outPath; +} + +async function main() { + const argv = process.argv.slice(2); + const manifestPath = argv.find((a) => !a.startsWith("--")); + if (!manifestPath) { + console.error("usage: render-cards.mjs <manifest.json> [--out <dir>] [--only <id>]"); + process.exit(2); + } + const flag = (name) => { + const i = argv.indexOf(name); + return i >= 0 ? argv[i + 1] : undefined; + }; + + const manifest = JSON.parse(await readFile(manifestPath, "utf8")); + const outDir = flag("--out") ?? path.join(path.dirname(path.resolve(manifestPath)), "out"); + const only = flag("--only"); + + await mkdir(path.join(outDir, "cards"), { recursive: true }); + + const cards = manifest.timeline.filter( + (e) => e.type === "card" && (!only || e.id === only), + ); + for (const card of cards) { + const p = await renderCard(card, manifest.render, outDir, manifest.timelineNodes); + console.log(`card ${card.id} -> ${p}`); + } + console.log(`${cards.length} card(s) rendered`); +} + +if (import.meta.url === `file://${process.argv[1]}`) { + main().catch((err) => { + console.error(err); + process.exit(1); + }); +} diff --git a/scripts/report-to-video/resolve-windows.mjs b/scripts/report-to-video/resolve-windows.mjs @@ -0,0 +1,241 @@ +#!/usr/bin/env node +// resolve-windows.mjs — widen a manifest's clip windows to whole sentences. +// +// A manifest window starts life as the cue span covering a quote, and a cue +// boundary is a bad place to cut: ASR breaks cues where the caption line wrapped, +// which is routinely mid-sentence and often mid-word. Cutting there drops the +// lead-in that makes a quote make sense, and clips audibly start and stop in the +// middle of speech. +// +// This walks outward from the cue span to the nearest sentence boundary in the +// transcript — a cue whose text ends in . ? or ! — so the clip carries the whole +// thought. Word-level alignment is a separate, audio-side problem: build-video.mjs +// snaps the actual cut to a silence (see --fetch-pad / snapping there). +// +// Expansion is capped so a run-on passage can't drag a clip out to a minute. +// +// In the app: not used. On the CLI: +// node scripts/report-to-video/resolve-windows.mjs <manifest.json> [--write] +// +// Options: +// --write Rewrite the manifest in place (default: dry run, print a table) +// --max-lead <s> Max seconds to expand backwards (default 9) +// --max-tail <s> Max seconds to expand forwards (default 12) +// --site-origin <url> Archive to read cues from when there is no local corpus +// (defaults to the manifest's provenance.siteOrigin) +// --resolve-site-ids On a published-id miss, find the record by scanning the +// channel's shards. Slow; see cues.mjs. +// --cue-source <which> auto (default) | local | http. The two can disagree +// once a corpus moves past its last publish — see cues.mjs. +// +// A clip entry may set `lockStart` / `lockEnd` to pin that edge exactly. + +import { readFile, writeFile } from "node:fs/promises"; + +import { createCueSource, siteOriginFromManifest } from "./cues.mjs"; + + +const ENDS_SENTENCE = /[.!?]["'”’)\]]*\s*$/; + +// A cue that is only "[music]" or "[ __ ]" (the profanity bleep) carries no +// sentence signal; treat it as transparent so expansion walks past it. +const IS_FILLER = /^\s*(\[[^\]]*\]|>>|♪|—|-)*\s*$/; + +// The manifest stores times rounded to 2 dp, so a value read back from it can sit +// a hair BELOW the cue end it came from. Without a tolerance the end lookup then +// lands on the previous cue, the forward search runs on to the next sentence, and +// the clip grows a little every time this is run — it has to be a fixed point. +const EPS = 0.02; + + +function indexAt(cues, t, which) { + // First cue whose span contains t, else the nearest one on the right side. + let idx = cues.findIndex((c) => c.end > t); + if (idx < 0) idx = cues.length - 1; + if (which === "end") { + let j = cues.findIndex((c) => c.end >= t - EPS); + if (j < 0) j = cues.length - 1; + idx = j; + } + return idx; +} + +export function widen(cues, start, end, { maxLead = 8, maxTail = 12 } = {}) { + const isBoundary = (c) => ENDS_SENTENCE.test(c.text) && !IS_FILLER.test(c.text); + const i0 = indexAt(cues, start, "start"); + const i1 = indexAt(cues, end, "end"); + + // START: the latest cue that OPENS a sentence (i.e. its predecessor closes + // one) at or before the quote, within the lead budget. Finding no such cue + // means every candidate lead-in is a sentence fragment, so take none at all — + // a fragment is the irrelevant context we are trying to avoid, not context. + let si = null; + for (let i = i0; i > 0; i -= 1) { + if (start - cues[i].start > maxLead) break; + if (isBoundary(cues[i - 1])) { + si = i; + break; + } + } + if (si === null) si = i0; + + // END: the first cue that CLOSES a sentence at or after the quote. Never + // clamp to a budget here — stopping partway through a sentence is exactly the + // mid-thought ending this is meant to remove, so the budget only decides how + // far to look, and failing to find one falls back to the original cue end. + let ei = null; + for (let j = i1; j < cues.length; j += 1) { + if (cues[j].end - end > maxTail) break; + if (isBoundary(cues[j])) { + ei = j; + break; + } + } + if (ei === null) ei = i1; + + return { + start: cues[si].start, + end: cues[ei].end, + leadCues: i0 - si, + tailCues: ei - i1, + }; +} + +async function main() { + const argv = process.argv.slice(2); + const manifestPath = argv.find((a) => !a.startsWith("--")); + if (!manifestPath) { + console.error("usage: resolve-windows.mjs <manifest.json> [--write]"); + process.exit(2); + } + const num = (name, dflt) => { + const i = argv.indexOf(name); + return i >= 0 ? Number(argv[i + 1]) : dflt; + }; + // Lead is where the context lives — it is the run-up that makes a quote make + // sense. Tail only needs to finish the sentence, so it gets a smaller budget. + const opts = { maxLead: num("--max-lead", 8), maxTail: num("--max-tail", 12) }; + + const manifest = JSON.parse(await readFile(manifestPath, "utf8")); + const slug = manifest.provenance.channelSlug; + const cache = new Map(); + + // Cues come from a local corpus when there is one, and from the published + // archive the manifest was built against when there is not — so this runs in a + // clone with no `transcripts/` at all. See cues.mjs. + const flag = (name) => { + const i = argv.indexOf(name); + return i >= 0 ? argv[i + 1] : undefined; + }; + const cues = createCueSource({ + siteOrigin: flag("--site-origin") ?? process.env.SITE_ORIGIN ?? siteOriginFromManifest(manifest), + resolveSiteIds: argv.includes("--resolve-site-ids"), + prefer: flag("--cue-source") ?? "auto", + log: (m) => console.error(` · ${m}`), + }); + const loadCues = (videoId, channelSlug, hints) => + cues.load(channelSlug, videoId, hints).then((r) => r.cues); + + let changed = 0; + for (const e of manifest.timeline) { + if (e.type !== "clip") continue; + // A compilation can span several archived channels (the same streamer's VODs + // are mirrored across more than one), so a clip may name its own. Key the + // cache by channel too — the same id under a different slug is a different file. + // An author can trim a clip to land mid-cue on purpose — a cue often carries + // a whole paragraph, and cutting a quote short is an editorial decision. + // Widening would undo exactly that, so `lock` opts the clip out. + if (e.lock) { + console.log(`${e.id.padEnd(4)} ${e.video.padEnd(12)} locked, left at ${e.start.toFixed(1)}–${e.end.toFixed(1)}`); + continue; + } + const chan = e.channel ?? slug; + const key = `${chan}/${e.video}`; + if (!cache.has(key)) { + cache.set( + key, + await loadCues(e.video, chan, { siteChannel: e.siteChannel, siteVideo: e.siteVideo }), + ); + } + const cues = cache.get(key); + + const before = { start: e.start, end: e.end }; + const w = widen(cues, e.start, e.end, opts); + + // `lockStart` / `lockEnd` pin an edge to exactly what the author wrote. The + // escape hatch exists because sentence detection is only as good as the ASR's + // punctuation, and some uploads have none at all — and because an utterance's + // real trailing pause does not always line up with its last cue's end. + if (e.lockStart) w.start = before.start; + if (e.lockEnd) w.end = before.end; + const dLead = (before.start - w.start).toFixed(1); + const dTail = (w.end - before.end).toFixed(1); + const dur = (w.end - w.start).toFixed(1); + + // Ignore sub-frame drift so a re-run on an already-resolved manifest is a + // genuine no-op rather than a rewrite that nudges every window. + const moved = + Math.abs(w.start - before.start) > 0.05 || Math.abs(w.end - before.end) > 0.05; + if (moved) changed += 1; + console.log( + `${e.id.padEnd(4)} ${e.video.padEnd(12)} ` + + `${before.start.toFixed(1)}–${before.end.toFixed(1)} -> ` + + `${w.start.toFixed(1)}–${w.end.toFixed(1)} (+${dLead}s lead, +${dTail}s tail, ${dur}s)`, + ); + + if (moved) { + e.start = Number(w.start.toFixed(2)); + e.end = Number(w.end.toFixed(2)); + } + } + + // De-overlap clips that come from the SAME video. Widening is per-clip and + // blind to its neighbours, so a tail that finds no sentence boundary runs to + // the budget and can swallow the next clip's material — which plays as the + // same footage twice. (Real case: a 2024 upload whose ASR carries no + // punctuation at all in that stretch, so nothing stopped the search.) + // The later clip's start is the deliberate one, so trim the earlier clip's tail. + const byVideo = new Map(); + for (const e of manifest.timeline) { + if (e.type !== "clip") continue; + if (!byVideo.has(e.video)) byVideo.set(e.video, []); + byVideo.get(e.video).push(e); + } + for (const [video, list] of byVideo) { + if (list.length < 2) continue; + list.sort((a, b) => a.start - b.start); + for (let i = 0; i < list.length - 1; i += 1) { + const a = list[i]; + const b = list[i + 1]; + if (a.end <= b.start) continue; + const overlap = a.end - b.start; + if (a.lockEnd) { + console.log(` ⚠ ${a.id} overlaps ${b.id} by ${overlap.toFixed(1)}s but has lockEnd — not trimmed`); + continue; + } + a.end = Number(b.start.toFixed(2)); + changed += 1; + console.log( + ` de-overlap ${video}: ${a.id} trimmed ${overlap.toFixed(1)}s off its tail ` + + `(it ran into ${b.id})`, + ); + if (a.end - a.start < 3) { + console.log(` ⚠ ${a.id} is now only ${(a.end - a.start).toFixed(1)}s — check it`); + } + } + } + + if (argv.includes("--write")) { + await writeFile(manifestPath, JSON.stringify(manifest, null, 2) + "\n", "utf8"); + console.log(`\nwrote ${manifestPath} (${changed} window(s) changed)`); + } else { + console.log(`\ndry run — ${changed} window(s) would change; pass --write to apply`); + } +} + +if (import.meta.url === `file://${process.argv[1]}`) { + main().catch((err) => { + console.error(err); + process.exit(1); + }); +} diff --git a/scripts/report-to-video/verify-build.mjs b/scripts/report-to-video/verify-build.mjs @@ -0,0 +1,110 @@ +#!/usr/bin/env node +// verify-build.mjs — is the file that came out the file that was asked for? +// +// A build can exit 0 and still be wrong in ways nothing else notices: a concat +// that produced a zero-length file, a chapter pass that silently dropped +// markers, a timeline that lost a clip because --continue-on-error let it. Each +// of those looks like success at the terminal and like a finished video in a +// directory listing. +// +// So the last step of a build measures the deliverable and compares it to the +// manifest. Cheap (one ffprobe) and the only thing that closes the loop. +// +// node scripts/report-to-video/verify-build.mjs <manifest.json> [--out <dir>] +// [--variant sourced|full] [--json] + +import { execFile } from "node:child_process"; +import { promisify } from "node:util"; +import { readFile, stat } from "node:fs/promises"; +import path from "node:path"; + +import { selectVariant, variantPaths } from "./build-video.mjs"; + +const execFileP = promisify(execFile); +const FFPROBE = process.env.FFPROBE_BIN ?? "ffprobe"; + +export async function verifyBuild(manifestPath, { outDir, variant = "sourced" } = {}) { + // The SAME filter the build ran. Verifying the whole manifest against one + // variant's file would report a missing chapter for every entry the other cut + // carries -- i.e. it would be red exactly when the build was right. + const manifest = selectVariant( + JSON.parse(await readFile(manifestPath, "utf8")), + variant, + ); + const root = outDir ?? path.join(path.dirname(path.resolve(manifestPath)), "out"); + const file = variantPaths(root, manifest.slug, variant).final; + const problems = []; + + const st = await stat(file).catch(() => null); + if (!st) return { ok: false, file, problems: [`${file} does not exist`] }; + if (st.size < 1024) problems.push(`${file} is ${st.size} bytes`); + + const { stdout } = await execFileP(FFPROBE, [ + "-v", "error", + "-show_entries", "format=duration,size", + "-show_chapters", + "-of", "json", + file, + ], { maxBuffer: 1 << 24 }); + const probe = JSON.parse(stdout); + const duration = Number(probe.format?.duration ?? 0); + const chapters = (probe.chapters ?? []).length; + const entries = (manifest.timeline ?? []).length; + + if (!(duration > 0)) problems.push("duration is not greater than zero"); + + // Every timeline entry becomes a chapter, so a mismatch means the timeline and + // the file disagree about what is in it -- which is exactly the failure + // --continue-on-error is allowed to cause and must never cause silently. + if (chapters > 0 && chapters !== entries) { + problems.push(`${chapters} chapter(s) for ${entries} timeline entr(ies) — the cut is missing something`); + } + + // A rough floor: the sum of the windows, less the crossfades. Well under the + // real duration because snapping moves the cuts, but a file that came out at + // half the expected length did not build what was asked for. + const wanted = (manifest.timeline ?? []).reduce( + // `seconds` covers cards and the two end-sequence kinds (scroll, chart); + // only a clip's length has to be derived from its window. + (n, e) => n + (e.type === "clip" ? Math.max(0, (e.end ?? 0) - (e.start ?? 0)) : (e.seconds ?? 0)), + 0, + ); + if (wanted > 0 && duration < wanted * 0.5) { + problems.push(`${duration.toFixed(1)}s out of a timeline that asks for about ${wanted.toFixed(0)}s`); + } + + return { ok: problems.length === 0, variant, file, duration, chapters, entries, size: st.size, problems }; +} + +async function main() { + const argv = process.argv.slice(2); + const manifestPath = argv.find((a) => !a.startsWith("--")); + if (!manifestPath) { + console.error("usage: verify-build.mjs <manifest.json> [--out <dir>] [--variant sourced|full] [--json]"); + process.exit(2); + } + const flag = (n) => { const i = argv.indexOf(n); return i >= 0 ? argv[i + 1] : undefined; }; + const res = await verifyBuild(manifestPath, { + outDir: flag("--out"), + variant: flag("--variant") ?? "sourced", + }); + + if (argv.includes("--json")) { + console.log(JSON.stringify(res, null, 2)); + } else { + console.log( + `${res.file} (${res.variant})\n ${res.duration?.toFixed(1) ?? "?"}s · ${res.chapters ?? 0} chapter(s) for ` + + `${res.entries ?? 0} entr(ies) · ${((res.size ?? 0) / 1e6).toFixed(1)} MB`, + ); + for (const p of res.problems) console.log(` ** ${p}`); + if (res.ok) console.log(" ok"); + } + process.exit(res.ok ? 0 : 1); +} + +if (import.meta.url === `file://${process.argv[1]}`) { + main().catch((err) => { + console.error(err.message ?? err); + process.exit(1); + }); +} diff --git a/umtool/app/api/browse/decisions/route.ts b/umtool/app/api/browse/decisions/route.ts @@ -1,4 +1,5 @@ -import { countBySeverity, decisionsForSong, openDecisions } from "@/lib/decisions"; +import { countBySeverity, decisionsForSong } from "@/lib/decisions"; +import { openDecisions } from "@/lib/projects"; import { isSegment } from "@/lib/browse"; export const dynamic = "force-dynamic"; diff --git a/umtool/app/api/browse/poster/route.ts b/umtool/app/api/browse/poster/route.ts @@ -3,6 +3,8 @@ import path from "node:path"; import { SONG_REPORTS } from "@/lib/paths"; import { readSong, resolveRendition, type Song } from "@/lib/browse"; import { posterFrame } from "@/lib/poster"; +import { projectRef, summariseProject } from "@/lib/projects"; +import { resolveInRoots } from "@/lib/paths"; export const dynamic = "force-dynamic"; @@ -39,10 +41,38 @@ function send(buf: Buffer, file: string, source: string) { export async function GET(request: Request) { const url = new URL(request.url); + const project = url.searchParams.get("project"); const id = url.searchParams.get("song") ?? ""; const rel = url.searchParams.get("rel"); const width = Math.min(1280, Math.max(120, Number(url.searchParams.get("w")) || 480)); + // `?project=` is the general form; `?song=` stays because every existing song + // link and every existing spec uses it. + // + // A project poster takes NO client-supplied rel. The frame it draws is the + // one its own summariser chose -- the deliverable, else a built segment (which + // already carries the chrome, so the card looks like the video mid-build), + // else a raw clip. That is a server-derived path, which is why this route can + // widen to every kind without widening what a caller can ask it to open. + if (project) { + const ref = await projectRef(project); + if (!ref) return new Response("no such project", { status: 404 }); + const summary = await summariseProject(ref); + if (!summary.posterRel) return new Response("nothing to draw", { status: 404 }); + const abs = resolveInRoots(path.join(ref.dir, summary.posterRel)); + if (!abs) return new Response("outside the roots", { status: 400 }); + if (TYPES[path.extname(abs).toLowerCase()]) { + try { + return send(await readFile(abs), abs, "file"); + } catch { + return new Response("nothing to draw", { status: 404 }); + } + } + const poster = await posterFrame(abs, width); + if (!poster) return new Response(null, { status: 404 }); + return send(await readFile(poster.file), poster.file, poster.cached ? "cached" : "fresh"); + } + const song = await readSong(id); if (!song) return new Response("no such song", { status: 404 }); diff --git a/umtool/app/api/browse/projects/route.ts b/umtool/app/api/browse/projects/route.ts @@ -0,0 +1,34 @@ +import { decisionCounts, indexHealth, listFolders, listProjects } from "@/lib/projects"; + +export const dynamic = "force-dynamic"; + +// Every project, as JSON. The index page's data, for an agent or a script. +// +// `x-index` reports how much of that came from the persistent index and how much +// was read fresh. It is the only place the index is observable at all, which is +// the intent: a stale index self-heals on the next load and the user sees +// nothing but latency, so the health has to be surfaced deliberately or not at +// all. The specs assert on it. +export async function GET(request: Request) { + const url = new URL(request.url); + const projects = await listProjects(); + const health = indexHealth(); + + const counts = url.searchParams.get("decisions") === "1" ? await decisionCounts() : null; + const folders = [...(await listFolders()).values()]; + + return Response.json( + { + projects: counts + ? projects.map((p) => ({ ...p, decisions: counts.get(p.id) ?? null })) + : projects, + folders, + }, + { + headers: { + "cache-control": "no-store", + "x-index": health.ok ? `${health.fresh}/${health.total} fresh` : "off", + }, + }, + ); +} diff --git a/umtool/app/api/mix/files/route.ts b/umtool/app/api/mix/files/route.ts @@ -1,12 +1,25 @@ -import { listMedia } from "@/lib/media"; import { MEDIA_ROOTS } from "@/lib/paths"; +import { listMedia } from "@/lib/media"; +import { mediaGroups } from "@/lib/projects"; export const dynamic = "force-dynamic"; -// Everything the bench could load, newest first -- which is almost always the -// order you want, because the file you are about to judge is the one that just -// finished rendering. -export async function GET() { +// Everything the bench could load. +// +// `?group=project` is the shape the picker uses now. The flat list stays for +// anything that just wants "the newest renders", and because a flat newest-first +// list is still almost always the order you want when the file you are about to +// judge is the one that just finished rendering -- it is only COVERAGE it is bad +// at, and coverage is what a picker over twelve projects needs. +export async function GET(request: Request) { + const grouped = new URL(request.url).searchParams.get("group") === "project"; + if (grouped) { + const { groups, other } = await mediaGroups(); + return Response.json( + { roots: MEDIA_ROOTS, groups, other }, + { headers: { "cache-control": "no-store" } }, + ); + } const files = await listMedia(); return Response.json( { roots: MEDIA_ROOTS, files }, diff --git a/umtool/app/api/report/build/route.ts b/umtool/app/api/report/build/route.ts @@ -0,0 +1,141 @@ +import { rename, stat } from "node:fs/promises"; +import path from "node:path"; +import { cancelJob, getJob, jobView, recentJobs, runningJob, startJob } from "@/lib/jobs"; +import { PRESETS, buildSteps } from "@/lib/report/driver.mjs"; +import { clipsOf, readManifest } from "@/lib/projects/report.mjs"; +import { projectRef } from "@/lib/projects"; + +export const dynamic = "force-dynamic"; + +// Building a report video. +// +// The client sends a PROJECT ID and a PRESET NAME. It never sends a path, an +// argv or an env map -- the same contract /api/browse/build keeps, and the +// reason the output path is computed here rather than accepted. +// +// POLLING, NOT STREAMING, like every other job in this app. The progress that +// matters is per-clip, and that arrives as NDJSON events the driver collects. + +/** `20260817-1408`, the stamp shape promote already uses for a demoted cut. */ +function stamp(d = new Date()): string { + const p = (n: number) => String(n).padStart(2, "0"); + return `${d.getFullYear()}${p(d.getMonth() + 1)}${p(d.getDate())}-${p(d.getHours())}${p(d.getMinutes())}`; +} + +export async function GET(request: Request) { + const url = new URL(request.url); + const id = url.searchParams.get("job"); + const headers = { "cache-control": "no-store" }; + if (!id) { + const running = runningJob(); + return Response.json( + { + running: running ? jobView(running) : null, + jobs: recentJobs(5).map((j) => jobView(j, j.log.length, j.events.length)), + presets: Object.entries(PRESETS).map(([k, v]) => ({ id: k, label: v.label })), + }, + { headers }, + ); + } + const job = getJob(id); + if (!job) return Response.json({ error: "no such job" }, { status: 404 }); + return Response.json( + jobView(job, Number(url.searchParams.get("since") ?? 0), Number(url.searchParams.get("sinceEvent") ?? 0)), + { headers }, + ); +} + +export async function POST(request: Request) { + const url = new URL(request.url); + const dry = url.searchParams.get("dry") === "1"; + const replace = url.searchParams.get("replace") === "1"; + const cancel = url.searchParams.get("cancel"); + + if (cancel) { + // Safe at any point, and worth saying why: every artefact is + // content-addressed -- a fetched window by its window, a segment by its clip + // id -- so re-running skips whatever finished. A cancelled build is paused. + return Response.json({ cancelled: cancelJob(cancel) }); + } + + const body = (await request.json().catch(() => ({}))) as Record<string, unknown>; + const projectId = String(body.project ?? ""); + const preset = String(body.preset ?? "fast"); + const only = body.only ? String(body.only) : null; + const skipFetch = !!body.skipFetch; + + if (!(preset in PRESETS)) { + return Response.json( + { error: `preset must be one of ${Object.keys(PRESETS).join(", ")}` }, + { status: 400 }, + ); + } + + const project = await projectRef(projectId); + if (!project) return Response.json({ error: "no such project" }, { status: 404 }); + const manifest = await readManifest(project.dir); + if (!manifest) return Response.json({ error: "no manifest" }, { status: 400 }); + if (only && !clipsOf(manifest).some((e: { id: string }) => e.id === only)) { + return Response.json({ error: `no clip with id ${only}` }, { status: 400 }); + } + + const clipCount = clipsOf(manifest).length; + const steps = buildSteps(project, { preset, only, skipFetch, clipCount }); + const view = { + project: project.id, + preset, + only, + steps: steps.map((s: { label: string; argv: string[]; cwd: string; timeoutMs?: number }) => ({ + label: s.label, + argv: s.argv, + cwd: s.cwd, + timeoutMs: s.timeoutMs ?? null, + })), + }; + + // The exact argv before anything runs. Same contract as /api/mix/render?dry=1. + if (dry) return Response.json({ dry: true, ...view }, { headers: { "cache-control": "no-store" } }); + + // ---- the overwrite guard -------------------------------------------------- + // + // build-video.mjs always passes -y. A deliverable that cost an hour of network + // fetches must not be destroyed to make a new one, so an existing output that + // is NEWER than the manifest is refused; on ?replace=1 it is stamped aside + // rather than overwritten. + const finalPath = path.join(project.dir, "out", `${manifest.slug}.mp4`); + if (!only) { + const [fin, man] = await Promise.all([ + stat(finalPath).catch(() => null), + stat(path.join(project.dir, "video.manifest.json")).catch(() => null), + ]); + if (fin && man && fin.mtimeMs >= man.mtimeMs) { + if (!replace) { + return Response.json( + { + error: + `${path.basename(finalPath)} is newer than the manifest — building would overwrite ` + + "a deliverable nothing has asked to change. Pass replace=1 to stamp it aside.", + needsReplace: true, + }, + { status: 409 }, + ); + } + await rename(finalPath, finalPath.replace(/\.mp4$/, `.${stamp()}.mp4`)); + } + } + + const running = runningJob(); + if (running) { + return Response.json( + { error: `a job is already running (${running.kind})`, running: jobView(running) }, + { status: 409 }, + ); + } + + try { + const job = startJob(`build ${project.id} (${preset})`, steps); + return Response.json({ ok: true, ...view, job: jobView(job) }, { headers: { "cache-control": "no-store" } }); + } catch (e) { + return Response.json({ error: e instanceof Error ? e.message : String(e) }, { status: 409 }); + } +} diff --git a/umtool/app/api/report/claim/route.ts b/umtool/app/api/report/claim/route.ts @@ -0,0 +1,65 @@ +import { StaleToken, manifestToken, updateClaim } from "@/lib/report/manifest.mjs"; +import { resolveClaim } from "@/lib/report/serve.mjs"; +import { readClaimDetail } from "@/lib/projects/report.mjs"; + +export const dynamic = "force-dynamic"; + +// One ledger claim, and the ruling on it. +// +// GET mirrors /api/report/clip: everything the bench needs in one request, plus +// the manifest's mtime as a WRITE TOKEN, so a save can be refused when somebody +// -- another tab, an agent, a `resolve-windows --write` -- has written in +// between. Losing that write would be losing human judgement, which is the one +// thing this whole surface exists to collect. + +export async function GET(request: Request) { + const url = new URL(request.url); + const projectId = url.searchParams.get("project") ?? ""; + const claimId = url.searchParams.get("claim") ?? ""; + const padArg = Number(url.searchParams.get("pad")); + + const r = await resolveClaim(projectId, claimId); + if ("error" in r) return Response.json({ error: r.error }, { status: r.status }); + + const detail = await readClaimDetail(r.project.dir, claimId, { + manifest: r.manifest, + ...(Number.isFinite(padArg) && padArg > 0 ? { pad: Math.min(600, padArg) } : {}), + }); + if (!detail) return Response.json({ error: "no such claim" }, { status: 404 }); + + return Response.json( + { project: r.project.id, ...detail, token: await manifestToken(r.project.dir) }, + { headers: { "cache-control": "no-store" } }, + ); +} + +export async function PUT(request: Request) { + const body = (await request.json().catch(() => ({}))) as Record<string, unknown>; + const projectId = String(body.project ?? ""); + const claimId = String(body.claim ?? ""); + + const r = await resolveClaim(projectId, claimId); + if ("error" in r) return Response.json({ error: r.error }, { status: r.status }); + + const patch: Record<string, unknown> = {}; + for (const k of ["scope", "scopeBasis", "scopeConfidence", "population", "valueKind", "flags", "roles", "note"]) { + if (body[k] !== undefined) patch[k] = body[k]; + } + if (!Object.keys(patch).length) { + return Response.json({ error: "nothing to write" }, { status: 400 }); + } + + try { + const { entry, token } = await updateClaim(r.project.dir, claimId, patch, { + token: body.token === undefined ? null : String(body.token), + }); + const detail = await readClaimDetail(r.project.dir, claimId); + return Response.json({ claim: entry, gaps: detail?.gaps ?? [], token }); + } catch (err) { + // A stale token is a 409 and never a silent overwrite. + if (err instanceof StaleToken) { + return Response.json({ error: err.message, stale: true }, { status: 409 }); + } + return Response.json({ error: (err as Error).message }, { status: 400 }); + } +} diff --git a/umtool/app/api/report/clip/route.ts b/umtool/app/api/report/clip/route.ts @@ -0,0 +1,75 @@ +import path from "node:path"; +import { resolveClip, windowsFor } from "@/lib/report/serve.mjs"; +import { manifestToken } from "@/lib/report/manifest.mjs"; +import { cuesInWindow, readClipDetail } from "@/lib/projects/report.mjs"; + +export const dynamic = "force-dynamic"; + +// Everything the bench needs to open a clip, in one request. +// +// The window, the cached source files it can draw from, the cues around it, and +// the manifest's mtime as a WRITE TOKEN -- handed out here so a save can be +// refused when somebody (a `resolve-windows --write`, an agent, another tab) +// has written in between. +export async function GET(request: Request) { + const url = new URL(request.url); + const projectId = url.searchParams.get("project") ?? ""; + const clipId = url.searchParams.get("clip") ?? ""; + + const r = await resolveClip(projectId, clipId); + if ("error" in r) return Response.json({ error: r.error }, { status: r.status }); + const { project, manifest, clip } = r; + + const windows = await windowsFor(project, clip); + const detail = await readClipDetail(project.dir, { manifest }); + const entry = detail?.entries.find((e: { id: string }) => e.id === clipId) ?? null; + + // The view is the widest cached file when there is one, else a generous span + // around the window so the waveform is not a blank rectangle before a fetch. + const widest = windows[0] ?? null; + const view = widest + ? { from: widest.from, to: widest.to } + : { from: Math.max(0, clip.start - 20), to: clip.end + 20 }; + + const cues = await cuesInWindow(project.dir, clipId, view.from, view.to); + + return Response.json( + { + project: project.id, + dir: project.dir, + clip: { + id: clip.id, + video: clip.video, + channel: entry?.channel ?? null, + start: clip.start, + end: clip.end, + cite: clip.cite ?? null, + quote: clip.quote ?? null, + note: clip.note ?? null, + lock: !!clip.lock, + lockStart: !!clip.lockStart, + lockEnd: !!clip.lockEnd, + }, + view, + windows: windows.map((w: { name: string; from: number; to: number }) => ({ + name: w.name, + from: w.from, + to: w.to, + })), + // What resolve-windows WOULD do, computed in-process because widen() is + // pure once the cues are read. "Run the widener and see" stops being a + // leap of faith. + proposed: entry?.proposed ?? null, + endsSentence: entry?.endsSentence ?? null, + noPunctuation: entry?.noPunctuation ?? false, + sourceDuration: entry?.duration ?? null, + segment: entry?.segment ?? null, + cues: cues?.cues ?? [], + punctuationRate: cues?.punctuationRate ?? null, + token: await manifestToken(project.dir), + fetchPad: manifest.render?.fetchPad ?? 3, + outLabel: path.posix.join(project.id, "out"), + }, + { headers: { "cache-control": "no-store" } }, + ); +} diff --git a/umtool/app/api/report/cues/route.ts b/umtool/app/api/report/cues/route.ts @@ -0,0 +1,26 @@ +import { resolveClip } from "@/lib/report/serve.mjs"; +import { cuesInWindow } from "@/lib/projects/report.mjs"; + +export const dynamic = "force-dynamic"; + +// The cues around a clip, so "what am I cutting off" is READ rather than +// inferred from a waveform. Each carries whether it closes a sentence, computed +// with the same regex resolve-windows.mjs uses -- imported, not re-written, so +// the rail and the widener can never disagree about where a sentence ends. +export async function GET(request: Request) { + const url = new URL(request.url); + const projectId = url.searchParams.get("project") ?? ""; + const clipId = url.searchParams.get("clip") ?? ""; + const from = Number(url.searchParams.get("from")); + const to = Number(url.searchParams.get("to")); + if (!Number.isFinite(from) || !Number.isFinite(to) || to <= from) { + return Response.json({ error: "bad window" }, { status: 400 }); + } + + const r = await resolveClip(projectId, clipId); + if ("error" in r) return Response.json({ error: r.error }, { status: r.status }); + + const cues = await cuesInWindow(r.project.dir, clipId, from, to); + if (!cues) return Response.json({ error: "no cue file for this source" }, { status: 404 }); + return Response.json(cues, { headers: { "cache-control": "no-store" } }); +} diff --git a/umtool/app/api/report/fetch/route.ts b/umtool/app/api/report/fetch/route.ts @@ -0,0 +1,49 @@ +import { getJob, jobView, runningJob, startJob } from "@/lib/jobs"; +import { fetchSteps } from "@/lib/report/driver.mjs"; +import { resolveClip } from "@/lib/report/serve.mjs"; + +export const dynamic = "force-dynamic"; + +// "Fetch 20s more", from the clip bench. +// +// A DRAG NEVER DOWNLOADS. Dragging past the cached window clamps and offers this +// button instead, because a handle that silently starts a 12-second network +// fetch is a handle you stop trusting. It runs the pipeline's own fetch path +// (build-video.mjs --fetch-only), so the file it writes is named the way the +// build expects, gets the same format pin, and inherits the Rumble HLS retry -- +// and containing-window reuse then makes this generous fetch BE the build's +// cache rather than a second one. + +const MAX_PAD = 120; + +export async function POST(request: Request) { + const body = (await request.json().catch(() => ({}))) as Record<string, unknown>; + const projectId = String(body.project ?? ""); + const clipId = String(body.clip ?? ""); + const pad = Math.max(1, Math.min(MAX_PAD, Number(body.pad ?? 20))); + + const r = await resolveClip(projectId, clipId); + if ("error" in r) return Response.json({ error: r.error }, { status: r.status }); + + const running = runningJob(); + if (running) { + return Response.json( + { error: `a job is already running (${running.kind})`, job: jobView(running) }, + { status: 409 }, + ); + } + + const job = startJob(`fetch ${projectId}/${clipId}`, fetchSteps(r.project, clipId, pad)); + return Response.json({ job: jobView(job) }, { status: 202 }); +} + +export async function GET(request: Request) { + const url = new URL(request.url); + const id = url.searchParams.get("job"); + const job = id ? getJob(id) : runningJob(); + if (!job) return Response.json({ job: null }, { headers: { "cache-control": "no-store" } }); + return Response.json( + { job: jobView(job, Number(url.searchParams.get("since") ?? 0), Number(url.searchParams.get("sinceEvent") ?? 0)) }, + { headers: { "cache-control": "no-store" } }, + ); +} diff --git a/umtool/app/api/report/peaks/route.ts b/umtool/app/api/report/peaks/route.ts @@ -0,0 +1,64 @@ +import { absOf, pickWindow, resolveClip, windowsFor } from "@/lib/report/serve.mjs"; +import { ENV_RATE, analyseMedia } from "@/lib/media"; + +export const dynamic = "force-dynamic"; + +// The envelope of a cached source window, in ABSOLUTE SOURCE SECONDS. +// +// Everything in the bench is in source seconds -- the manifest's, the cues', +// the QR's. The file's own start is the only relative number, and it is added +// here so nothing downstream has to remember to. +// +// analyseMedia decodes once and caches to MIX_CACHE keyed by mtime and size. A +// raw clip's window is IN ITS NAME, so the file is immutable and that cache can +// never miss twice for the same window. + +export async function GET(request: Request) { + const url = new URL(request.url); + const r = await resolveClip(url.searchParams.get("project") ?? "", url.searchParams.get("clip") ?? ""); + if ("error" in r) return Response.json({ error: r.error }, { status: r.status }); + + const windows = await windowsFor(r.project, r.clip); + const win = pickWindow(windows, url.searchParams.get("file")); + if (!win) return Response.json({ error: "no cached window for this clip" }, { status: 404 }); + const abs = absOf(win); + if (!abs) return Response.json({ error: "outside the roots" }, { status: 400 }); + + const buckets = Math.max(64, Math.min(4000, Number(url.searchParams.get("n") ?? 1200))); + + let analysis; + try { + analysis = await analyseMedia(abs); + } catch (e) { + return Response.json({ error: e instanceof Error ? e.message : String(e) }, { status: 500 }); + } + + const env = analysis.peak; + const n = env.length; + const min: number[] = []; + const max: number[] = []; + for (let b = 0; b < buckets; b += 1) { + const i0 = Math.floor((b / buckets) * n); + const i1 = Math.max(i0 + 1, Math.floor(((b + 1) / buckets) * n)); + let p = 0; + for (let i = i0; i < i1 && i < n; i += 1) if (env[i] > p) p = env[i]; + max.push(p); + // analyseMedia's envelope is positive-only. Mirroring it is what makes the + // canvas draw the symmetric shape a waveform is expected to be, rather than + // a row of upward spikes. + min.push(-p); + } + + return Response.json( + { + from: win.from, + to: win.from + (n / ENV_RATE), + duration: analysis.duration, + fetchStart: win.from, + window: win.name, + min, + max, + }, + { headers: { "cache-control": "private, max-age=3600" } }, + ); +} diff --git a/umtool/app/api/report/raw/route.ts b/umtool/app/api/report/raw/route.ts @@ -0,0 +1,95 @@ +import { createReadStream } from "node:fs"; +import { stat } from "node:fs/promises"; +import { Readable } from "node:stream"; +import { absOf, pickWindow, resolveClaim, resolveClip, windowsFor } from "@/lib/report/serve.mjs"; +import { readClaimDetail } from "@/lib/projects/report.mjs"; + +export const dynamic = "force-dynamic"; + +// The cached source window, served WHOLE with byte ranges. +// +// The alternative -- an ffmpeg slice per requested window, like +// /api/clip/[key]/video does -- would put an encode in front of every drag. The +// file is small (one clip plus its pad), immutable (its window is in its name), +// and the browser only needs to be told where to seek: the bench sets +// currentTime = t - fetchStart and windows entirely on the client. +// +// Range support is not optional for that. Without a 206 the <video> element +// will not seek in a stream it did not fully download. + +export async function GET(request: Request) { + const url = new URL(request.url); + const projectId = url.searchParams.get("project") ?? ""; + const claimId = url.searchParams.get("claim"); + + // A CLAIM names a moment rather than a window, so its candidate files are the + // cached windows that CONTAIN its cite. Everything below -- membership, + // root-resolution, ranges, the immutable cache header -- is the clip path's, + // unchanged: the two differ only in which list of windows they may pick from. + let windows: Array<{ name: string; from: number; to: number; path: string }>; + if (claimId) { + const c = await resolveClaim(projectId, claimId); + if ("error" in c) return new Response(c.error, { status: c.status }); + const detail = await readClaimDetail(c.project.dir, claimId, { manifest: c.manifest }); + windows = (detail?.windows ?? []) as typeof windows; + // readClaimDetail returns names, not paths; re-resolve against the scan. + const all = await windowsFor(c.project, { video: c.claim.video }); + windows = all.filter((w: { name: string }) => windows.some((x) => x.name === w.name)); + } else { + const r = await resolveClip(projectId, url.searchParams.get("clip") ?? ""); + if ("error" in r) return new Response(r.error, { status: r.status }); + windows = await windowsFor(r.project, r.clip); + } + + // A member of the server's own scan, never a path from the client. + const win = pickWindow(windows, url.searchParams.get("file")); + if (!win) return new Response("no cached window covering that moment", { status: 404 }); + const abs = absOf(win); + if (!abs) return new Response("outside the roots", { status: 400 }); + + const st = await stat(abs).catch(() => null); + if (!st) return new Response("gone", { status: 404 }); + + const headers: Record<string, string> = { + "content-type": "video/mp4", + "accept-ranges": "bytes", + // The file is immutable -- rename the window and it is a different file -- + // so it may be cached hard. That is what makes dragging feel instant. + "cache-control": "private, max-age=3600, immutable", + // The absolute source second the file starts at, so the client can convert + // without a second request. + "x-fetch-start": String(win.from), + "x-fetch-end": String(win.to), + "x-window": win.name, + }; + + const range = request.headers.get("range"); + const m = range ? /^bytes=(\d*)-(\d*)$/.exec(range.trim()) : null; + if (m) { + const size = st.size; + let start = m[1] ? Number(m[1]) : 0; + let end = m[2] ? Number(m[2]) : size - 1; + if (!m[1] && m[2]) { + // A suffix range: the LAST n bytes. + start = Math.max(0, size - Number(m[2])); + end = size - 1; + } + if (!Number.isFinite(start) || !Number.isFinite(end) || start > end || start >= size) { + return new Response(null, { status: 416, headers: { "content-range": `bytes */${size}` } }); + } + end = Math.min(end, size - 1); + const stream = createReadStream(abs, { start, end }); + return new Response(Readable.toWeb(stream) as ReadableStream, { + status: 206, + headers: { + ...headers, + "content-range": `bytes ${start}-${end}/${size}`, + "content-length": String(end - start + 1), + }, + }); + } + + return new Response(Readable.toWeb(createReadStream(abs)) as ReadableStream, { + headers: { ...headers, "content-length": String(st.size) }, + }); +} diff --git a/umtool/app/api/report/window/route.ts b/umtool/app/api/report/window/route.ts @@ -0,0 +1,52 @@ +import { StaleToken, updateClip } from "@/lib/report/manifest.mjs"; +import { resolveClip } from "@/lib/report/serve.mjs"; + +export const dynamic = "force-dynamic"; + +// Saving a window. +// +// The client sends a project id, a clip id, numbers, and the TOKEN it was given +// when it opened the clip. It never sends a path, and it cannot ask for an +// entry to move: re-ordering recomputes `sectionEnter` and changes the cut, so +// it is a different operation with a different button. +// +// A stale token is a 409 with both values, not a silent overwrite. The other +// writer is usually `resolve-windows --write` or an agent running `umtool`, and +// what it wrote is somebody's judgement. +export async function PUT(request: Request) { + let body: Record<string, unknown>; + try { + body = await request.json(); + } catch { + return Response.json({ error: "expected JSON" }, { status: 400 }); + } + + const projectId = String(body.project ?? ""); + const clipId = String(body.clip ?? ""); + const r = await resolveClip(projectId, clipId); + if ("error" in r) return Response.json({ error: r.error }, { status: r.status }); + + const patch: Record<string, unknown> = {}; + for (const k of ["start", "end", "lock", "lockStart", "lockEnd", "note"]) { + if (body[k] !== undefined) patch[k] = body[k]; + } + if (!Object.keys(patch).length) return Response.json({ error: "nothing to change" }, { status: 400 }); + + try { + const res = await updateClip(r.project.dir, clipId, patch, { + token: body.token === undefined ? null : String(body.token), + }); + return Response.json( + { ok: true, entry: res.entry, before: res.before, token: res.token }, + { headers: { "cache-control": "no-store" } }, + ); + } catch (e) { + if (e instanceof StaleToken) { + return Response.json( + { error: e.message, expected: e.expected, got: e.got, stale: true }, + { status: 409 }, + ); + } + return Response.json({ error: e instanceof Error ? e.message : String(e) }, { status: 400 }); + } +} diff --git a/umtool/app/browse/[...path]/page.tsx b/umtool/app/browse/[...path]/page.tsx @@ -0,0 +1,88 @@ +import Link from "next/link"; +import { notFound } from "next/navigation"; +import BrowseHeader from "@/components/BrowseHeader"; +import ProjectView from "@/components/projects/ProjectView"; +import ProjectGrid from "@/components/projects/ProjectGrid"; +import { listProjects, resolveProjectPath } from "@/lib/projects"; + +export const dynamic = "force-dynamic"; + +// --------------------------------------------------------------------------- +// Everything below /browse that is not a tool page. +// +// This replaced app/browse/[song]/ rather than joining it: two dynamic segments +// at the same level is a Next routing conflict, so it is a delete-and-move. The +// song pages themselves are unchanged -- they moved to components/projects/ and +// are handed an id instead of awaiting params. /browse/yoshi/wide is still the +// same URL and still renders the same page. +// +// Resolution is LONGEST-PREFIX: a project id is a path, so `a/b` being a project +// must not stop `a/b/c` from being one too. Whatever is left over is a VIEW, and +// which views exist is the kind's business, not this file's. +// --------------------------------------------------------------------------- + +export default async function BrowsePathPage({ + params, + searchParams, +}: { + params: Promise<{ path?: string[] }>; + searchParams: Promise<Record<string, string | undefined>>; +}) { + const { path: segments = [] } = await params; + const search = await searchParams; + const resolved = await resolveProjectPath(segments); + if (!resolved) notFound(); + + if (resolved.project) { + return <ProjectView project={resolved.project} rest={resolved.rest} search={search} />; + } + + // Two projects answer to the same bare name. Which one you meant is not + // guessable, and guessing is how a link quietly starts pointing at the wrong + // thing -- so the page says so and offers both canonical paths. + if ("ambiguous" in resolved && resolved.ambiguous) { + return ( + <div className="flex h-full flex-col"> + <BrowseHeader + crumbs={[{ href: "/browse", label: "projects" }, { label: segments.join("/") }]} + note={`${resolved.ambiguous.length} projects answer to that name`} + /> + <main className="deck-main flex-1 p-4"> + <p className="mb-2 text-[12px] text-[var(--color-dim)]"> + <code className="font-mono">{segments[0]}</code> is the name of more than one project, + so it cannot be a shortcut to any of them. Use a full path: + </p> + <ul className="space-y-1"> + {resolved.ambiguous.map((p) => ( + <li key={p.id}> + <Link + href={`/browse/${p.id}${segments.slice(1).map((s) => `/${s}`).join("")}`} + className="font-mono text-[12px] text-[var(--color-sel)] hover:underline" + > + /browse/{p.id} + </Link> + </li> + ))} + </ul> + </main> + </div> + ); + } + + // A folder: the same grid, filtered to what is under it. + const folder = resolved.folder; + if (!folder) notFound(); + const all = await listProjects(); + const mine = all.filter((p) => p.id.startsWith(`${folder.path}/`)); + return ( + <div className="flex h-full flex-col"> + <BrowseHeader + crumbs={[{ href: "/browse", label: "projects" }, { label: folder.label }]} + note={`${mine.length} project${mine.length === 1 ? "" : "s"}`} + /> + <main className="deck-main flex-1 p-4"> + <ProjectGrid projects={mine} /> + </main> + </div> + ); +} diff --git a/umtool/app/browse/[song]/[cut]/page.tsx b/umtool/app/browse/[song]/[cut]/page.tsx @@ -1,122 +0,0 @@ -import Link from "next/link"; -import { notFound } from "next/navigation"; -import BrowseHeader from "@/components/BrowseHeader"; -import CutBench from "@/components/CutBench"; -import ProvenancePanel from "@/components/ProvenancePanel"; -import { isCutName, probeAll, readSong } from "@/lib/browse"; -import { fmtBytes, fmtDur } from "@/lib/format"; -import { buildStatus } from "@/lib/manifest"; -import { readProvenance } from "@/lib/provenance"; -import { readNotes } from "@/lib/notes"; -import { readCompares } from "@/lib/compares"; -import LoudnessTable from "@/components/LoudnessTable"; -import { cachedLoudness } from "@/lib/loudness"; -import { DEFAULT_TARGET } from "@/lib/loudness-types"; -import { readSpec } from "@/lib/spec"; -import { resolveRendition } from "@/lib/browse"; - -export const dynamic = "force-dynamic"; - -export default async function CutPage({ - params, - searchParams, -}: { - params: Promise<{ song: string; cut: string }>; - searchParams: Promise<{ v?: string; plan?: string; notes?: string }>; -}) { - const { song: id, cut: cutName } = await params; - const { v, plan, notes: notesParam } = await searchParams; - if (!isCutName(cutName)) notFound(); - - const song = await readSong(id); - if (!song) notFound(); - const cut = song.cuts.find((c) => c.name === cutName); - if (!cut) notFound(); - - const rels = [cut.shipped?.rel, ...cut.variants.map((x) => x.rel)].filter( - (r): r is string => !!r, - ); - const info = await probeAll(id, rels); - - const durations: Record<string, number> = {}; - for (const [rel, i] of Object.entries(info)) durations[rel] = i.duration; - - // The recipe is recorded per FILE, so the shipped cut is the one it describes. - // A named plan in build.json is a fact; the ?plan= override is the user's - // choice where no builder recorded one. - const build = cut.shipped - ? await buildStatus(id, cut.shipped.rel) - : { entry: null, stale: null }; - const prov = await readProvenance(id, plan ?? build.entry?.plans[0] ?? null); - const notes = await readNotes(id); - const compares = await readCompares(id); - - // Cached figures only. Measuring here would put an ffmpeg decode per file in - // front of every navigation to this page. - const spec = await readSpec(id); - const loudness = await Promise.all( - rels.map(async (rel) => { - const abs = resolveRendition(id, rel); - return { rel, loudness: abs ? await cachedLoudness(abs) : null }; - }), - ); - const target = { - lufs: spec.loudness?.targetLufs ?? DEFAULT_TARGET.lufs, - truePeak: spec.loudness?.truePeak ?? DEFAULT_TARGET.truePeak, - }; - - return ( - <div className="flex h-full flex-col"> - <BrowseHeader - crumbs={[ - { href: "/browse", label: "songs" }, - { href: `/browse/${song.id}`, label: song.id }, - { label: cutName }, - ]} - note={ - cut.shipped - ? `${fmtDur(durations[cut.shipped.rel] ?? 0)} · ${fmtBytes(cut.shipped.size)}` - : "not built" - } - /> - <main className="deck-main flex-1"> - {!cut.shipped && cut.variants.length === 0 ? ( - <div className="p-4 text-[12px] text-[var(--color-dim)]"> - Nothing here yet — no <code className="font-mono">{cutName}.mp4</code> and no variants - named for it.{" "} - <Link href={`/browse/${song.id}`} className="text-[var(--color-sel)] underline"> - back to {song.id} - </Link> - </div> - ) : ( - <> - <CutBench - song={song.id} - cut={cutName} - shipped={cut.shipped} - variants={cut.variants} - durations={durations} - initialVariant={v ?? null} - notes={notes} - compares={compares} - /> - <div className="px-4 pb-4"> - <LoudnessTable song={song.id} initial={loudness} target={target} /> - </div> - <div className="px-4 pb-4"> - <ProvenancePanel - song={song.id} - cut={cutName} - prov={prov} - build={build} - planParam={plan ?? null} - notes={notes} - allNotes={notesParam === "all"} - /> - </div> - </> - )} - </main> - </div> - ); -} diff --git a/umtool/app/browse/[song]/page.tsx b/umtool/app/browse/[song]/page.tsx @@ -1,254 +0,0 @@ -import Link from "next/link"; -import { notFound } from "next/navigation"; -import BrowseHeader from "@/components/BrowseHeader"; -import VerdictChip from "@/components/VerdictChip"; -import SpecSheet from "@/components/SpecSheet"; -import NoteField from "@/components/NoteField"; -import CopyButton from "@/components/CopyButton"; -import { readSong, probeAll, thumbAliasesFor, type Cut, type Song } from "@/lib/browse"; -import ThumbBench from "@/components/ThumbBench"; -import { thumbView } from "@/lib/thumbs"; -import { fmtAgo, fmtBytes, fmtDur } from "@/lib/format"; -import { Markdown } from "@/lib/markdown"; -import { operationsFor, readSpec, validateSpec } from "@/lib/spec"; -import { readNotes, type NoteMap } from "@/lib/notes"; -import { TRIM_SETS } from "@/lib/trim"; - -export const dynamic = "force-dynamic"; - -export default async function SongPage({ params }: { params: Promise<{ song: string }> }) { - const { song: id } = await params; - const song = await readSong(id); - if (!song) notFound(); - - // Durations are worth an ffprobe HERE but not on the index: this page is a - // handful of files and "which of these is the short cut" is exactly the - // question it answers. Memoised by mtime in lib/browse.ts. - const rels = [ - ...song.cuts.flatMap((c) => [c.shipped?.rel, ...c.variants.map((v) => v.rel)]), - ...song.unattributed.map((v) => v.rel), - ].filter((r): r is string => !!r); - const info = await probeAll(id, rels); - - const spec = await readSpec(id); - const problems = await validateSpec(spec); - const notes = await readNotes(id); - // Two small JSON reads, no probing -- the bench draws what the manifests say. - const thumbs = await thumbView(id, thumbAliasesFor(id)); - - return ( - <div className="flex h-full flex-col"> - <BrowseHeader - crumbs={[{ href: "/browse", label: "songs" }, { label: song.id }]} - note={`${song.present}/${song.cuts.length} cuts · ${song.variantCount} variants`} - /> - <div className="flex flex-wrap items-start gap-3 border-b border-[var(--color-line)] px-4 py-2"> - <div className="min-w-0 flex-1"> - {/* The one obvious place to write about the song, open by default -- - everything else on the page is collapsed until asked for. */} - <NoteField song={id} target="song" initial={notes.song ?? null} label="notes on this song" /> - </div> - <CopyButton - label="copy this song" - title="this song as markdown — spec, cuts, verdicts, notes and resolved marks" - url={`/api/browse/context?song=${encodeURIComponent(id)}`} - /> - </div> - <main className="deck-main flex-1"> - <div className="grid gap-4 p-4 xl:grid-cols-[minmax(0,2fr)_minmax(0,1fr)]"> - <div className="space-y-3"> - {song.cuts.map((cut) => ( - <CutCard key={cut.name} song={song} cut={cut} info={info} notes={notes} /> - ))} - - {song.unattributed.length > 0 && ( - <section className="rounded border border-[var(--color-dirty)]/40 bg-[var(--color-panel)] p-3"> - <div className="micro mb-2"> - matches no cut name — attribute by renaming, never by guessing - </div> - <ul className="space-y-1"> - {song.unattributed.map((v) => ( - <li key={v.rel} className="num text-[12px] text-[var(--color-dim)]"> - <span className="font-mono text-[var(--color-text)]">{v.rel}</span>{" "} - {fmtBytes(v.size)} - </li> - ))} - </ul> - </section> - )} - </div> - - <aside className="space-y-4"> - <ThumbBench song={song.id} view={thumbs} notes={notes} /> - <SpecSheet - song={song.id} - initial={spec} - problems={problems} - operations={operationsFor(spec, song)} - plans={song.plans.map((p) => p.name)} - trimSets={Object.values(TRIM_SETS).map((t) => ({ id: t.id, label: t.label }))} - /> - {song.readme && ( - <section className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] p-3"> - <Markdown text={song.readme} /> - </section> - )} - <section className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] p-3"> - <div className="micro mb-2">plans</div> - {song.plans.length === 0 ? ( - <p className="text-[12px] text-[var(--color-dim)]">no plan/ directory</p> - ) : ( - <ul className="space-y-0.5"> - {song.plans.map((p) => ( - <li key={p.name} className="num text-[11px] text-[var(--color-dim)]"> - <span className="flex gap-2"> - <span className="truncate font-mono text-[var(--color-text)]">{p.name}</span> - <span className="ml-auto shrink-0">{fmtBytes(p.size)}</span> - </span> - <NoteField - song={song.id} - target={`plan:${p.name}`} - initial={notes[`plan:${p.name}`] ?? null} - label="what this plan is" - rows={3} - /> - </li> - ))} - </ul> - )} - {song.hasClipsCsv && <div className="micro mt-2">clips.csv present</div>} - </section> - </aside> - </div> - </main> - </div> - ); -} - -function CutCard({ - song, - cut, - info, - notes, -}: { - song: Song; - cut: Cut; - info: Record<string, { duration: number; width: number; height: number }>; - notes: NoteMap; -}) { - const shipped = cut.shipped; - const live = cut.variants.filter((v) => !v.retired); - const retired = cut.variants.filter((v) => v.retired); - - return ( - <section - data-cut={cut.name} - data-present={shipped ? "1" : "0"} - className="rounded border border-[var(--color-line)] bg-[var(--color-panel)]" - > - <div className="flex items-start gap-3 p-3"> - {shipped ? ( - /* eslint-disable-next-line @next/next/no-img-element */ - <img - src={`/api/browse/poster?song=${encodeURIComponent(song.id)}&rel=${encodeURIComponent(shipped.rel)}&w=320`} - alt="" - width={160} - height={90} - className="w-40 shrink-0 rounded bg-[var(--color-panel-2)] object-cover" - /> - ) : ( - <div className="flex h-[90px] w-40 shrink-0 items-center justify-center rounded border border-dashed border-[var(--color-line)] text-[11px] text-[var(--color-dim)]"> - not built - </div> - )} - - <div className="min-w-0 flex-1 space-y-1"> - <div className="flex flex-wrap items-baseline gap-2"> - <Link - href={`/browse/${song.id}/${cut.name}`} - className="font-mono text-[13px] text-[var(--color-text)] hover:text-[var(--color-sel)]" - > - {cut.name} - </Link> - {shipped ? ( - <span className="num text-[11px] text-[var(--color-meter)]"> - {fmtDur(info[shipped.rel]?.duration ?? 0)} - </span> - ) : ( - <span className="text-[11px] text-[var(--color-dim)]">— a hole in the set</span> - )} - {shipped && info[shipped.rel] && ( - <span className="num text-[11px] text-[var(--color-dim)]"> - {info[shipped.rel].width}×{info[shipped.rel].height} - </span> - )} - <span className="num ml-auto text-[11px] text-[var(--color-dim)]"> - {shipped ? `${fmtBytes(shipped.size)} · ${fmtAgo(shipped.mtimeMs)}` : ""} - </span> - </div> - - {shipped && ( - <VerdictChip song={song.id} rel={shipped.rel} initial={shipped.verdict} /> - )} - - {/* Two different notes, deliberately. The `cut:` one is about the SLOT - -- it survives a promote and can be written about a cut that has - not been built. The `file:` one is about these bytes. */} - <NoteField - song={song.id} - target={`cut:${cut.name}`} - initial={notes[`cut:${cut.name}`] ?? null} - label={`notes on ${cut.name}`} - rows={3} - /> - {shipped && ( - <NoteField - song={song.id} - target={`file:${shipped.rel}`} - initial={notes[`file:${shipped.rel}`] ?? null} - label="notes on this file" - rows={3} - /> - )} - - {live.length > 0 && ( - <ul className="space-y-1 pt-1"> - {live.map((v) => ( - <li - key={v.rel} - data-variant={v.rel} - data-variant-tag={v.tag} - className="flex flex-wrap items-center gap-2 rounded bg-[var(--color-panel-2)] px-2 py-1" - > - <span className="font-mono text-[12px] text-[var(--color-text)]">{v.tag}</span> - <span className="num text-[11px] text-[var(--color-meter)]"> - {fmtDur(info[v.rel]?.duration ?? 0)} - </span> - <span className="num text-[11px] text-[var(--color-dim)]">{fmtBytes(v.size)}</span> - <div className="ml-auto"> - <VerdictChip song={song.id} rel={v.rel} initial={v.verdict} /> - </div> - <div className="w-full"> - <NoteField - song={song.id} - target={`file:${v.rel}`} - initial={notes[`file:${v.rel}`] ?? null} - label={`notes on ${v.tag}`} - rows={3} - /> - </div> - </li> - ))} - </ul> - )} - - {retired.length > 0 && ( - <div className="micro pt-1" data-retired={retired.length}> - {retired.length} retired: {retired.map((v) => v.tag).join(", ")} - </div> - )} - </div> - </div> - </section> - ); -} diff --git a/umtool/app/browse/at/page.tsx b/umtool/app/browse/at/page.tsx @@ -0,0 +1,53 @@ +import Link from "next/link"; +import { notFound } from "next/navigation"; +import ProjectView from "@/components/projects/ProjectView"; +import { projectRefs } from "@/lib/projects"; + +export const dynamic = "force-dynamic"; + +// --------------------------------------------------------------------------- +// The escape hatch for a project that cannot be reached at its own URL. +// +// Two ways that happens, and both used to fail silently. A project whose first +// path segment is a tool page name -- `find`, `trim`, `decisions` -- can never +// win the route, because a static segment beats a dynamic one; it was listed, +// linked, and the link rendered the phrase console. And a project whose name +// isSegment() dislikes (a space is enough) simply vanished from the listing. +// +// `?path=` is validated by MEMBERSHIP of the current scan, not by inspecting the +// string: it either is one of the projects the walk found or it is a 404. That +// is the same rule every other name that crosses the wire in this app follows, +// and it is why this page cannot be turned into a file reader. +// --------------------------------------------------------------------------- + +export default async function AtPage({ + searchParams, +}: { + searchParams: Promise<Record<string, string | undefined>>; +}) { + const search = await searchParams; + const wanted = search.path; + const project = wanted ? (await projectRefs()).find((p) => p.id === wanted) : null; + if (!project) notFound(); + + const why = + project.routing === "shadowed" + ? `/browse/${project.id.split("/")[0]} is a tool page, so this project can never be opened at its own URL. Rename the directory, or work here.` + : project.routing === "unroutable" + ? "This directory's name cannot be a URL segment, so it has no address of its own. Rename it to letters, digits, dots, dashes and underscores to give it one." + : null; + + return ( + <> + {why && ( + <div className="border-b border-[var(--color-bad)] bg-[color-mix(in_srgb,var(--color-bad)_10%,transparent)] px-4 py-2 text-[12px] text-[var(--color-bad)]"> + {why}{" "} + <Link href="/browse" className="underline"> + every project + </Link> + </div> + )} + <ProjectView project={project} rest={[]} search={search} /> + </> + ); +} diff --git a/umtool/app/browse/decisions/page.tsx b/umtool/app/browse/decisions/page.tsx @@ -1,14 +1,15 @@ import Link from "next/link"; +import { badgeVariants, type BadgeVariants } from "@/components/ui/badge"; import BrowseHeader from "@/components/BrowseHeader"; import CopyButton from "@/components/CopyButton"; import { SEVERITIES, countBySeverity, openCount, - openDecisions, type Decision, type Severity, } from "@/lib/decisions"; +import { openDecisions } from "@/lib/projects"; import { fmtAgo } from "@/lib/format"; export const dynamic = "force-dynamic"; @@ -22,10 +23,13 @@ export const dynamic = "force-dynamic"; // Zero client JS. The filters are links that change searchParams; nothing here // changes without a navigation, and the list is short by construction. -const TONE: Record<Severity, string> = { - blocking: "text-[var(--color-bad)] border-[var(--color-bad)]", - open: "text-[var(--color-dirty)] border-[var(--color-dirty)]", - info: "text-[var(--color-dim)] border-[var(--color-line)]", +// The three severities ARE three badge variants, one for one. That is not a +// coincidence to be tidied away: the palette reserves its hues for verdicts, +// and a severity is the closest thing this list has to one. +const TONE: Record<Severity, BadgeVariants["variant"]> = { + blocking: "blocking", + open: "open", + info: "info", }; export default async function DecisionsPage({ @@ -136,9 +140,7 @@ export default async function DecisionsPage({ data-target={d.target} className="flex flex-wrap items-baseline gap-2 rounded border border-[var(--color-line)] bg-[var(--color-panel)] px-3 py-1.5" > - <span - className={`rounded border px-1.5 py-0.5 text-[10px] uppercase tracking-wider ${TONE[d.severity]}`} - > + <span className={badgeVariants({ variant: TONE[d.severity], size: "sm" })}> {d.severity} </span> <Link @@ -173,17 +175,16 @@ function Chip({ href: string; on: boolean; label: string; - tone?: string; + tone?: BadgeVariants["variant"]; }) { return ( <Link href={href} + // Set by hand, not by the variant. It is what the e2e suite asserts on, + // and it is the actual accessibility statement -- the colour is only the + // visible half of it. aria-current={on ? "true" : undefined} - className={`rounded border px-1.5 py-0.5 font-mono text-[11px] ${ - on - ? "border-[var(--color-sel)] text-[var(--color-sel)]" - : (tone ?? "border-[var(--color-line)] text-[var(--color-dim)] hover:text-[var(--color-text)]") - }`} + className={badgeVariants({ variant: on ? "on" : (tone ?? "neutral") })} > {label} </Link> diff --git a/umtool/app/browse/page.tsx b/umtool/app/browse/page.tsx @@ -1,35 +1,184 @@ import Link from "next/link"; +import { badgeVariants, type BadgeVariants } from "@/components/ui/badge"; import BrowseHeader from "@/components/BrowseHeader"; import NewSongForm from "@/components/NewSongForm"; import CopyButton from "@/components/CopyButton"; -import { CUT_NAMES, listSongs } from "@/lib/browse"; -import { fmtAgo } from "@/lib/format"; +import ProjectGrid from "@/components/projects/ProjectGrid"; +import { KINDS, decisionCounts, indexHealth, listFolders, listProjects } from "@/lib/projects"; +import { PROJECT_STATES } from "@/lib/project-types"; export const dynamic = "force-dynamic"; // A STATIC segment, so it wins over app/[mode]/page.tsx -- which would // otherwise catch /browse, fail isMode(), and 404. Same trick as /mix. // -// Zero client JS on this level. It is a list of five cards; every poster is an -// <img> the browser fetches on its own, and nothing here changes without a -// navigation. +// ZERO CLIENT JS, still. The four filters are links that change searchParams, +// exactly as /browse/decisions does it, and `?q=` is a plain GET form carrying +// the others as hidden inputs -- the /browse/find idiom -- so any filtered view +// is one pasteable URL. Nothing here hydrates. +// +// Counts on the chips come from the UNFILTERED set on purpose. A chip whose +// number changes when you click a different chip moves under the cursor, and +// the whole point of a filter row is to say how much is behind each one. + +type Search = { + kind?: string; + template?: string; + state?: string; + open?: string; + q?: string; + sort?: string; +}; + +export default async function BrowsePage({ + searchParams, +}: { + searchParams: Promise<Search>; +}) { + const sp = await searchParams; + const { kind, template, state, open, q, sort } = sp; + + const all = await listProjects(); + const counts = await decisionCounts(); + const folders = await listFolders(); + + const needle = (q ?? "").trim().toLowerCase(); + const match = (p: (typeof all)[number]) => { + const c = counts.get(p.id); + return ( + (!kind || p.kind === kind) && + (!template || p.template === template) && + (!state || p.state === state) && + (!open || + (open === "blocking" ? (c?.blocking ?? 0) > 0 : (c?.blocking ?? 0) + (c?.open ?? 0) > 0)) && + (!needle || p.haystack.includes(needle)) + ); + }; + + const items = all.filter(match); + const sorted = + sort === "name" ? [...items].sort((a, b) => a.id.localeCompare(b.id)) : items; + + const nOf = (pred: (p: (typeof all)[number]) => boolean) => all.filter(pred).length; + + const qs = (next: Partial<Search>) => { + const p = new URLSearchParams(); + for (const [k, v] of Object.entries({ ...sp, ...next })) if (v) p.set(k, String(v)); + const s = p.toString(); + return `/browse${s ? `?${s}` : ""}`; + }; -export default async function BrowsePage() { - const songs = await listSongs(); + // Group by folder, using the COLLAPSED labels but never collapsed URLs. + const groups = new Map<string, typeof sorted>(); + for (const p of sorted) { + // Which displayed folder owns this project: the deepest collapsed node + // whose path is a prefix of the project's id. + let owner = ""; + for (const f of folders.keys()) { + if (f && (p.id === f || p.id.startsWith(`${f}/`)) && f.length > owner.length) owner = f; + } + const list = groups.get(owner); + if (list) list.push(p); + else groups.set(owner, [p]); + } + + const blocking = [...counts.values()].reduce((n, c) => n + c.blocking, 0); + const openN = [...counts.values()].reduce((n, c) => n + c.open, 0); + + // The index's only visible surface. It self-heals silently, so its health has + // to be said deliberately or it cannot be observed at all. + const ix = indexHealth(); return ( <div className="flex h-full flex-col"> <BrowseHeader - crumbs={[{ label: "songs" }]} - note={`${songs.length} songs · ${CUT_NAMES.length} cuts each`} + crumbs={[{ label: "projects" }]} + note={`${all.length} projects · ${blocking} blocking · ${openN} open`} /> <main className="deck-main flex-1 p-4"> + {/* --- filters, as links ---------------------------------------- */} + <div className="mb-2 flex flex-wrap items-center gap-1.5"> + <span className="micro">kind</span> + <Chip href={qs({ kind: "", template: "" })} on={!kind && !template} label={`all ${all.length}`} /> + {KINDS.map((k) => ( + <Chip + key={k.id} + href={qs({ kind: k.id, template: "" })} + on={kind === k.id && !template} + label={`${k.label} ${nOf((p) => p.kind === k.id)}`} + /> + ))} + </div> + + <div className="mb-2 flex flex-wrap items-center gap-1.5"> + <span className="micro">state</span> + <Chip href={qs({ state: "" })} on={!state} label="any" /> + {PROJECT_STATES.filter((s) => nOf((p) => p.state === s) > 0).map((s) => ( + <Chip + key={s} + href={qs({ state: s })} + on={state === s} + label={`${s} ${nOf((p) => p.state === s)}`} + /> + ))} + + <span className="micro ml-3">decisions</span> + <Chip href={qs({ open: "" })} on={!open} label="any" /> + <Chip + href={qs({ open: "blocking" })} + on={open === "blocking"} + label={`blocking ${nOf((p) => (counts.get(p.id)?.blocking ?? 0) > 0)}`} + tone="blocking" + /> + <Chip + href={qs({ open: "1" })} + on={open === "1"} + label={`waiting ${nOf( + (p) => (counts.get(p.id)?.blocking ?? 0) + (counts.get(p.id)?.open ?? 0) > 0, + )}`} + /> + + <span className="micro ml-3">sort</span> + <Chip href={qs({ sort: "" })} on={sort !== "name"} label="recent" /> + <Chip href={qs({ sort: "name" })} on={sort === "name"} label="name" /> + </div> + <div className="mb-3 flex flex-wrap items-center gap-2"> + {/* A GET form so the result is a pasteable URL, and hidden inputs so + searching does not throw away the filters you already set. */} + <form method="get" action="/browse" className="flex items-center gap-1.5"> + {(["kind", "template", "state", "open", "sort"] as const).map((k) => + sp[k] ? <input key={k} type="hidden" name={k} value={sp[k]} /> : null, + )} + <input + type="search" + name="q" + defaultValue={q ?? ""} + placeholder="title, id, source…" + aria-label="filter projects" + className="w-56 rounded border border-[var(--color-line)] bg-[var(--color-ink)] px-2 py-1 font-mono text-[12px] text-[var(--color-text)] placeholder:text-[var(--color-dim)]" + /> + <button + type="submit" + className="rounded border border-[var(--color-line)] px-2 py-1 text-[11px] text-[var(--color-dim)] hover:text-[var(--color-text)]" + > + filter + </button> + {q && ( + <Link href={qs({ q: "" })} className="micro hover:text-[var(--color-text)]"> + clear + </Link> + )} + </form> + <NewSongForm /> <div className="ml-auto flex items-center gap-2"> - {/* One paste that describes the whole project: every song, what - ships, every judgement with its reason, every mark resolved to - its source moment. */} + <Link + href="/browse/decisions" + className="rounded border border-[var(--color-line)] px-2.5 py-1 text-[12px] text-[var(--color-dim)] hover:text-[var(--color-text)]" + > + decisions + </Link> <CopyButton label="copy the whole picture" title="every song as markdown — spec, cuts, verdicts, notes and resolved marks" @@ -38,69 +187,60 @@ export default async function BrowsePage() { /> </div> </div> - {songs.length === 0 ? ( + + {all.length === 0 ? ( <p className="text-[12px] text-[var(--color-dim)]"> - Nothing under <code className="font-mono">videos/</code>. Point{" "} - <code className="font-mono">SONG_REPORTS_DIR</code> at a tree of songs. + No projects under <code className="font-mono">REPORTS_DIR</code>. A directory becomes + one by holding a <code className="font-mono">video.manifest.json</code>, a{" "} + <code className="font-mono">spec.json</code>, or a sweep report. </p> ) : ( - <ul className="grid gap-3 sm:grid-cols-2 lg:grid-cols-3 2xl:grid-cols-4"> - {songs.map((s) => ( - <li key={s.id}> - <Link - href={`/browse/${s.id}`} - data-song={s.id} - className="block overflow-hidden rounded border border-[var(--color-line)] bg-[var(--color-panel)] transition-colors hover:border-[var(--color-sel)]" - > - {/* eslint-disable-next-line @next/next/no-img-element */} - <img - src={`/api/browse/poster?song=${encodeURIComponent(s.id)}&w=640`} - alt="" - width={640} - height={360} - className="aspect-video w-full bg-[var(--color-panel-2)] object-cover" - /> - <div className="space-y-1.5 p-3"> - <div className="truncate text-[13px] font-medium text-[var(--color-text)]"> - {s.title} - </div> - <div className="num flex flex-wrap items-center gap-x-3 gap-y-1 text-[11px] text-[var(--color-dim)]"> - <span data-cuts={s.present}> - <span className="text-[var(--color-meter)]">{s.present}</span>/ - {CUT_NAMES.length} cuts - </span> - <span data-variants={s.variantCount}> - <span className="text-[var(--color-meter)]">{s.variantCount}</span> variants - </span> - <span className="ml-auto">{fmtAgo(s.newestMtimeMs)}</span> - </div> - {/* The verdict tally is the only place these three hues - appear outside a verdict control. Undecided is - deliberately NOT coloured -- it is the absence of a - judgement, not a third one. */} - <div className="num flex items-center gap-2 text-[11px]"> - <span className="text-[var(--color-good)]" data-keep={s.counts.keep}> - {s.counts.keep} keep - </span> - <span className="text-[var(--color-bad)]" data-reject={s.counts.reject}> - {s.counts.reject} reject - </span> - <span className="text-[var(--color-dim)]" data-undecided={s.counts.undecided}> - {s.counts.undecided} undecided - </span> - </div> - {s.missing.length > 0 && ( - <div className="micro" data-missing={s.missing.join(",")}> - no {s.missing.join(", ")} - </div> - )} - </div> - </Link> - </li> + <div className="space-y-5"> + {[...groups.entries()].map(([folderPath, list]) => ( + <section key={folderPath || "_root"} data-folder={folderPath}> + {folderPath && ( + <h2 className="micro mb-1.5"> + {folders.get(folderPath)?.label ?? folderPath} + </h2> + )} + <ProjectGrid projects={list} counts={counts} /> + </section> ))} - </ul> + </div> + )} + + {ix.ok && ix.total > 0 && ( + <p className="micro mt-6" data-index={`${ix.fresh}/${ix.total}`}> + index: {ix.fresh}/{ix.total} fresh — the filesystem is the model; this only + caches what it said + </p> )} </main> </div> ); } + +function Chip({ + href, + on, + label, + tone, +}: { + href: string; + on: boolean; + label: string; + tone?: BadgeVariants["variant"]; +}) { + return ( + <Link + href={href} + // Set by hand, not by the variant. It is what the e2e suite asserts on, + // and it is the actual accessibility statement -- the colour is only the + // visible half of it. + aria-current={on ? "true" : undefined} + className={badgeVariants({ variant: on ? "on" : (tone ?? "neutral") })} + > + {label} + </Link> + ); +} diff --git a/umtool/app/globals.css b/umtool/app/globals.css @@ -1,4 +1,19 @@ -@import "tailwindcss"; +/* Tailwind v4 auto-detects its sources, and that is too broad here. + * + * It honours .gitignore but scans everything else -- including MARKDOWN. This + * file's own documentation (umtool/docs/browse.md) described the failure mode + * by quoting the corrupt class name it produces, Tailwind extracted that quote + * as a candidate, and every page 500d on a CSS parse error again. Documenting + * the trap re-created the trap. + * + * So detection is turned off and the three directories that actually hold + * class names are named. This makes the whole class of problem impossible: a + * scratch dist dir, a test artefact, a fixture or a doc can no longer poison + * the stylesheet. */ +@import "tailwindcss" source(none); +@source "../app"; +@source "../components"; +@source "../lib"; /* A dark, dense, keyboard-first bench. Nothing here is deployed; the only user is someone judging thousands of clips, or auditioning one transition, in a @@ -35,6 +50,44 @@ /* measurement ONLY -- the pitch rail, and every number that is a reading */ --color-meter: #56d4c4; + + /* --------------------------------------------------------------------- + shadcn/ui's token names, MAPPED ONTO THE PALETTE ABOVE. Not imported. + + Its default theme ships `primary`, `accent` and `destructive` as their own + hues. Adopting those would walk straight back into the rule this file + opens with: a second green marked the active nav link while competing with + --color-good two inches away, so a navigation state wore a verdict's + clothes, and --color-accent was DELETED rather than retuned. + + So every shadcn name is an alias for a colour that already means something + here, and `accent` is mapped to a SURFACE rather than to a hue -- shadcn + uses it for hover backgrounds, which is a surface change, not a statement. + No new colour enters the app, and `cn()` + cva give us the variants and the + class-merging without the theme. + --------------------------------------------------------------------- */ + --color-background: var(--color-ink); + --color-foreground: var(--color-text); + --color-card: var(--color-panel); + --color-card-foreground: var(--color-text); + --color-popover: var(--color-panel); + --color-popover-foreground: var(--color-text); + --color-muted: var(--color-panel-2); + --color-muted-foreground: var(--color-dim); + /* A surface, never a hue. */ + --color-accent: var(--color-panel-2); + --color-accent-foreground: var(--color-text); + --color-secondary: var(--color-panel-2); + --color-secondary-foreground: var(--color-text); + --color-border: var(--color-line); + --color-input: var(--color-line); + --color-primary: var(--color-sel); + --color-primary-foreground: var(--color-ink); + --color-ring: var(--color-sel); + --color-destructive: var(--color-bad); + --color-destructive-foreground: var(--color-ink); + + --radius: 4px; } html, diff --git a/umtool/app/mix/page.tsx b/umtool/app/mix/page.tsx @@ -1,25 +1,64 @@ +import Link from "next/link"; import AppNav from "@/components/AppNav"; import MixBench from "@/components/MixBench"; import { MEDIA_ROOTS } from "@/lib/paths"; +import { isRefusal, resolvePreset } from "@/lib/mix-preset"; export const dynamic = "force-dynamic"; // A STATIC segment, so it wins over app/[mode]/page.tsx -- which would // otherwise catch /mix, fail isMode(), and 404. -export default function MixPage() { +// +// The deep link (?body=&start=&end=&from=&clip=) is resolved HERE, server-side. +// That is what lets a clip reach the bench with no useSearchParams and no +// Suspense boundary: this page already ran on the server, it just never read its +// own searchParams. +export default async function MixPage({ + searchParams, +}: { + searchParams: Promise<{ body?: string; start?: string; end?: string; from?: string; clip?: string }>; +}) { + const sp = await searchParams; + const resolved = await resolvePreset(sp); + const refused = isRefusal(resolved) ? resolved : null; + const preset = isRefusal(resolved) ? null : resolved; + return ( <div className="flex h-full flex-col"> <header className="flex flex-wrap items-center gap-3 border-b border-[var(--color-line)] bg-[var(--color-panel)] px-4 py-2"> <AppNav active="mix" /> - <div className="text-[12px] text-[var(--color-dim)]"> - set the handover and the end point, hear it, then render it - </div> + {preset?.fromHref ? ( + <Link + href={preset.fromHref} + className="text-[12px] text-[var(--color-sel)] hover:underline" + data-mix-from={preset.from ?? ""} + > + ← {preset.from} + {preset.clip ? ` / ${preset.clip}` : ""} + </Link> + ) : ( + <div className="text-[12px] text-[var(--color-dim)]"> + set the handover and the end point, hear it, then render it + </div> + )} <div className="num ml-auto hidden text-[10px] text-[var(--color-dim)] lg:block"> {MEDIA_ROOTS.length} roots </div> </header> + + {refused && ( + // Refused, not clamped. A link that quietly opened a DIFFERENT file than + // it named would be worse than this message. + <p + data-mix-refused="" + className="border-b border-[var(--color-bad)] bg-[color-mix(in_srgb,var(--color-bad)_10%,transparent)] px-4 py-2 text-[12px] text-[var(--color-bad)]" + > + that link was refused — {refused.reason} + </p> + )} + <main className="deck-main flex-1"> - <MixBench /> + <MixBench preset={preset} /> </main> </div> ); diff --git a/umtool/bin/umtool.mjs b/umtool/bin/umtool.mjs @@ -0,0 +1,594 @@ +#!/usr/bin/env node +// umtool — the project tree, from a terminal. +// +// The audience is an AI assistant working in this repo, which is why every +// command takes --json and why `check` exits non-zero. It reads the SAME +// lib/projects/*.mjs the app does, so `umtool ls` and /browse cannot disagree +// about what a project is, and `umtool check` and the decisions inbox cannot +// disagree about what is wrong with one. +// +// Honouring REPORTS_DIR / SONG_REPORTS_DIR / SONG_DIR / CHANNELS_DIR means it +// can be pointed at the e2e fixture, which is how it is tested. +// +// umtool ls [--kind K] [--template T] [--state S] [--open] [--blocking] +// [--q TEXT] [--sort name|recent] [--json] +// umtool show <project> [--json] +// umtool check [<project> | --all] [--json] exit 1 on anything blocking +// umtool decisions [--json] +// umtool folders [--json] +// umtool kinds [--json] +// umtool window <project> <clip> [--start S] [--end E] [--lock] [--lock-end] ... +// umtool build <project> [--preset preview|fast|final] [--only ID] [--dry] +// umtool index [--rebuild] [--prune] [--since MS] [--json] +// umtool new <slug> [--kind report-video] [--from <sweep-report.md>] +import process from "node:process"; +import { + PROJECT_KINDS, + REPORTS_ROOT, + decisionsAreComplete, + decisionsFor, + folders, + projectRefs, + resolveProject, + summarise, +} from "../lib/projects/core.mjs"; +import { readClipDetail, readManifest } from "../lib/projects/report.mjs"; +import path from "node:path"; +import { mkdir, readFile, writeFile, stat } from "node:fs/promises"; +import { updateClip } from "../lib/report/manifest.mjs"; +import { buildSteps, PRESETS } from "../lib/report/driver.mjs"; +import { openIndex, signRecord } from "../lib/projects/index-db.mjs"; + +const argv = process.argv.slice(2); +const cmd = argv.find((a) => !a.startsWith("-")) ?? "help"; +const rest = argv.filter((a) => a !== cmd); +const has = (n) => rest.includes(n) || argv.includes(n); +const val = (n) => { + const i = argv.indexOf(n); + return i >= 0 ? argv[i + 1] : undefined; +}; +const json = has("--json"); +const positional = argv.filter((a, i) => { + if (a.startsWith("-")) return false; + if (a === cmd && argv.indexOf(a) === argv.indexOf(cmd)) return false; + // A value that belongs to the flag before it is not a positional. + return !(i > 0 && argv[i - 1].startsWith("--")); +}); + +const out = (v) => console.log(json ? JSON.stringify(v, null, 2) : v); +const die = (msg, code = 2) => { + console.error(msg); + process.exit(code); +}; + +const RANK = { blocking: 0, open: 1, info: 2 }; +const sortDecisions = (ds) => + [...ds].sort((a, b) => RANK[a.severity] - RANK[b.severity] || a.project.localeCompare(b.project)); + +const ago = (ms) => { + if (!ms) return "—"; + const s = Math.max(0, (Date.now() - ms) / 1000); + if (s < 90) return `${Math.round(s)}s`; + if (s < 5400) return `${Math.round(s / 60)}m`; + if (s < 129600) return `${Math.round(s / 3600)}h`; + return `${Math.round(s / 86400)}d`; +}; + +async function summaries() { + const refs = await projectRefs(); + return Promise.all(refs.map((p) => summarise(p))); +} + +async function allDecisions() { + const refs = await projectRefs(); + const per = await Promise.all(refs.map((p) => decisionsFor(p))); + return sortDecisions(per.flat()); +} + +// --------------------------------------------------------------------------- + +async function cmdLs() { + let items = await summaries(); + const counts = new Map(); + const refs = await projectRefs(); + for (const p of refs) { + const ds = await decisionsFor(p); + counts.set(p.id, { + blocking: ds.filter((d) => d.severity === "blocking").length, + open: ds.filter((d) => d.severity === "open").length, + complete: decisionsAreComplete(p.kind), + }); + } + + const kind = val("--kind"); + const template = val("--template"); + const state = val("--state"); + const q = (val("--q") ?? "").trim().toLowerCase(); + items = items.filter( + (p) => + (!kind || p.kind === kind) && + (!template || p.template === template) && + (!state || p.state === state) && + (!q || p.haystack.includes(q)) && + (!has("--blocking") || (counts.get(p.id)?.blocking ?? 0) > 0) && + (!has("--open") || + (counts.get(p.id)?.blocking ?? 0) + (counts.get(p.id)?.open ?? 0) > 0), + ); + items.sort( + val("--sort") === "name" + ? (a, b) => a.id.localeCompare(b.id) + : (a, b) => b.newestMtimeMs - a.newestMtimeMs || a.id.localeCompare(b.id), + ); + + if (json) return out(items.map((p) => ({ ...p, decisions: counts.get(p.id) }))); + + if (!items.length) return console.log("no projects match"); + const w = Math.max(...items.map((p) => p.id.length)); + for (const p of items) { + const c = counts.get(p.id); + const marks = [ + c.blocking ? `${c.blocking} blocking` : "", + c.open ? `${c.open} open` : "", + ...p.flags, + ].filter(Boolean); + console.log( + `${p.id.padEnd(w)} ${p.badge.padEnd(6)} ${p.state.padEnd(8)} ${ago(p.newestMtimeMs).padStart(4)} ` + + `${p.facts.join(" · ")}${marks.length ? ` [${marks.join(" · ")}]` : ""}`, + ); + } + console.log(`\n${items.length} project(s) under ${REPORTS_ROOT}`); +} + +async function pick(arg) { + if (!arg) die("which project? pass an id, a name, or a directory"); + const r = await resolveProject(arg); + if (r.ambiguous) { + die( + `"${arg}" is the name of ${r.ambiguous.length} projects — say which:\n` + + r.ambiguous.map((p) => ` ${p.id}`).join("\n"), + ); + } + if (!r.project) die(`no project matches "${arg}"`); + return r.project; +} + +async function cmdShow() { + const p = await pick(positional[0]); + const s = await summarise(p); + const ds = await decisionsFor(p); + const detail = p.kind === "report-video" ? await readClipDetail(p.dir) : null; + + if (json) return out({ ...s, decisions: ds, entries: detail?.entries ?? null }); + + console.log(`${s.title}`); + console.log(`${p.id} [${p.kind} · ${p.template}] ${s.state}`); + if (s.subtitle) console.log(s.subtitle); + console.log(`${p.dir}`); + if (s.facts.length) console.log(`\n${s.facts.join(" · ")}`); + if (s.flags.length) console.log(`FLAGS: ${s.flags.join(" · ")}`); + + if (detail) { + console.log(`\ncues from ${detail.channelsDir}${detail.shadowExists ? " (this project's shadow tree)" : ""}`); + console.log("\nthe cut:"); + for (const e of detail.entries) { + // A CLIP is `type === "clip"`. Everything else -- a card, and the `scroll` + // and `chart` entries one real manifest carries -- has no window and no + // source, and the vocabulary is open, so it is printed generically rather + // than assumed to be one of two things. + if (e.kind !== "clip") { + const label = e.heading ?? e.title ?? e.label ?? ""; + const secs = e.seconds != null ? `${e.seconds}s` : ""; + console.log(` ${e.id.padEnd(5)} ${String(e.kind).padEnd(9)} ${secs.padStart(4)} ${label}`); + continue; + } + const marks = [ + e.lock ? "locked" : "", + !e.lock && e.lockStart ? "start pinned" : "", + !e.lock && e.lockEnd ? "end pinned" : "", + e.cached ? "cached" : "NOT FETCHED", + e.segment ? "segment" : "", + e.hasCues ? "" : "NO CUES", + e.endsSentence === false && !e.lockEnd && !e.lock ? "ends mid-sentence" : "", + e.noPunctuation ? "source unpunctuated" : "", + e.proposed ? `widen -> ${e.proposed.start}–${e.proposed.end}` : "", + ].filter(Boolean); + console.log( + ` ${e.id.padEnd(5)} ${String(e.video).padEnd(14)} ` + + `${e.start.toFixed(2)}–${e.end.toFixed(2)} (${(e.end - e.start).toFixed(1)}s)` + + `${marks.length ? ` [${marks.join(" · ")}]` : ""}`, + ); + } + } + + if (ds.length) { + console.log(""); + for (const d of sortDecisions(ds)) { + console.log(` ${d.severity.toUpperCase().padEnd(8)} ${d.kind} ${d.target} — ${d.why}`); + } + } + if (!decisionsAreComplete(p.kind)) { + console.log(`\n(this kind's decisions are computed by the app — see /browse/decisions)`); + } +} + +async function cmdCheck() { + const one = positional[0]; + const refs = one ? [await pick(one)] : await projectRefs(); + const rows = []; + for (const p of refs) rows.push(...(await decisionsFor(p))); + const sorted = sortDecisions(rows); + const blocking = sorted.filter((d) => d.severity === "blocking"); + + if (json) { + out({ ok: blocking.length === 0, blocking: blocking.length, decisions: sorted }); + } else { + for (const d of sorted) { + if (d.severity === "info" && !has("--all-severities")) continue; + console.log(`${d.severity.toUpperCase().padEnd(8)} ${d.project} ${d.kind} ${d.target}`); + console.log(` ${d.why}`); + } + const partial = refs.filter((p) => !decisionsAreComplete(p.kind)).length; + console.log( + `\n${refs.length} project(s), ${blocking.length} blocking, ` + + `${sorted.filter((d) => d.severity === "open").length} open`, + ); + if (partial) { + console.log( + `(${partial} of them are kinds whose decisions the app computes — this checked their routing and nothing else)`, + ); + } + } + // The whole point: a build script can gate on this. + process.exit(blocking.length ? 1 : 0); +} + +async function cmdDecisions() { + const ds = await allDecisions(); + if (json) return out(ds); + let last = ""; + for (const d of ds) { + if (d.project !== last) { + console.log(`\n${d.project}`); + last = d.project; + } + console.log(` ${d.severity.toUpperCase().padEnd(8)} ${d.kind} ${d.target} — ${d.why}`); + } + console.log(`\n${ds.filter((d) => d.severity !== "info").length} waiting of ${ds.length}`); +} + +async function cmdFolders() { + const f = await folders(); + if (json) return out([...f.values()]); + for (const n of f.values()) { + if (!n.path) continue; + console.log(`${n.path} "${n.label}" ${n.projects.length} project(s)`); + } +} + +function cmdKinds() { + const meta = PROJECT_KINDS.map((k) => ({ + id: k.id, + template: k.template, + label: k.label, + badge: k.badge, + decisionKinds: k.decisionKinds, + hasDecisions: !!k.decisions, + })); + if (json) return out(meta); + for (const k of meta) { + console.log(`${k.id.padEnd(14)} ${k.template.padEnd(16)} ${k.label}`); + if (k.decisionKinds.length) console.log(` decisions: ${k.decisionKinds.join(", ")}`); + } +} + +function usage() { + console.log( + [ + "umtool — the project tree, from a terminal", + "", + " umtool ls [--kind K] [--template T] [--state S] [--open] [--blocking]", + " [--q TEXT] [--sort name|recent] [--json]", + " umtool show <project> [--json]", + " umtool check [<project>] [--json] exit 1 on anything blocking", + " umtool decisions [--json]", + " umtool folders [--json]", + " umtool kinds [--json]", + " umtool window <project> <clip> [--start S] [--end E] [--lock|--lock-end|…]", + " umtool build <project> [--preset preview|fast|final] [--only ID]", + " umtool index [--rebuild] [--prune] [--since MS] [--json]", + " umtool new <slug> [--kind report-video] [--from <sweep-report.md>]", + "", + `reading ${REPORTS_ROOT} (set REPORTS_DIR to move it)`, + "", + "Run `check` before every build. It is what catches a manifest with no", + "siteOrigin — the defect that shipped 19 QR codes reading `undefined/?v=…`.", + ].join("\n"), + ); +} + +const COMMANDS = { + ls: cmdLs, + window: cmdWindow, + build: cmdBuild, + index: cmdIndex, + new: cmdNew, + show: cmdShow, + check: cmdCheck, + decisions: cmdDecisions, + folders: cmdFolders, + kinds: cmdKinds, + help: usage, +}; + +const run = COMMANDS[cmd]; +if (!run) die(`unknown command "${cmd}"\n\nRun \`umtool help\`.`); +await run(); + + +// --------------------------------------------------------------------------- +// Writing. +// --------------------------------------------------------------------------- + +async function cmdWindow() { + const p = await pick(positional[0]); + const clipId = positional[1]; + if (!clipId) die("which clip? `umtool window <project> <clip> --start S --end E`"); + + const patch = {}; + const num = (n) => { + const v = val(n); + return v === undefined ? undefined : Number(v); + }; + if (num("--start") !== undefined) patch.start = num("--start"); + if (num("--end") !== undefined) patch.end = num("--end"); + // A flag and its negation, because `false` REMOVES the key -- the manifests + // are read by humans and `"lockEnd": false` reads like a decision. + for (const [flag, key] of [ + ["--lock", "lock"], + ["--lock-start", "lockStart"], + ["--lock-end", "lockEnd"], + ]) { + if (has(flag)) patch[key] = true; + if (has(`--no-${flag.slice(2)}`)) patch[key] = false; + } + if (val("--note") !== undefined) patch.note = val("--note"); + if (!Object.keys(patch).length) die("nothing to change"); + + try { + // Through the SAME writer the bench uses: 2 dp, the CLI's own formatting, + // tmp+rename, one .bak. A second implementation here is how the two would + // start disagreeing about a window. + const res = await updateClip(p.dir, clipId, patch); + if (json) return out({ ok: true, ...res }); + console.log( + `${clipId}: ${res.before.start}–${res.before.end} -> ${res.entry.start}–${res.entry.end}`, + ); + const marks = ["lock", "lockStart", "lockEnd"].filter((k) => res.entry[k]); + if (marks.length) console.log(` ${marks.join(", ")}`); + console.log(`\nRun resolve-windows to see whether the widener agrees:`); + console.log(` node scripts/report-to-video/resolve-windows.mjs ${p.dir}/video.manifest.json`); + } catch (e) { + die(e?.message ?? String(e)); + } +} + +async function cmdBuild() { + const p = await pick(positional[0]); + const preset = val("--preset") ?? "fast"; + if (!(preset in PRESETS)) die(`--preset must be one of ${Object.keys(PRESETS).join(", ")}`); + const manifest = await readManifest(p.dir); + if (!manifest) die("no manifest"); + const clipCount = (manifest.timeline ?? []).filter((e) => e.type === "clip").length; + + const steps = buildSteps(p, { + preset, + only: val("--only") ?? null, + skipFetch: has("--skip-fetch"), + clipCount, + }); + + if (json) return out({ project: p.id, preset, steps }); + + // Print, never run. The app runs the chain through lib/jobs.ts, which owns the + // cancellation, the timeouts and the process-group kill; a second runner here + // would be a second set of those, and the one that got them right is not this. + console.log(`# ${p.id} — ${PRESETS[preset].label}`); + console.log(`# check first: umtool check ${p.id}\n`); + for (const s of steps) { + console.log(`# ${s.label}${s.timeoutMs ? ` (up to ${Math.round(s.timeoutMs / 60000)}m)` : ""}`); + console.log(`(cd ${s.cwd} && ${s.argv.join(" ")})\n`); + } + if (!has("--dry")) { + console.log("# Nothing was run. This prints the chain; the button on the project page runs it,"); + console.log("# because cancellation and the process-group kill live in the server's job runner."); + } +} + +async function cmdIndex() { + const ix = await openIndex(); + if (!ix.ok) { + if (json) return out({ ok: false, reason: "no index — the filesystem is the model anyway" }); + console.log("no index (that is a normal state: everything still works, just slower)"); + return; + } + + if (has("--prune")) { + const refs = await projectRefs(); + const live = new Set(refs.map((p) => p.id)); + let dropped = 0; + for (const rec of ix.recent(10_000)) { + if (live.has(rec.id)) continue; + ix.del(rec.id); + dropped += 1; + } + if (!json) console.log(`pruned ${dropped} record(s) for projects that are gone`); + } + + if (has("--rebuild")) { + // Re-reads every project and writes the record back. Never needed for + // correctness -- every read verifies its own signature -- but it makes the + // first page load after a big change fast instead of merely correct. + for (const p of await projectRefs()) { + const s = await summarise(p); + const { kindById } = await import("../lib/projects/kinds.mjs"); + const k = kindById(p.kind); + const kindSig = k?.signature ? String(await k.signature(p.dir)) : "0"; + ix.put({ ...s, sig: signRecord({ kindSig }) }); + } + if (!json) console.log("rebuilt"); + } + + const sinceArg = val("--since"); + if (sinceArg !== undefined) { + const rows = ix.since(Number(sinceArg)); + if (json) return out(rows); + for (const r of rows) console.log(`${r.id} ${r.state} ${new Date(r.newestMtimeMs).toISOString()}`); + console.log(`\n${rows.length} project(s) changed since ${new Date(Number(sinceArg)).toISOString()}`); + await ix.close(); + return; + } + + const st = ix.stats(); + if (json) return out(st); + console.log(`${st.records} record(s), schema ${st.schema}`); + console.log(st.path); + console.log(`built ${st.builtAt ? new Date(st.builtAt).toISOString() : "never"}`); + console.log("\nSafe to delete at any time. Every read verifies its own signature"); + console.log("against the filesystem, so a stale record self-heals on the next load."); + await ix.close(); +} + + +// --------------------------------------------------------------------------- +// Scaffolding. +// --------------------------------------------------------------------------- + +/** + * A manifest skeleton, and deliberately an EMPTY timeline. + * + * It would be easy to derive first-draft clips from a report's citations: the + * shape is regular (`> "quote"` then `— [title @ h:mm:ss](…?v=slug%2Fid&t=sec)`). + * It is not done, and that is the honest position rather than a missing feature. + * A report records ONE second per citation; a window needs a start AND an end + * taken from transcript.cues.json, and matching a quote to its cues is the + * actual work of authoring a cut. A generated timeline of guessed windows would + * look finished and be wrong, and every clip would have to be opened anyway. + * + * So this writes what can be known -- the slug, the origin, the channel, the + * render block -- lists the citations it found as a checklist, and says what to + * do next. + */ +function skeleton(slug, title, provenance) { + return { + schemaVersion: 1, + slug, + title, + subtitle: "", + generatedOn: new Date().toISOString().slice(0, 10), + provenance: { + // The field that shipped broken TWICE. It is first, and it is empty rather + // than plausible, so `umtool check` blocks until somebody sets it. + siteOrigin: "", + channelSlug: "", + channel: "", + ...provenance, + }, + render: { + width: 1920, + height: 1080, + fps: 30, + audioRate: 48000, + audioChannels: 2, + maxHeightSource: 1080, + fontRegular: "/usr/share/fonts/TTF/FiraSans-Regular.ttf", + fontBold: "/usr/share/fonts/TTF/FiraSans-Bold.ttf", + palette: { bg: "#12100c", fg: "#f6f1e6", muted: "#a2957f", accent: "#c8752a", amber: "#ffc860" }, + transition: 0.4, + fetchPad: 3, + snapWindow: 1.6, + silenceMinDur: 0.09, + silenceRelDb: 6, + headerHeight: 56, + footerHeight: 0, + crf: 21, + preset: "slow", + qr: { scale: 4, quiet: 3, ecc: "M", margin: 28 }, + }, + timelineNodes: [], + timeline: [], + }; +} + +async function cmdNew() { + const slug = positional[0]; + if (!slug) die("usage: umtool new <slug> [--kind report-video] [--from <sweep-report.md>]"); + if (!/^[a-z0-9][a-z0-9-]*$/.test(slug)) { + die(`"${slug}" will not route — use lower-case letters, digits and dashes`); + } + const kind = val("--kind") ?? "report-video"; + if (kind !== "report-video") die(`only report-video can be scaffolded so far, not ${kind}`); + + const dir = path.join(REPORTS_ROOT, slug); + if (await stat(dir).then(() => true, () => false)) die(`${dir} already exists`); + + let title = slug.replace(/-/g, " "); + const citations = []; + const fromArg = val("--from"); + if (fromArg) { + const text = await readFile(fromArg, "utf8").catch(() => null); + if (text === null) die(`could not read ${fromArg}`); + title = text.match(/^#\s+(.+)$/m)?.[1]?.trim() ?? title; + // `?v=<channel>%2F<id>&t=<sec>` -- the shape the viewer's share links use. + for (const m of text.matchAll(/\]\([^)]*[?&]v=([^&)]+)&t=(\d+)/g)) { + const [chan, id] = decodeURIComponent(m[1]).split("/"); + citations.push({ channel: chan, video: id, second: Number(m[2]) }); + } + } + + const channels = [...new Set(citations.map((c) => c.channel))]; + const doc = skeleton(slug, title, channels.length === 1 ? { channelSlug: channels[0] } : {}); + + await mkdir(dir, { recursive: true }); + await writeFile(path.join(dir, "video.manifest.json"), JSON.stringify(doc, null, 2) + "\n", "utf8"); + + if (fromArg) { + await writeFile( + path.join(dir, "sweep-report.md"), + await readFile(fromArg, "utf8"), + "utf8", + ); + } + + const lines = [ + `# ${title}`, + "", + "What this cut argues, which sources it draws on, and anything cut short on", + "purpose (with why — that is what a `lock` in the manifest means).", + "", + "## Windows still to write", + "", + citations.length + ? "Each of these is ONE second from the report. A clip needs a start AND an" + + " end, read from the source's transcript.cues.json — that matching is the work." + : "No `?v=` citations were found, so there is nothing to work from yet.", + "", + ...citations.map( + (c, i) => `- [ ] c${String(i).padStart(2, "0")} ${c.channel}/${c.video} @ ${c.second}s`, + ), + ]; + await writeFile(path.join(dir, "README.md"), lines.join("\n") + "\n", "utf8"); + + if (json) return out({ ok: true, dir, citations: citations.length, channels }); + + console.log(`${dir}`); + console.log(` video.manifest.json an EMPTY timeline — see README.md`); + if (fromArg) console.log(` sweep-report.md copied from ${fromArg}`); + console.log(` README.md ${citations.length} citation(s) as a checklist`); + console.log(""); + console.log("Next, in order:"); + console.log(` 1. set provenance.siteOrigin — it is EMPTY, and \`check\` blocks until it is not.`); + console.log(` Two finished videos shipped with QR codes that resolve to nothing.`); + console.log(` 2. write the timeline (umtool/docs/authoring.md)`); + console.log(` 3. umtool check ${slug}`); + console.log(` 4. umtool build ${slug} --preset fast`); +} diff --git a/umtool/components.json b/umtool/components.json @@ -0,0 +1,21 @@ +{ + "$schema": "https://ui.shadcn.com/schema.json", + "style": "new-york", + "rsc": true, + "tsx": true, + "tailwind": { + "config": "", + "css": "app/globals.css", + "baseColor": "slate", + "cssVariables": true, + "prefix": "" + }, + "iconLibrary": "lucide", + "aliases": { + "components": "@/components", + "utils": "@/lib/utils", + "ui": "@/components/ui", + "lib": "@/lib", + "hooks": "@/hooks" + } +} diff --git a/umtool/components/MixBench.tsx b/umtool/components/MixBench.tsx @@ -4,6 +4,7 @@ import { useCallback, useEffect, useRef, useState } from "react"; import MixLanes, { type Analysis, type Marker } from "./MixLanes"; type FileRow = { path: string; label: string; size: number; mtimeMs: number }; +type FileGroup = { project: string; label: string; kind: string; files: FileRow[] }; type Info = { path: string; label: string; duration: number; hasVideo: boolean; hasAudio: boolean }; type Track = { info: Info; analysis: Analysis | null }; @@ -59,8 +60,24 @@ export function bgGainAt(t: number, s: Pick<Spec, "handover" | "fade" | "bgGain" return s.bgGain + (s.duck - s.bgGain) * ((t - s.handover) / s.fade); } -export default function MixBench() { - const [files, setFiles] = useState<FileRow[]>([]); +export type MixPresetProp = { + body: string; + label: string; + start: number; + end: number; + duration: number; + from: string | null; + clip: string | null; + fileFrom: number; + fileTo: number; +} | null; + +export default function MixBench({ preset = null }: { preset?: MixPresetProp }) { + const [groups, setGroups] = useState<FileGroup[]>([]); + const [other, setOther] = useState<FileRow[]>([]); + // The only client state the picker adds. A twelve-project tree makes the + // select long enough that scrolling it is worse than typing three letters. + const [filter, setFilter] = useState(""); const [spec, setSpec] = useState<Spec>(BLANK); const [body, setBody] = useState<Track | null>(null); const [bg, setBg] = useState<Track | null>(null); @@ -86,14 +103,35 @@ export default function MixBench() { useEffect(() => { void (async () => { const [f, s] = await Promise.all([ - fetch("/api/mix/files", { cache: "no-store" }).then((r) => r.json()), + fetch("/api/mix/files?group=project", { cache: "no-store" }).then((r) => r.json()), fetch("/api/mix/session", { cache: "no-store" }).then((r) => r.json()), ]); - setFiles(f.files ?? []); + setGroups(f.groups ?? []); + setOther(f.other ?? []); + + // PRECEDENCE: preset > the saved pair > BLANK. `last` is used only when + // there is no preset -- a link that named a file and a window must not + // lose to whatever this bench was last pointed at. + if (preset) { + // The per-pair knobs are still worth having: handover, fade, gain and + // duck are things learned about THIS pair, while body/start/end came + // from the link. + const pair = s?.byPair?.[`${preset.body}|`] ?? {}; + setSpec({ + ...BLANK, + ...pair, + body: preset.body, + bg: null, + start: preset.start, + end: preset.end, + }); + return; + } const last = s?.last ? s.byPair?.[s.last] : null; if (last?.body) setSpec({ ...BLANK, ...last }); })(); - }, []); + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [preset?.body, preset?.start, preset?.end]); const loadTrack = useCallback(async (which: "body" | "bg", path: string | null) => { if (!path) { @@ -328,21 +366,71 @@ export default function MixBench() { setView({ from: Math.max(0, spec.handover - 3), to: Math.min(duration || spec.handover + 3, spec.handover + 3) }); }, [spec.handover, duration]); + // ---- the picker ---------------------------------------------------------- + // + // Grouped by project, because a flat newest-first list with a cap could not + // show them all: four of six report deliverables fell off the end of a + // 600-entry list, and per-song cuts never appeared at all. + // + // The <option value> stays the ABSOLUTE path. That is what dodges the + // relative-binding hazard in resolveInRoots -- a relative label is tried + // against each root in order and never stats, so it binds to the first root it + // COULD live under whether or not it is there. Displaying the short label and + // sending the full path is what keeps that from mattering. + const needle = filter.trim().toLowerCase(); + const keep = (f: FileRow) => !needle || f.label.toLowerCase().includes(needle); + const shownGroups = groups + .map((g) => ({ ...g, files: g.files.filter(keep) })) + .filter((g) => g.files.length > 0); + const shownOther = other.filter(keep); + const nShown = shownGroups.reduce((n, g) => n + g.files.length, 0) + shownOther.length; + const picker = (which: "body" | "bg") => ( <select value={(which === "body" ? spec.body : spec.bg) ?? ""} onChange={(e) => set(which === "body" ? { body: e.target.value } : { bg: e.target.value || null })} + data-picker={which} className="w-full rounded border border-[var(--color-line)] bg-[var(--color-panel-2)] px-2 py-1.5 text-[13px] outline-none focus:border-[var(--color-sel)]" > <option value="">{which === "bg" ? "— none (song only) —" : "— choose a render —"}</option> - {files.map((f) => ( - <option key={f.path} value={f.path}> - {f.label} - </option> + {shownGroups.map((g) => ( + <optgroup key={g.project} label={g.label}> + {g.files.map((f) => ( + <option key={f.path} value={f.path}> + {f.label} + </option> + ))} + </optgroup> ))} + {shownOther.length > 0 && ( + <optgroup label="— other —"> + {shownOther.map((f) => ( + <option key={f.path} value={f.path}> + {f.label} + </option> + ))} + </optgroup> + )} </select> ); + const pickerFilter = ( + <label className="flex items-center gap-1.5"> + <input + type="search" + value={filter} + onChange={(e) => setFilter(e.target.value)} + placeholder="filter files…" + aria-label="filter the file list" + data-picker-filter="" + className="w-full rounded border border-[var(--color-line)] bg-[var(--color-panel-2)] px-2 py-1 font-mono text-[11px] outline-none focus:border-[var(--color-sel)]" + /> + <span className="micro whitespace-nowrap" data-picker-count={nShown}> + {nShown} + </span> + </label> + ); + const field = ( label: string, key: "handover" | "fade" | "bgGain" | "duck" | "start" | "end", @@ -364,6 +452,19 @@ export default function MixBench() { return ( <div className="mx-auto w-full max-w-6xl p-4"> + {preset && ( + // The window came from a clip, but the FILE is wider than the clip -- + // saying so is what stops the material outside it looking unreachable. + <p className="mb-2 text-[11px] text-[var(--color-dim)]" data-mix-preset={preset.clip ?? ""}> + this window came from{" "} + <span className="font-mono text-[var(--color-text)]"> + {preset.from} + {preset.clip ? ` / ${preset.clip}` : ""} + </span>{" "} + — the file itself runs {preset.fileFrom.toFixed(2)}–{preset.fileTo.toFixed(2)}s + </p> + )} + <div className="mb-2">{pickerFilter}</div> <div className="mb-3 grid gap-3 md:grid-cols-2"> <div> <div className="micro mb-1">song — the render with the picture</div> diff --git a/umtool/components/projects/ClaimBench.tsx b/umtool/components/projects/ClaimBench.tsx @@ -0,0 +1,481 @@ +"use client"; + +import Link from "next/link"; +import { useCallback, useEffect, useMemo, useRef, useState } from "react"; +import { + POPULATIONS, + SCOPES, + SCOPE_CONFIDENCE, + VALUE_KINDS, + rosterLine, +} from "report-to-video/ledger-totals"; + +// Ruling on one claim. +// +// The layout puts the CONTEXT first and the controls second, on purpose. The +// temptation this page exists to resist is adjudicating from the quote -- the +// quote is what produced the four hazards in the first place, and every one of +// them is invisible inside it. So the cited cue is highlighted inside a +// ninety-second paragraph, and the audio starts a few seconds early. + +export type ClaimBenchData = { + project: string; + claim: { + id: string; + date: string; + company: string | null; + value: number | null; + display: string | null; + label: string | null; + quote: string | null; + src: string | null; + video: string | null; + channel: string | null; + cite: number | null; + scope: string | null; + scopeBasis: string | null; + scopeConfidence: string | null; + population: string | null; + valueKind: string | null; + flags: string[] | null; + roles: Array<{ role: string; count: number; verbatim: string }> | null; + }; + at: number | null; + view: { from: number; to: number }; + pad: number; + cues: Array<{ start: number; end: number; text: string; cited: boolean }>; + source: { title: string | null; uploadDate: string | null; duration: number | null; webpageUrl: string | null } | null; + noCues: boolean; + windows: Array<{ name: string; from: number; to: number }>; + gaps: string[]; + fired: Array<{ rule: string; text: string }>; + prev: string | null; + next: string | null; + token: string | null; +}; + +const hms = (t: number) => { + const s = Math.max(0, Math.floor(t)); + const h = Math.floor(s / 3600); + const m = Math.floor((s % 3600) / 60); + const r = s % 60; + return h ? `${h}:${String(m).padStart(2, "0")}:${String(r).padStart(2, "0")}` : `${m}:${String(r).padStart(2, "0")}`; +}; + +/** What each vocabulary term means, so nobody has to guess at the radio. */ +const HELP: Record<string, string> = { + media: "The Quartering — the media team", + coffee: "Coffee Brand Coffee", + publica: "The Publica", + all: "spans everything he owns", + clear: "the quote itself settles it", + read: "the context settles it; the quote alone would not", + unresolved: "genuinely ambiguous — feeds NEITHER total", + uttered: "he says this number, as one number, for this scope", + derived: "our arithmetic over his per-company claims", + synthetic: "our midpoint of a range he gave", + employees: "“employees”", + "full-time": "“full-time”", + salaried: "“salaried”", + contractor: "“contractors”", + "1099": "“1099”", + people: "“people”", +}; + +function Choice({ + label, + options, + value, + onChange, + name, +}: { + label: string; + options: readonly string[]; + value: string | null; + onChange: (v: string) => void; + name: string; +}) { + return ( + <div> + <div className="micro mb-1">{label}</div> + <div className="flex flex-wrap gap-1"> + {options.map((o) => ( + <button + key={o} + type="button" + data-testid={`${name}-${o}`} + aria-pressed={value === o} + title={HELP[o] ?? o} + onClick={() => onChange(o)} + className={ + "rounded border px-2 py-1 text-[11px] " + + (value === o + ? "border-[var(--color-sel)] bg-[var(--color-sel)]/15 text-[var(--color-sel)]" + : "border-[var(--color-line)] text-[var(--color-dim)] hover:text-[var(--color-fg)]") + } + > + {o} + </button> + ))} + </div> + {value && HELP[value] ? ( + <div className="mt-1 text-[10px] text-[var(--color-dim)]">{HELP[value]}</div> + ) : null} + </div> + ); +} + +type Role = { role: string; count: number; verbatim: string }; + +const rolesToText = (roles: ClaimBenchData["claim"]["roles"]) => + (roles ?? []).map((r) => `${r.count} | ${r.role} | ${r.verbatim}`).join("\n"); + +/** + * `2 | video editor | two video editors` per line. + * + * `verbatim` is required and is the point of the field: the count and the role + * name are OUR reading, and without his own words beside them nobody can check + * the reading against the audio. + */ +function parseRoles(text: string): { ok: true; list: Role[] } | { ok: false; error: string } { + const lines = text.split("\n").map((l) => l.trim()).filter(Boolean); + const list: Role[] = []; + for (const [i, line] of lines.entries()) { + const parts = line.split("|").map((s) => s.trim()); + if (parts.length !== 3) return { ok: false, error: `line ${i + 1} needs count | role | his words` }; + const count = Number(parts[0]); + if (!Number.isFinite(count) || count < 0) return { ok: false, error: `line ${i + 1}: count is not a number` }; + if (!parts[1]) return { ok: false, error: `line ${i + 1}: no role` }; + if (!parts[2]) return { ok: false, error: `line ${i + 1}: quote his words for it` }; + list.push({ count, role: parts[1], verbatim: parts[2] }); + } + return { ok: true, list }; +} + +export default function ClaimBench({ data }: { data: ClaimBenchData }) { + const { claim, at, view, cues } = data; + const video = useRef<HTMLVideoElement>(null); + + const [scope, setScope] = useState(claim.scope); + const [confidence, setConfidence] = useState(claim.scopeConfidence); + const [population, setPopulation] = useState(claim.population); + const [valueKind, setValueKind] = useState(claim.valueKind); + const [basis, setBasis] = useState(claim.scopeBasis ?? ""); + const [flagText, setFlagText] = useState((claim.flags ?? []).join("\n")); + // The roster, as one line per role: `2 | video editor | two video editors`. + // A textarea rather than a row of inputs because most claims have none and + // the handful that do are a two-line list -- a repeater widget would be more + // chrome than the field it edits. + const [roleText, setRoleText] = useState(rolesToText(claim.roles)); + const [token, setToken] = useState(data.token ?? ""); + const [busy, setBusy] = useState<string | null>(null); + const [note, setNote] = useState<string | null>(null); + const [fetching, setFetching] = useState(false); + + // The widest cached window containing the cite. When there is one the moment + // is playable with no download at all -- which is most of them once a build + // has run, because the build already over-fetched around every clip. + const cached = data.windows[0] ?? null; + + const roles = useMemo(() => parseRoles(roleText), [roleText]); + const rosterPreview = roles.ok ? rosterLine(roles.list) : null; + + const dirty = + roleText !== rolesToText(claim.roles) || + scope !== claim.scope || + confidence !== claim.scopeConfidence || + population !== claim.population || + valueKind !== claim.valueKind || + basis !== (claim.scopeBasis ?? "") || + flagText !== (claim.flags ?? []).join("\n"); + + const complete = !!(scope && confidence && population && valueKind && basis.trim()); + + // Start a few seconds EARLY. Landing exactly on the cited word is landing + // mid-sentence, which is the position this page exists to get out of. + const seekTo = useCallback( + (t: number) => { + const el = video.current; + if (!el || !cached) return; + el.currentTime = Math.max(0, t - cached.from); + void el.play().catch(() => {}); + }, + [cached], + ); + + useEffect(() => { + if (at != null && cached) { + const el = video.current; + if (el) el.currentTime = Math.max(0, at - cached.from - 4); + } + }, [at, cached]); + + const save = useCallback(async () => { + setBusy("saving…"); + setNote(null); + const r = await fetch("/api/report/claim", { + method: "PUT", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + project: data.project, + claim: claim.id, + token, + scope, + scopeConfidence: confidence, + population, + valueKind, + scopeBasis: basis, + flags: flagText.split("\n").map((s) => s.trim()).filter(Boolean), + roles: roles.ok ? roles.list : undefined, + }), + }); + const j = (await r.json()) as Record<string, unknown>; + setBusy(null); + if (!r.ok) { + setNote( + j.stale + ? "the manifest changed since you opened this — reload before saving, or your ruling would overwrite whatever was written" + : `could not save: ${String(j.error ?? r.status)}`, + ); + return; + } + setToken(String(j.token ?? "")); + const gaps = (j.gaps as string[]) ?? []; + setNote(gaps.length ? `saved — still missing ${gaps.join(", ")}` : "saved — adjudicated"); + }, [data.project, claim.id, token, scope, confidence, population, valueKind, basis, flagText, roles]); + + const fetchWindow = useCallback(async () => { + setFetching(true); + setNote(null); + const r = await fetch("/api/report/fetch", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ project: data.project, clip: claim.id, pad: data.pad + 5 }), + }); + const j = (await r.json()) as Record<string, unknown>; + if (!r.ok) { + setFetching(false); + setNote(`could not fetch: ${String(j.error ?? r.status)}`); + return; + } + // The job runs on the server; poll it rather than guess how long it takes. + const job = j.job as { id: string } | undefined; + for (let i = 0; i < 300 && job; i += 1) { + await new Promise((res) => setTimeout(res, 1000)); + const s = await fetch(`/api/report/fetch?job=${job.id}`, { cache: "no-store" }); + const sj = (await s.json()) as { job?: { state?: string } }; + if (!sj.job || sj.job.state === "done" || sj.job.state === "failed") break; + } + setFetching(false); + window.location.reload(); + }, [data.project, claim.id, data.pad]); + + const paragraph = useMemo( + () => + cues.map((c, i) => ( + <span + key={`${c.start}-${i}`} + data-cited={c.cited ? "1" : undefined} + onClick={() => seekTo(c.start)} + className={ + "cursor-pointer " + + (c.cited + ? "bg-[var(--color-sel)]/25 text-[var(--color-fg)]" + : "text-[var(--color-dim)] hover:text-[var(--color-fg)]") + } + > + {c.text}{" "} + </span> + )), + [cues, seekTo], + ); + + return ( + <div className="space-y-3" data-claim={claim.id}> + <div className="flex items-baseline gap-3"> + <div className="font-mono text-[13px] text-[var(--color-fg)]">{claim.id}</div> + <div className="text-[12px] text-[var(--color-dim)]">{claim.date}</div> + <div className="text-[13px] text-[var(--color-meter)]"> + {claim.display ?? (claim.value == null ? "—" : String(claim.value))} + </div> + <div className="flex-1" /> + {data.prev ? ( + <Link className="text-[11px] text-[var(--color-sel)] hover:underline" href={`/browse/${data.project}/claim/${data.prev}`}> + ← {data.prev} + </Link> + ) : null} + {data.next ? ( + <Link className="text-[11px] text-[var(--color-sel)] hover:underline" href={`/browse/${data.project}/claim/${data.next}`}> + {data.next} → + </Link> + ) : null} + </div> + + <div className="grid gap-3 lg:grid-cols-[minmax(0,1fr)_360px]"> + <div className="space-y-3"> + {cached ? ( + <video + ref={video} + data-testid="claim-video" + src={`/api/report/raw?project=${encodeURIComponent(data.project)}&claim=${encodeURIComponent(claim.id)}&file=${encodeURIComponent(cached.name)}`} + className="aspect-video w-full rounded border border-[var(--color-line)] bg-black" + preload="metadata" + controls + /> + ) : ( + <div className="flex aspect-video w-full flex-col items-center justify-center gap-2 rounded border border-dashed border-[var(--color-line)] text-[12px] text-[var(--color-dim)]"> + <div>no cached window covers this moment</div> + <button + type="button" + data-testid="claim-fetch" + disabled={fetching || !claim.video} + onClick={() => void fetchWindow()} + className="rounded border border-[var(--color-line)] px-2 py-1 text-[11px] hover:text-[var(--color-fg)] disabled:opacity-40" + > + {fetching ? "fetching…" : `fetch ±${data.pad}s`} + </button> + </div> + )} + + <div className="rounded border border-[var(--color-line)] p-3"> + <div className="micro mb-2"> + ±{data.pad}s of context · {hms(view.from)}–{hms(view.to)} + {at != null ? ` · cited at ${hms(at)}` : ""} + </div> + {data.noCues ? ( + <div className="text-[12px] text-[var(--color-dim)]"> + no transcript.cues.json for {claim.channel}/{claim.video} — on Rumble, check the + directory is the URL slug and not the site id + </div> + ) : ( + <p + data-testid="claim-context" + className="max-h-[320px] overflow-y-auto text-[12px] leading-relaxed" + > + {paragraph} + </p> + )} + </div> + + <div className="rounded border border-[var(--color-line)] p-3 text-[12px]"> + <div className="micro mb-1">as recorded</div> + <div className="text-[var(--color-fg)]">“{claim.quote}”</div> + <div className="mt-1 text-[11px] text-[var(--color-dim)]"> + {claim.label} · {claim.src} + </div> + </div> + </div> + + <div className="space-y-3 text-[12px]"> + {data.gaps.length ? ( + <div className="rounded border border-[var(--color-warn,#E8A33F)] p-2 text-[11px] text-[var(--color-dim)]"> + unadjudicated — missing {data.gaps.join(", ")} + </div> + ) : ( + <div className="rounded border border-[var(--color-line)] p-2 text-[11px] text-[var(--color-dim)]"> + adjudicated + </div> + )} + + <Choice name="scope" label="scope" options={SCOPES} value={scope} onChange={setScope} /> + <Choice + name="confidence" + label="scope confidence" + options={SCOPE_CONFIDENCE} + value={confidence} + onChange={setConfidence} + /> + <Choice + name="population" + label="population" + options={POPULATIONS} + value={population} + onChange={setPopulation} + /> + <Choice + name="valuekind" + label="value kind" + options={VALUE_KINDS} + value={valueKind} + onChange={setValueKind} + /> + + <div> + <div className="micro mb-1">scope basis — the phrase that settles it</div> + <textarea + data-testid="claim-basis" + value={basis} + onChange={(e) => setBasis(e.target.value)} + rows={3} + className="w-full rounded border border-[var(--color-line)] bg-transparent p-2 text-[11px]" + placeholder="quote the words from the context that decide the scope" + /> + </div> + + <div> + <div className="micro mb-1"> + roster — one role per line: <span className="font-mono">count | role | his words</span> + </div> + <textarea + data-testid="claim-roles" + value={roleText} + onChange={(e) => setRoleText(e.target.value)} + rows={3} + className="w-full rounded border border-[var(--color-line)] bg-transparent p-2 text-[11px]" + placeholder="2 | video editor | two video editors" + /> + <div className="mt-1 text-[10px] text-[var(--color-dim)]"> + {roleText.trim() === "" + ? "only for a claim where he ENUMERATES who works for him" + : roles.ok + ? `reads as “${rosterPreview}”` + : `not saveable yet — ${roles.error}`} + </div> + </div> + + <div> + <div className="micro mb-1">flags — one per line, free text</div> + <textarea + data-testid="claim-flags" + value={flagText} + onChange={(e) => setFlagText(e.target.value)} + rows={2} + className="w-full rounded border border-[var(--color-line)] bg-transparent p-2 text-[11px]" + placeholder="e.g. reading someone else's tweet aloud" + /> + </div> + + {data.fired.length ? ( + <div className="rounded border border-[var(--color-line)] p-2"> + <div className="micro mb-1">predicates firing on this claim</div> + <ul className="space-y-1 text-[11px] text-[var(--color-dim)]"> + {data.fired.map((f, i) => ( + <li key={i}> + <span className="font-mono text-[10px] text-[var(--color-meter)]">{f.rule}</span>{" "} + {f.text} + </li> + ))} + </ul> + </div> + ) : null} + + <div className="flex items-center gap-2"> + <button + type="button" + data-testid="claim-save" + disabled={!dirty || !!busy || !roles.ok} + onClick={() => void save()} + className="rounded border border-[var(--color-sel)] px-3 py-1 text-[11px] text-[var(--color-sel)] disabled:opacity-40" + > + {busy ?? "save ruling"} + </button> + {!complete ? ( + <span className="text-[10px] text-[var(--color-dim)]">all six fields, or it stays blocking</span> + ) : null} + </div> + {note ? <div data-testid="claim-note" className="text-[11px] text-[var(--color-dim)]">{note}</div> : null} + </div> + </div> + </div> + ); +} diff --git a/umtool/components/projects/ClaimBenchPage.tsx b/umtool/components/projects/ClaimBenchPage.tsx @@ -0,0 +1,101 @@ +import { notFound } from "next/navigation"; +import BrowseHeader from "@/components/BrowseHeader"; +import ClaimBench, { type ClaimBenchData } from "./ClaimBench"; +import { manifestToken } from "@/lib/report/manifest.mjs"; +import { readClaimDetail, readManifest } from "@/lib/projects/report.mjs"; +import { ledgerTotals } from "report-to-video/ledger-totals"; +import type { ProjectRef } from "@/lib/project-types"; + +// The server half of the claim bench. +// +// Same rule as the clip bench: the id is validated as a MEMBER of the ledger, +// never as a path. The difference is what gets read -- a clip wants its window +// and its waveform, a claim wants ROOM. Ninety seconds either side, because a +// first-person quote is routinely the host reading someone else's words or +// being sarcastic, and the quote alone cannot show you which. + +export default async function ClaimBenchPage({ + project, + claimId, +}: { + project: ProjectRef; + claimId: string; +}) { + const manifest = await readManifest(project.dir); + if (!manifest) notFound(); + + const detail = await readClaimDetail(project.dir, claimId, { manifest }); + if (!detail) notFound(); + + const ledger = (manifest.ledger ?? []) as Array<Record<string, unknown>>; + const i = ledger.findIndex((e) => e.id === claimId); + + // Which predicates fire on THIS claim, computed over whatever has been ruled + // on so far. `strict: false` because the ledger is by definition half-worked + // while somebody is standing on this page. + let fired: Array<{ rule: string; text: string }> = []; + try { + const totals = ledgerTotals(ledger, { strict: false }); + fired = totals.steps.find((s: { id: string }) => s.id === claimId)?.flags ?? []; + } catch { + /* a malformed ledger shows as gaps, not as a broken page */ + } + + const data: ClaimBenchData = { + project: project.id, + claim: { + id: String(detail.claim.id), + date: String(detail.claim.date ?? ""), + company: (detail.claim.company as string) ?? null, + value: (detail.claim.value as number | null) ?? null, + display: (detail.claim.display as string) ?? null, + label: (detail.claim.label as string) ?? null, + quote: (detail.claim.quote as string) ?? null, + src: (detail.claim.src as string) ?? null, + video: (detail.claim.video as string) ?? null, + channel: (detail.claim.channel as string) ?? null, + cite: (detail.claim.cite as number | null) ?? null, + scope: (detail.claim.scope as string) ?? null, + scopeBasis: (detail.claim.scopeBasis as string) ?? null, + scopeConfidence: (detail.claim.scopeConfidence as string) ?? null, + population: (detail.claim.population as string) ?? null, + valueKind: (detail.claim.valueKind as string) ?? null, + flags: Array.isArray(detail.claim.flags) ? (detail.claim.flags as string[]) : null, + roles: Array.isArray(detail.claim.roles) + ? (detail.claim.roles as ClaimBenchData["claim"]["roles"]) + : null, + }, + at: detail.at, + view: detail.view, + pad: detail.pad, + cues: detail.cues, + source: detail.source, + noCues: detail.noCues, + windows: detail.windows, + gaps: detail.gaps, + fired, + prev: i > 0 ? String(ledger[i - 1].id) : null, + next: i >= 0 && i < ledger.length - 1 ? String(ledger[i + 1].id) : null, + token: await manifestToken(project.dir), + }; + + const outstanding = ledger.filter( + (e) => !e.scope || !e.scopeBasis || !e.scopeConfidence || !e.population || !e.valueKind || !Array.isArray(e.flags), + ).length; + + return ( + <div className="flex h-full flex-col"> + <BrowseHeader + crumbs={[ + { href: "/browse", label: "projects" }, + { href: `/browse/${project.id}`, label: project.name }, + { label: claimId }, + ]} + note={`claim ${i + 1} of ${ledger.length} · ${outstanding} unadjudicated`} + /> + <main className="deck-main flex-1 p-4"> + <ClaimBench data={data} /> + </main> + </div> + ); +} diff --git a/umtool/components/projects/ClipBench.tsx b/umtool/components/projects/ClipBench.tsx @@ -0,0 +1,549 @@ +"use client"; + +import { useCallback, useEffect, useRef, useState } from "react"; +import Link from "next/link"; +import Waveform, { type Peaks } from "@/components/Waveform"; +import { badgeVariants } from "@/components/ui/badge"; +import { buttonVariants } from "@/components/ui/button"; + +// --------------------------------------------------------------------------- +// Editing a clip's window against the audio and the words. +// +// "How much context does this clip need" was a loop of hand-editing JSON, +// re-running two CLIs and watching an mp4. This is that loop, in one place. +// +// EVERYTHING IS IN ABSOLUTE SOURCE SECONDS -- the manifest's, the cue file's, +// the QR's. The cached file's own start (`fetchStart`) is the only relative +// number in the whole component, and it exists solely to set video.currentTime. +// The moment those two are allowed to mix is the moment a window is off by the +// pad and nobody can see why. +// +// A DRAG NEVER DOWNLOADS. Dragging past the cached window clamps and offers a +// button, because a handle that silently starts a 12-second network fetch is a +// handle you stop trusting. +// --------------------------------------------------------------------------- + +type Cue = { start: number; end: number; text: string; endsSentence: boolean }; +type Win = { name: string; from: number; to: number }; + +type Clip = { + id: string; + video: string; + channel: string | null; + start: number; + end: number; + cite: number | null; + quote: string | null; + note: string | null; + lock: boolean; + lockStart: boolean; + lockEnd: boolean; +}; + +export type ClipBenchData = { + project: string; + clip: Clip; + view: { from: number; to: number }; + windows: Win[]; + proposed: { start: number; end: number } | null; + endsSentence: boolean | null; + noPunctuation: boolean; + sourceDuration: number | null; + segment: string | null; + cues: Cue[]; + token: string | null; + fetchPad: number; + /** A /mix deep link for this clip's cached window, or null. Server-built. */ + mixHref: string | null; +}; + +const hms = (t: number) => { + const s = Math.max(0, t); + const m = Math.floor(s / 60); + const sec = s - m * 60; + const h = Math.floor(m / 60); + return h > 0 + ? `${h}:${String(m % 60).padStart(2, "0")}:${sec.toFixed(2).padStart(5, "0")}` + : `${m}:${sec.toFixed(2).padStart(5, "0")}`; +}; + +const round2 = (n: number) => Number(n.toFixed(2)); + +export default function ClipBench({ data }: { data: ClipBenchData }) { + const [clip, setClip] = useState<Clip>(data.clip); + const [token, setToken] = useState(data.token); + const [windows, setWindows] = useState<Win[]>(data.windows); + const [cues, setCues] = useState<Cue[]>(data.cues); + const [proposed, setProposed] = useState(data.proposed); + const [view, setView] = useState(data.view); + const [sel, setSel] = useState({ from: data.clip.start, to: data.clip.end }); + const [peaks, setPeaks] = useState<Peaks | null>(null); + const [playhead, setPlayhead] = useState<number | null>(null); + const [note, setNote] = useState<string | null>(null); + const [busy, setBusy] = useState<string | null>(null); + const [dirty, setDirty] = useState(false); + + const video = useRef<HTMLVideoElement | null>(null); + const stopAt = useRef<number | null>(null); + + // The widest cached file is the one the bench draws from: it is how much room + // there is to drag before anything has to be fetched. + const cached = windows[0] ?? null; + const fetchStart = cached?.from ?? 0; + const cachedTo = cached?.to ?? 0; + + // ---- peaks -------------------------------------------------------------- + useEffect(() => { + if (!cached) return; + let live = true; + void (async () => { + const r = await fetch( + `/api/report/peaks?project=${encodeURIComponent(data.project)}&clip=${encodeURIComponent(clip.id)}&file=${encodeURIComponent(cached.name)}`, + { cache: "no-store" }, + ); + if (!r.ok || !live) return; + setPeaks((await r.json()) as Peaks); + })(); + return () => { + live = false; + }; + }, [cached, data.project, clip.id]); + + // ---- audition ----------------------------------------------------------- + const play = useCallback( + (from: number, to: number) => { + const el = video.current; + if (!el || !cached) return; + el.currentTime = Math.max(0, from - fetchStart); + stopAt.current = to; + void el.play(); + }, + [cached, fetchStart], + ); + + useEffect(() => { + const el = video.current; + if (!el) return; + const tick = () => { + const t = el.currentTime + fetchStart; + setPlayhead(t); + if (stopAt.current != null && t >= stopAt.current) { + el.pause(); + stopAt.current = null; + } + }; + el.addEventListener("timeupdate", tick); + return () => el.removeEventListener("timeupdate", tick); + }, [fetchStart]); + + // ---- the selection ------------------------------------------------------ + // + // Clamped to the cached file. Past its edge the handle stops and the "fetch + // more" button appears; past the source's own duration it stops for good. + const clampTo = Math.min(cachedTo || Infinity, data.sourceDuration ?? Infinity); + const onSel = useCallback( + (next: { from: number; to: number }, dragging: boolean) => { + const from = Math.max(fetchStart, next.from); + const to = Math.min(clampTo, next.to); + setSel({ from, to }); + setDirty(true); + // Audition on pointer-UP, never during a drag: the Deck's rule, and the + // reason is that a sound restarting on every pointermove is unusable. + if (!dragging) play(from, Math.min(to, from + 6)); + }, + [fetchStart, clampTo, play], + ); + + const atStartEdge = sel.from <= fetchStart + 0.05 && fetchStart > 0; + const atEndEdge = + sel.to >= cachedTo - 0.05 && (data.sourceDuration == null || cachedTo < data.sourceDuration - 0.05); + + // ---- keyboard ----------------------------------------------------------- + useEffect(() => { + const nudge = (which: "from" | "to", by: number) => + onSel(which === "from" ? { ...sel, from: sel.from + by } : { ...sel, to: sel.to + by }, false); + + const onKey = (e: KeyboardEvent) => { + const el = e.target as HTMLElement | null; + if (el && /^(INPUT|TEXTAREA|SELECT)$/.test(el.tagName)) return; + const step = e.shiftKey ? 0.5 : 0.05; + switch (e.key) { + case "[": nudge("from", -step); break; + case "]": nudge("from", step); break; + case ",": nudge("to", -step); break; + case ".": nudge("to", step); break; + case " ": + e.preventDefault(); + play(sel.from, sel.to); + break; + case "r": + case "R": + setSel({ from: clip.start, to: clip.end }); + setDirty(false); + break; + default: + return; + } + e.preventDefault(); + }; + window.addEventListener("keydown", onKey); + return () => window.removeEventListener("keydown", onKey); + }, [sel, clip.start, clip.end, onSel, play]); + + // ---- saving ------------------------------------------------------------- + const save = useCallback( + async (patch: Partial<Clip> & { start?: number; end?: number }) => { + setBusy("saving…"); + setNote(null); + const r = await fetch("/api/report/window", { + method: "PUT", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ project: data.project, clip: clip.id, token, ...patch }), + }); + const j = (await r.json()) as Record<string, unknown>; + setBusy(null); + if (!r.ok) { + setNote( + j.stale + ? "the manifest changed since you opened this — reload before saving, or your edit would overwrite whatever was written" + : `could not save: ${String(j.error ?? r.status)}`, + ); + return false; + } + const entry = j.entry as Clip; + setClip((c) => ({ ...c, ...entry })); + setSel({ from: entry.start, to: entry.end }); + setToken(String(j.token ?? "")); + setDirty(false); + setNote("saved"); + // The widener's opinion changes when the window does. + void refresh(); + return true; + }, + [data.project, clip.id, token], + ); + + const refresh = useCallback(async () => { + const r = await fetch( + `/api/report/clip?project=${encodeURIComponent(data.project)}&clip=${encodeURIComponent(clip.id)}`, + { cache: "no-store" }, + ); + if (!r.ok) return; + const j = (await r.json()) as ClipBenchData; + setWindows(j.windows); + setCues(j.cues); + setProposed(j.proposed); + setToken(j.token); + if (j.windows[0]) setView({ from: j.windows[0].from, to: j.windows[0].to }); + }, [data.project, clip.id]); + + // ---- fetching more -------------------------------------------------------- + const fetchMore = useCallback(async () => { + setBusy("fetching a wider window…"); + setNote(null); + const r = await fetch("/api/report/fetch", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ project: data.project, clip: clip.id, pad: 20 }), + }); + if (!r.ok) { + const j = (await r.json()) as { error?: string }; + setBusy(null); + setNote(j.error ?? "could not start the fetch"); + return; + } + const { job } = (await r.json()) as { job: { id: string } }; + // Poll rather than stream: one 12-second job, and a dead poll is harmless. + for (let i = 0; i < 240; i += 1) { + await new Promise((res) => setTimeout(res, 500)); + const s = await fetch(`/api/report/fetch?job=${job.id}`, { cache: "no-store" }); + const sj = (await s.json()) as { job: { state: string; error: string | null } | null }; + if (!sj.job || sj.job.state === "running") continue; + setBusy(null); + if (sj.job.state === "failed") setNote(`the fetch failed: ${sj.job.error ?? "unknown"}`); + else { + setNote("fetched — the build will reuse this file, not download it again"); + await refresh(); + } + return; + } + setBusy(null); + setNote("the fetch is still running; reload to pick it up"); + }, [data.project, clip.id, refresh]); + + // ---- the warnings -------------------------------------------------------- + const endCue = cues.find((c) => sel.to >= c.start - 0.02 && sel.to <= c.end + 0.02) ?? null; + const endsSentence = data.noPunctuation ? null : endCue ? endCue.endsSentence : null; + const midSentence = endsSentence === false && !clip.lockEnd && !clip.lock; + + // Moving an edge somewhere widen() would not produce means the next + // `resolve-windows --write` reverts it. Offering the matching lock is what + // stops that being a silent loss -- and is why five of six real manifests are + // 100% locked. + const wouldRevert = + !clip.lock && + proposed != null && + (Math.abs(proposed.start - round2(sel.from)) > 0.05 || + Math.abs(proposed.end - round2(sel.to)) > 0.05); + + const span = Math.max(1e-6, view.to - view.from); + const pct = (t: number) => `${((t - view.from) / span) * 100}%`; + + return ( + <div className="space-y-3" data-bench={clip.id}> + {/* ---- the picture ---- */} + <div className="grid gap-3 lg:grid-cols-[minmax(0,1fr)_320px]"> + <div> + {cached ? ( + <video + ref={video} + data-testid="clip-video" + src={`/api/report/raw?project=${encodeURIComponent(data.project)}&clip=${encodeURIComponent(clip.id)}&file=${encodeURIComponent(cached.name)}`} + className="aspect-video w-full rounded border border-[var(--color-line)] bg-black" + preload="metadata" + controls + /> + ) : ( + <div className="flex aspect-video w-full items-center justify-center rounded border border-dashed border-[var(--color-line)] text-[12px] text-[var(--color-dim)]"> + nothing fetched for this clip yet + </div> + )} + </div> + + <div className="space-y-2 text-[12px]"> + <div className="num"> + <div className="micro">window (source seconds)</div> + <div className="text-[var(--color-meter)]"> + {hms(sel.from)} – {hms(sel.to)}{" "} + <span className="text-[var(--color-dim)]">({(sel.to - sel.from).toFixed(2)}s)</span> + </div> + {dirty && ( + <div className="text-[var(--color-dirty)]"> + unsaved — was {hms(clip.start)} – {hms(clip.end)} + </div> + )} + </div> + + <div className="flex flex-wrap gap-1.5"> + <button + type="button" + className={buttonVariants({ variant: "primary", size: "sm" })} + disabled={!dirty || !!busy} + onClick={() => void save({ start: round2(sel.from), end: round2(sel.to) })} + > + save window + </button> + <button + type="button" + className={buttonVariants({ size: "sm" })} + onClick={() => play(sel.from, sel.to)} + > + play selection + </button> + <button + type="button" + className={buttonVariants({ size: "sm" })} + disabled={!dirty} + onClick={() => { + setSel({ from: clip.start, to: clip.end }); + setDirty(false); + }} + > + reset + </button> + {data.mixHref && ( + // Built SERVER-SIDE, because a mix link carries an absolute path + // and this component has no business constructing one. It also + // carries the SAVED window rather than the current selection: a mix + // of an unsaved edit is a mix of something not in the cut. + <Link + href={data.mixHref} + data-mix-link={clip.id} + className={buttonVariants({ size: "sm" })} + > + mix + </Link> + )} + </div> + + <div className="micro"> + <kbd>[</kbd> <kbd>]</kbd> start · <kbd>,</kbd> <kbd>.</kbd> end · <kbd>space</kbd>{" "} + audition · <kbd>R</kbd> reset — hold shift for 0.5s + </div> + + {/* ---- lock, explained rather than labelled ---- */} + <div className="space-y-1 rounded border border-[var(--color-line)] p-2"> + <div className="micro">what resolve-windows may touch</div> + {( + [ + ["lock", "leave this clip alone entirely — the window is deliberate"], + ["lockStart", "pin the start exactly where it is"], + ["lockEnd", "pin the end exactly where it is"], + ] as const + ).map(([k, why]) => ( + <label key={k} className="flex items-start gap-2"> + <input + type="checkbox" + name={k} + checked={!!clip[k]} + disabled={!!busy} + onChange={(e) => void save({ [k]: e.target.checked } as Partial<Clip>)} + /> + <span> + <span className="font-mono text-[11px] text-[var(--color-text)]">{k}</span> + <span className="block text-[11px] text-[var(--color-dim)]">{why}</span> + </span> + </label> + ))} + </div> + + {data.segment && ( + <div className="micro">a segment is already built for this clip</div> + )} + {busy && <div className="text-[var(--color-meter)]">{busy}</div>} + {note && ( + <div data-bench-note="" className="text-[var(--color-dirty)]"> + {note} + </div> + )} + </div> + </div> + + {/* ---- the warnings, before the instrument ---- */} + {midSentence && ( + <p + data-warn="mid-sentence" + className="rounded border border-[var(--color-dirty)] bg-[color-mix(in_srgb,var(--color-dirty)_10%,transparent)] px-3 py-1.5 text-[12px] text-[var(--color-dirty)]" + > + This ends mid-sentence — the cut lands inside &ldquo;… + {endCue?.text.trim().slice(-56)}&rdquo;. Run it to the end of the sentence, or set{" "} + <code className="font-mono">lockEnd</code> to say you meant to cut here. + </p> + )} + {data.noPunctuation && ( + <p + data-warn="no-punctuation" + className="rounded border border-[var(--color-line)] px-3 py-1.5 text-[12px] text-[var(--color-dim)]" + > + This upload&rsquo;s ASR carries no punctuation, so there are no sentence boundaries to + find and widening cannot help. Set these edges by ear and lock them. + </p> + )} + {wouldRevert && ( + <p + data-warn="would-revert" + className="rounded border border-[var(--color-dirty)] px-3 py-1.5 text-[12px] text-[var(--color-dirty)]" + > + <code className="font-mono">resolve-windows --write</code> would make this{" "} + {hms(proposed!.start)} – {hms(proposed!.end)}, reverting your edit. Set the matching lock + if this window is deliberate.{" "} + <button + type="button" + className={buttonVariants({ variant: "primary", size: "sm" })} + onClick={() => + void save({ + start: round2(sel.from), + end: round2(sel.to), + lockStart: Math.abs(proposed!.start - round2(sel.from)) > 0.05, + lockEnd: Math.abs(proposed!.end - round2(sel.to)) > 0.05, + }) + } + > + save and lock + </button> + </p> + )} + + {/* ---- the waveform ---- */} + <Waveform + view={view} + sel={sel} + cand={{ from: clip.start, to: clip.end }} + words={cues.map((c) => ({ start: c.start, end: c.end, w: c.text.slice(0, 24) }))} + peaks={peaks} + playhead={playhead} + onSel={onSel} + onReachEdge={() => { + /* widening is a FETCH here, not a redraw -- see the button below */ + }} + /> + + {(atStartEdge || atEndEdge) && ( + <div className="flex items-center gap-2 text-[12px]"> + <span className="text-[var(--color-dirty)]"> + that is the edge of what is cached{cached ? ` (${cached.name})` : ""} + </span> + <button + type="button" + className={buttonVariants({ variant: "primary", size: "sm" })} + disabled={!!busy} + onClick={() => void fetchMore()} + > + fetch 20s more + </button> + </div> + )} + + {/* ---- the cue rail ---- */} + <div> + <div className="micro mb-1"> + what is being said — a marked cue closes a sentence; click an edge to snap to it + </div> + <div className="relative h-16 w-full overflow-hidden rounded border border-[var(--color-line)] bg-[var(--color-panel-2)]"> + {cues.map((c) => { + const inSel = c.end > sel.from && c.start < sel.to; + return ( + <button + type="button" + key={`${c.start}-${c.end}`} + data-cue={c.start} + data-ends-sentence={c.endsSentence ? "1" : "0"} + title={c.text} + onClick={() => { + // Snap the NEARER edge to this cue's nearer boundary. + const dStart = Math.abs(sel.from - c.start); + const dEnd = Math.abs(sel.to - c.end); + onSel(dStart <= dEnd ? { ...sel, from: c.start } : { ...sel, to: c.end }, false); + }} + className={`absolute top-0 h-full overflow-hidden border-l px-1 text-left text-[10px] leading-tight ${ + inSel + ? "border-[var(--color-sel)] text-[var(--color-text)]" + : "border-[var(--color-line)] text-[var(--color-dim)] opacity-60" + }`} + style={{ left: pct(c.start), width: `calc(${pct(c.end)} - ${pct(c.start)})` }} + > + {c.endsSentence && ( + <span className="mr-1 text-[var(--color-meter)]" aria-label="closes a sentence"> + ¶ + </span> + )} + {c.text} + </button> + ); + })} + </div> + </div> + + {clip.quote && ( + <p className="text-[12px] leading-snug text-[var(--color-dim)]"> + <span className="micro mr-2">the quote this clip exists for</span> + &ldquo;{clip.quote}&rdquo; + </p> + )} + + <div className="flex flex-wrap items-center gap-2 text-[11px]"> + <span className={badgeVariants({ variant: "info", size: "sm" })}>{clip.video}</span> + {clip.channel && <span className={badgeVariants({ size: "sm" })}>{clip.channel}</span>} + {windows.length > 1 && ( + <span className="micro">{windows.length} cached windows for this source</span> + )} + <Link + href={`/browse/${data.project}`} + className="ml-auto text-[var(--color-sel)] hover:underline" + > + ← the whole cut + </Link> + </div> + </div> + ); +} diff --git a/umtool/components/projects/ClipBenchPage.tsx b/umtool/components/projects/ClipBenchPage.tsx @@ -0,0 +1,100 @@ +import path from "node:path"; +import { notFound } from "next/navigation"; +import BrowseHeader from "@/components/BrowseHeader"; +import ClipBench, { type ClipBenchData } from "./ClipBench"; +import { manifestToken } from "@/lib/report/manifest.mjs"; +import { cuesInWindow, readClipDetail, readManifest } from "@/lib/projects/report.mjs"; +import { windowsFor } from "@/lib/report/serve.mjs"; +import type { ProjectRef } from "@/lib/project-types"; + +// The server half of the bench: resolve the clip, read what it needs, and hand +// it over. The clip id is validated as a MEMBER of the timeline, never as a +// path -- the same rule every other name that crosses the wire here follows. + +export default async function ClipBenchPage({ + project, + clipId, +}: { + project: ProjectRef; + clipId: string; +}) { + const manifest = await readManifest(project.dir); + if (!manifest) notFound(); + + const detail = await readClipDetail(project.dir, { manifest }); + if (!detail) notFound(); + const entry = detail.entries.find( + (e: { id: string; kind: string }) => e.id === clipId && e.kind === "clip", + ); + if (!entry) notFound(); + + const windows = await windowsFor(project, entry); + const widest = windows[0] ?? null; + const view = widest + ? { from: widest.from, to: widest.to } + : { from: Math.max(0, entry.start - 20), to: entry.end + 20 }; + const cues = await cuesInWindow(project.dir, clipId, view.from, view.to); + + const data: ClipBenchData = { + project: project.id, + clip: { + id: entry.id, + video: entry.video, + channel: entry.channel ?? null, + start: entry.start, + end: entry.end, + cite: entry.cite ?? null, + quote: entry.quote ?? null, + note: entry.note ?? null, + lock: !!entry.lock, + lockStart: !!entry.lockStart, + lockEnd: !!entry.lockEnd, + }, + view, + windows: windows.map((w: { name: string; from: number; to: number }) => ({ + name: w.name, + from: w.from, + to: w.to, + })), + proposed: entry.proposed ?? null, + endsSentence: entry.endsSentence ?? null, + noPunctuation: !!entry.noPunctuation, + sourceDuration: entry.duration ?? null, + segment: entry.segment ?? null, + cues: cues?.cues ?? [], + token: await manifestToken(project.dir), + fetchPad: manifest.render?.fetchPad ?? 3, + // The widest cached RAW window: it has no chrome burned in, which is what a + // mix is looking at, and it exists as soon as the clip has been fetched + // once. start/end are the window minus the file's own start, because a mix + // spec is relative to the file it names. + mixHref: widest + ? `/mix?${new URLSearchParams({ + body: path.join(project.dir, "out", "clips-raw", widest.name), + start: (entry.start - widest.from).toFixed(2), + end: (entry.end - widest.from).toFixed(2), + from: project.id, + clip: clipId, + })}` + : null, + }; + + const clips = detail.entries.filter((e: { kind: string }) => e.kind === "clip"); + const i = clips.findIndex((e: { id: string }) => e.id === clipId); + + return ( + <div className="flex h-full flex-col"> + <BrowseHeader + crumbs={[ + { href: "/browse", label: "projects" }, + { href: `/browse/${project.id}`, label: project.name }, + { label: clipId }, + ]} + note={`clip ${i + 1} of ${clips.length} · ${entry.video}`} + /> + <main className="deck-main flex-1 p-4"> + <ClipBench data={data} /> + </main> + </div> + ); +} diff --git a/umtool/components/projects/CutPage.tsx b/umtool/components/projects/CutPage.tsx @@ -0,0 +1,125 @@ +import Link from "next/link"; +import { notFound } from "next/navigation"; +import BrowseHeader from "@/components/BrowseHeader"; +import CutBench from "@/components/CutBench"; +import ProvenancePanel from "@/components/ProvenancePanel"; +import { isCutName, probeAll, readSong } from "@/lib/browse"; +import { fmtBytes, fmtDur } from "@/lib/format"; +import { buildStatus } from "@/lib/manifest"; +import { readProvenance } from "@/lib/provenance"; +import { readNotes } from "@/lib/notes"; +import { readCompares } from "@/lib/compares"; +import LoudnessTable from "@/components/LoudnessTable"; +import { cachedLoudness } from "@/lib/loudness"; +import { DEFAULT_TARGET } from "@/lib/loudness-types"; +import { readSpec } from "@/lib/spec"; +import { resolveRendition } from "@/lib/browse"; + +// Moved out of app/browse/[song]/[cut]/page.tsx unchanged. `/browse/<song>/wide` +// is now a project id plus a VIEW, resolved by the registry rather than by a +// route segment -- the URL is byte-identical either way, which is the point. + +export default async function CutPage({ + id, + cutName, + search, +}: { + id: string; + cutName: string; + search: { v?: string; plan?: string; notes?: string }; +}) { + const { v, plan, notes: notesParam } = search; + if (!isCutName(cutName)) notFound(); + + const song = await readSong(id); + if (!song) notFound(); + const cut = song.cuts.find((c) => c.name === cutName); + if (!cut) notFound(); + + const rels = [cut.shipped?.rel, ...cut.variants.map((x) => x.rel)].filter( + (r): r is string => !!r, + ); + const info = await probeAll(id, rels); + + const durations: Record<string, number> = {}; + for (const [rel, i] of Object.entries(info)) durations[rel] = i.duration; + + // The recipe is recorded per FILE, so the shipped cut is the one it describes. + // A named plan in build.json is a fact; the ?plan= override is the user's + // choice where no builder recorded one. + const build = cut.shipped + ? await buildStatus(id, cut.shipped.rel) + : { entry: null, stale: null }; + const prov = await readProvenance(id, plan ?? build.entry?.plans[0] ?? null); + const notes = await readNotes(id); + const compares = await readCompares(id); + + // Cached figures only. Measuring here would put an ffmpeg decode per file in + // front of every navigation to this page. + const spec = await readSpec(id); + const loudness = await Promise.all( + rels.map(async (rel) => { + const abs = resolveRendition(id, rel); + return { rel, loudness: abs ? await cachedLoudness(abs) : null }; + }), + ); + const target = { + lufs: spec.loudness?.targetLufs ?? DEFAULT_TARGET.lufs, + truePeak: spec.loudness?.truePeak ?? DEFAULT_TARGET.truePeak, + }; + + return ( + <div className="flex h-full flex-col"> + <BrowseHeader + crumbs={[ + { href: "/browse", label: "songs" }, + { href: `/browse/${song.id}`, label: song.id }, + { label: cutName }, + ]} + note={ + cut.shipped + ? `${fmtDur(durations[cut.shipped.rel] ?? 0)} · ${fmtBytes(cut.shipped.size)}` + : "not built" + } + /> + <main className="deck-main flex-1"> + {!cut.shipped && cut.variants.length === 0 ? ( + <div className="p-4 text-[12px] text-[var(--color-dim)]"> + Nothing here yet — no <code className="font-mono">{cutName}.mp4</code> and no variants + named for it.{" "} + <Link href={`/browse/${song.id}`} className="text-[var(--color-sel)] underline"> + back to {song.id} + </Link> + </div> + ) : ( + <> + <CutBench + song={song.id} + cut={cutName} + shipped={cut.shipped} + variants={cut.variants} + durations={durations} + initialVariant={v ?? null} + notes={notes} + compares={compares} + /> + <div className="px-4 pb-4"> + <LoudnessTable song={song.id} initial={loudness} target={target} /> + </div> + <div className="px-4 pb-4"> + <ProvenancePanel + song={song.id} + cut={cutName} + prov={prov} + build={build} + planParam={plan ?? null} + notes={notes} + allNotes={notesParam === "all"} + /> + </div> + </> + )} + </main> + </div> + ); +} diff --git a/umtool/components/projects/ProjectGrid.tsx b/umtool/components/projects/ProjectGrid.tsx @@ -0,0 +1,104 @@ +import Link from "next/link"; +import { fmtAgo } from "@/lib/format"; +import type { ProjectSummary } from "@/lib/project-types"; + +// --------------------------------------------------------------------------- +// One card per project, whatever kind it is. +// +// This file contains no kind ids and no per-kind branches, and that is the +// design working rather than an omission: a kind's summariser already rendered +// its own facts ("19 clips · 4m46s · 17 sources", "3/4 cuts · 2 variants") and +// its own flags. The grid lays out strings. Adding a kind changes nothing here. +// --------------------------------------------------------------------------- + +export default function ProjectGrid({ + projects, + counts, +}: { + projects: ProjectSummary[]; + counts?: Map<string, { blocking: number; open: number }>; +}) { + if (projects.length === 0) { + return ( + <p className="text-[12px] text-[var(--color-dim)]">Nothing matches that filter.</p> + ); + } + + return ( + <ul className="grid gap-3 sm:grid-cols-2 lg:grid-cols-3 2xl:grid-cols-4"> + {projects.map((p) => { + const c = counts?.get(p.id); + // An unroutable or shadowed project keeps its card and its link -- the + // link just goes to the escape hatch instead of to an address that + // renders something else. + const href = p.routing === "ok" ? `/browse/${p.id}` : `/browse/at?path=${encodeURIComponent(p.id)}`; + return ( + <li key={p.id}> + <Link + href={href} + data-project={p.id} + data-kind={p.kind} + data-template={p.template} + data-state={p.state} + data-routing={p.routing} + className="block overflow-hidden rounded border border-[var(--color-line)] bg-[var(--color-panel)] transition-colors hover:border-[var(--color-sel)]" + > + {/* eslint-disable-next-line @next/next/no-img-element */} + <img + src={`/api/browse/poster?project=${encodeURIComponent(p.id)}&w=640`} + alt="" + width={640} + height={360} + className="aspect-video w-full bg-[var(--color-panel-2)] object-cover" + /> + <div + className="space-y-1.5 p-3" + {...Object.fromEntries( + Object.entries(p.attrs ?? {}).map(([k, v]) => [`data-${k}`, v]), + )} + > + <div className="flex items-baseline gap-2"> + <span className="rounded border border-[var(--color-line)] px-1.5 py-0.5 text-[10px] uppercase tracking-wider text-[var(--color-dim)]"> + {p.badge} + </span> + <span className="truncate text-[13px] font-medium text-[var(--color-text)]"> + {p.title} + </span> + </div> + {p.subtitle && ( + <div className="truncate text-[11px] text-[var(--color-dim)]">{p.subtitle}</div> + )} + <div className="num flex flex-wrap items-center gap-x-3 gap-y-1 text-[11px] text-[var(--color-dim)]"> + {p.facts.map((f) => ( + <span key={f}>{f}</span> + ))} + <span className="ml-auto">{fmtAgo(p.newestMtimeMs)}</span> + </div> + <div className="flex flex-wrap items-center gap-2 text-[11px]"> + <span className="micro" data-project-state={p.state}> + {p.state} + </span> + {c && c.blocking > 0 && ( + <span className="text-[var(--color-bad)]" data-blocking={c.blocking}> + {c.blocking} blocking + </span> + )} + {c && c.open > 0 && ( + <span className="text-[var(--color-dirty)]" data-open={c.open}> + {c.open} open + </span> + )} + </div> + {p.flags.length > 0 && ( + <div className="text-[11px] text-[var(--color-bad)]" data-flags={p.flags.join(",")}> + {p.flags.join(" · ")} + </div> + )} + </div> + </Link> + </li> + ); + })} + </ul> + ); +} diff --git a/umtool/components/projects/ProjectView.tsx b/umtool/components/projects/ProjectView.tsx @@ -0,0 +1,70 @@ +import { notFound } from "next/navigation"; +import type { ProjectRef } from "@/lib/project-types"; +import SongProject from "./SongProject"; +import CutPage from "./CutPage"; +import ReportProject from "./ReportProject"; +import ClipBenchPage from "./ClipBenchPage"; +import ClaimBenchPage from "./ClaimBenchPage"; +import SweepProject from "./SweepProject"; + +// --------------------------------------------------------------------------- +// The one place a kind id is matched against a component. +// +// It is HERE rather than in app/browse/[...path]/page.tsx deliberately: the +// rule is that adding a kind costs a registry entry and one view, and an e2e +// spec enforces it by failing when a kind id appears as a string literal +// anywhere outside lib/projects/ and components/projects/. A page that grew an +// `if (kind === "report-video")` would break that rule silently, so the page +// never learns what a kind is -- it resolves a path and renders this. +// +// `rest` is the segments after the project. [] is the project itself; anything +// else is a VIEW the kind declares, and a view a kind does not declare is a 404 +// rather than a page that ignores half its URL. +// --------------------------------------------------------------------------- + +export default async function ProjectView({ + project, + rest, + search, +}: { + project: ProjectRef; + rest: string[]; + search: Record<string, string | undefined>; +}) { + switch (project.kind) { + case "song": { + if (rest.length === 0) return <SongProject id={project.name} />; + if (rest.length === 1) return <CutPage id={project.name} cutName={rest[0]} search={search} />; + return notFound(); + } + case "report-video": { + if (rest.length === 0) return <ReportProject project={project} search={search} />; + // `/browse/<project>/clip/<id>` -- the bench. A view the kind does not + // declare is a 404 rather than a page that silently drops half its URL. + if (rest.length === 2 && rest[0] === "clip") { + return <ClipBenchPage project={project} clipId={rest[1]} />; + } + // `/browse/<project>/claim/<id>` -- ruling on one ledger entry against + // the audio, rather than editing a window. + if (rest.length === 2 && rest[0] === "claim") { + return <ClaimBenchPage project={project} claimId={rest[1]} />; + } + return notFound(); + } + case "sweep-report": { + if (rest.length === 0) return <SweepProject project={project} />; + return notFound(); + } + default: + // A kind the registry knows and this file does not. That is exactly the + // state a half-added kind is in, so it says so instead of 404ing. + return ( + <main className="deck-main flex-1 p-4"> + <p className="text-[12px] text-[var(--color-dim)]"> + <code className="font-mono">{project.kind}</code> is a registered kind with no view + yet. Add one in <code className="font-mono">components/projects/</code>. + </p> + </main> + ); + } +} diff --git a/umtool/components/projects/ReportBuildChain.tsx b/umtool/components/projects/ReportBuildChain.tsx @@ -0,0 +1,308 @@ +"use client"; + +import { useCallback, useEffect, useRef, useState } from "react"; +import { badgeVariants } from "@/components/ui/badge"; +import { buttonVariants } from "@/components/ui/button"; + +// --------------------------------------------------------------------------- +// Running a build, and watching it per clip. +// +// BuildChain's step boxes are the right model for the chain -- preflight, +// resolve, build, verify -- but the build step alone is twenty minutes of work +// on nineteen clips, and "step 3 of 4, running" is not progress. So the build +// step also renders a grid: one box per timeline entry, lit by the NDJSON events +// the pipeline emits (fetch / snap / segment / entry-failed). +// +// Polling, not streaming, like every other job here. +// --------------------------------------------------------------------------- + +type StepView = { label: string; argv: string[]; cwd: string; timeoutMs: number | null }; +type Ev = Record<string, unknown> & { ev: string; id?: string }; +type JobView = { + id: string; + state: "running" | "done" | "failed"; + stepIndex: number; + steps: StepView[]; + error: string | null; + log: string[]; + next: number; + events: Ev[]; + nextEvent: number; +}; + +type Preset = { id: string; label: string }; + +const CLIP_STATE = { + pending: "border-[var(--color-line)] text-[var(--color-dim)]", + fetching: "border-[var(--color-meter)] text-[var(--color-meter)]", + cutting: "border-[var(--color-sel)] text-[var(--color-sel)]", + done: "border-[var(--color-good)] text-[var(--color-good)]", + failed: "border-[var(--color-bad)] text-[var(--color-bad)]", +} as const; +type ClipState = keyof typeof CLIP_STATE; + +export default function ReportBuildChain({ + project, + entries, +}: { + project: string; + entries: { id: string; kind: string }[]; +}) { + const [presets, setPresets] = useState<Preset[]>([]); + const [preset, setPreset] = useState("fast"); + const [only, setOnly] = useState(""); + const [dry, setDry] = useState<StepView[] | null>(null); + const [job, setJob] = useState<JobView | null>(null); + const [events, setEvents] = useState<Ev[]>([]); + const [error, setError] = useState<string | null>(null); + const [needsReplace, setNeedsReplace] = useState(false); + const [busy, setBusy] = useState(false); + const since = useRef(0); + const sinceEvent = useRef(0); + + useEffect(() => { + void fetch("/api/report/build", { cache: "no-store" }) + .then((r) => r.json()) + .then((j) => { + setPresets(j.presets ?? []); + // Adopt a build already running, so a reload does not lose it. + if (j.running) { + setJob(j.running as JobView); + setEvents((j.running as JobView).events ?? []); + } + }) + .catch(() => {}); + }, []); + + useEffect(() => { + if (!job || job.state !== "running") return; + const t = setInterval(async () => { + const r = await fetch( + `/api/report/build?job=${job.id}&since=${since.current}&sinceEvent=${sinceEvent.current}`, + { cache: "no-store" }, + ); + if (!r.ok) return; + const j = (await r.json()) as JobView; + since.current = j.next; + sinceEvent.current = j.nextEvent; + setEvents((prev) => [...prev, ...(j.events ?? [])]); + setJob((prev) => (prev ? { ...j, log: [...prev.log, ...j.log] } : j)); + }, 800); + return () => clearInterval(t); + }, [job]); + + const post = useCallback( + async (qs: string, extra: Record<string, unknown> = {}) => { + setBusy(true); + setError(null); + const r = await fetch(`/api/report/build${qs}`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ project, preset, only: only || null, ...extra }), + }); + const j = (await r.json()) as Record<string, unknown>; + setBusy(false); + if (!r.ok) { + setError(String(j.error ?? r.status)); + setNeedsReplace(!!j.needsReplace); + return null; + } + setNeedsReplace(false); + return j; + }, + [project, preset, only], + ); + + const clipStates = new Map<string, ClipState>(); + for (const e of events) { + const id = e.id as string | undefined; + if (!id) continue; + if (e.ev === "fetch") clipStates.set(id, e.cached ? "cutting" : "fetching"); + else if (e.ev === "clip" || e.ev === "card") clipStates.set(id, "cutting"); + else if (e.ev === "snap") clipStates.set(id, "cutting"); + else if (e.ev === "segment") clipStates.set(id, "done"); + else if (e.ev === "entry-failed") clipStates.set(id, "failed"); + } + const failed = events.filter((e) => e.ev === "entry-failed"); + + return ( + <section className="space-y-2 rounded border border-[var(--color-line)] bg-[var(--color-panel)] p-3"> + <div className="flex flex-wrap items-center gap-2"> + <span className="micro">build</span> + <select + value={preset} + onChange={(e) => setPreset(e.target.value)} + data-preset="" + className="rounded border border-[var(--color-line)] bg-[var(--color-panel-2)] px-2 py-1 text-[12px]" + > + {presets.map((p) => ( + <option key={p.id} value={p.id}> + {p.label} + </option> + ))} + </select> + {preset === "preview" && ( + <select + value={only} + onChange={(e) => setOnly(e.target.value)} + data-only="" + className="rounded border border-[var(--color-line)] bg-[var(--color-panel-2)] px-2 py-1 font-mono text-[12px]" + > + <option value="">— which clip —</option> + {entries.filter((e) => e.kind === "clip").map((e) => ( + <option key={e.id} value={e.id}> + {e.id} + </option> + ))} + </select> + )} + + <button + type="button" + data-action="dry" + className={buttonVariants({ size: "sm" })} + disabled={busy} + onClick={async () => { + const j = await post("?dry=1"); + if (j) setDry(j.steps as StepView[]); + }} + > + show the command + </button> + <button + type="button" + data-action="run" + className={buttonVariants({ variant: "primary", size: "sm" })} + disabled={busy || job?.state === "running"} + onClick={async () => { + const j = await post(""); + if (j?.job) { + since.current = 0; + sinceEvent.current = 0; + setEvents([]); + setJob(j.job as JobView); + } + }} + > + run + </button> + {job?.state === "running" && ( + <button + type="button" + data-action="cancel" + className={buttonVariants({ variant: "destructive", size: "sm" })} + onClick={() => void fetch(`/api/report/build?cancel=${job.id}`, { method: "POST" })} + > + cancel + </button> + )} + </div> + + {error && ( + <p data-build-error="" className="text-[12px] text-[var(--color-bad)]"> + {error} + {needsReplace && ( + <> + {" "} + <button + type="button" + data-action="replace" + className={buttonVariants({ variant: "destructive", size: "sm" })} + onClick={async () => { + const j = await post("?replace=1"); + if (j?.job) { + since.current = 0; + sinceEvent.current = 0; + setEvents([]); + setJob(j.job as JobView); + } + }} + > + stamp it aside and build + </button> + </> + )} + </p> + )} + + {dry && !job && ( + <pre + data-dry="" + className="max-h-56 overflow-auto rounded border border-[var(--color-line)] bg-[var(--color-ink)] p-2 font-mono text-[11px] text-[var(--color-dim)]" + > + {dry.map((s) => `# ${s.label}\n${s.argv.join(" ")}\n`).join("\n")} + </pre> + )} + + {job && ( + <div className="space-y-2"> + <ol className="flex flex-wrap gap-1.5"> + {job.steps.map((s, i) => ( + <li + key={s.label} + data-step={i} + data-step-state={ + job.state !== "running" && i <= job.stepIndex + ? job.state + : i < job.stepIndex + ? "done" + : i === job.stepIndex + ? "running" + : "pending" + } + className={badgeVariants({ + variant: + i < job.stepIndex || (job.state === "done" && i <= job.stepIndex) + ? "meter" + : i === job.stepIndex && job.state === "running" + ? "on" + : job.state === "failed" && i === job.stepIndex + ? "blocking" + : "neutral", + })} + > + {s.label} + </li> + ))} + </ol> + + {/* One box per entry, lit by the pipeline's own events. "Step 3 of 4, + running" is not progress when step 3 is twenty minutes long. */} + <div className="flex flex-wrap gap-1"> + {entries.map((e) => { + const st = clipStates.get(e.id) ?? "pending"; + return ( + <span + key={e.id} + data-clip-box={e.id} + data-clip-state={st} + className={`rounded border px-1.5 py-0.5 font-mono text-[10px] ${CLIP_STATE[st]}`} + > + {e.id} + </span> + ); + })} + </div> + + {failed.length > 0 && ( + <p data-build-failed="" className="text-[12px] text-[var(--color-bad)]"> + {failed.length} entr{failed.length === 1 ? "y" : "ies"} failed ( + {failed.map((f) => String(f.id)).join(", ")}). Everything buildable was built and the + segments are kept — but the timeline was NOT concatenated, because a finished file + quietly missing a citation looks complete. + </p> + )} + + <pre className="max-h-56 overflow-auto rounded border border-[var(--color-line)] bg-[var(--color-ink)] p-2 font-mono text-[11px] text-[var(--color-dim)]"> + {job.log.slice(-120).join("\n")} + </pre> + + <div className="micro" data-job-state={job.state}> + {job.state} + {job.error ? ` — ${job.error}` : ""} + </div> + </div> + )} + </section> + ); +} diff --git a/umtool/components/projects/ReportProject.tsx b/umtool/components/projects/ReportProject.tsx @@ -0,0 +1,338 @@ +import path from "node:path"; +import Link from "next/link"; +import BrowseHeader from "@/components/BrowseHeader"; +import { fmtAgo, fmtBytes } from "@/lib/format"; +import { readClipDetail } from "@/lib/projects/report.mjs"; +import ReportBuildChain from "./ReportBuildChain"; +import { decisionsForProject } from "@/lib/projects"; +import { badgeVariants, type BadgeVariants } from "@/components/ui/badge"; +import type { Severity } from "@/lib/decisions"; +import type { ProjectRef } from "@/lib/project-types"; + +// --------------------------------------------------------------------------- +// A report video, as a page. +// +// The manifest IS the cut -- array order, absolute source seconds, one entry per +// clip -- so the page is the timeline, read back. What it adds is the three +// things the JSON cannot show you: whether each clip's material is actually on +// disk, whether its window ends where a sentence does, and what the widener +// would do to it if you ran it. +// --------------------------------------------------------------------------- + +const hms = (t: number) => { + const s = Math.max(0, Math.floor(t)); + const h = Math.floor(s / 3600); + const m = Math.floor((s % 3600) / 60); + const sec = s % 60; + return h > 0 + ? `${h}:${String(m).padStart(2, "0")}:${String(sec).padStart(2, "0")}` + : `${m}:${String(sec).padStart(2, "0")}`; +}; + +const TONE: Record<Severity, BadgeVariants["variant"]> = { + blocking: "blocking", + open: "open", + info: "info", +}; + +function Pill({ tone, children }: { tone?: BadgeVariants["variant"]; children: React.ReactNode }) { + return <span className={badgeVariants({ variant: tone ?? "info", size: "sm" })}>{children}</span>; +} + +export default async function ReportProject({ + project, + search, +}: { + project: ProjectRef; + search: Record<string, string | undefined>; +}) { + const detail = await readClipDetail(project.dir); + const decisions = await decisionsForProject(project); + + if (!detail) { + return ( + <div className="flex h-full flex-col"> + <BrowseHeader crumbs={[{ href: "/browse", label: "projects" }, { label: project.id }]} /> + <main className="deck-main flex-1 p-4"> + <p className="text-[12px] text-[var(--color-bad)]"> + <code className="font-mono">video.manifest.json</code> could not be parsed. + </p> + </main> + </div> + ); + } + + const { manifest: m, build, entries, channelsDir, shadowExists } = detail; + const clips = entries.filter((e) => e.kind === "clip"); + const nonClips = entries.filter((e) => e.kind !== "clip"); + const runtime = clips.reduce((n, e) => n + Math.max(0, e.end - e.start), 0); + const showAll = search.all === "1"; + const p = m.provenance ?? {}; + + // ---- what a clip's "mix" link opens --------------------------------------- + // + // The widest cached RAW window, not the built segment. Three reasons: it + // exists as soon as a clip has been fetched once (segments only exist after a + // build), it has NO CHROME burned in -- which is what a mix is looking at -- + // and it is the file the bench already has peaks for. The built segment is the + // fallback, and the link says which it is. + // + // start/end are the clip's window MINUS the file's own start, because a mix + // spec is relative to the file it names. The arithmetic is exact and known + // here; making the client do it would be a second place to get it wrong. + const mixHref = (e: { + id: string; + start: number; + end: number; + widest: { name: string; from: number } | null; + segment: string | null; + }): string | null => { + const q = new URLSearchParams(); + if (e.widest) { + q.set("body", path.join(project.dir, "out", "clips-raw", e.widest.name)); + q.set("start", (e.start - e.widest.from).toFixed(2)); + q.set("end", (e.end - e.widest.from).toFixed(2)); + } else if (e.segment) { + q.set("body", path.join(project.dir, e.segment)); + } else { + return null; + } + q.set("from", project.id); + q.set("clip", e.id); + return `/mix?${q}`; + }; + + return ( + <div className="flex h-full flex-col"> + <BrowseHeader + crumbs={[{ href: "/browse", label: "projects" }, { label: project.id }]} + note={ + `${clips.length} clips` + + (nonClips.length ? ` · ${nonClips.length} other` : "") + + ` · ${hms(runtime)} · ${build.built ? "built" : "not built"}` + } + /> + + <div className="flex flex-wrap items-start gap-3 border-b border-[var(--color-line)] px-4 py-2"> + <div className="min-w-0 flex-1"> + <h1 className="truncate text-[15px] text-[var(--color-text)]">{m.title}</h1> + {m.subtitle && <p className="text-[12px] text-[var(--color-dim)]">{m.subtitle}</p>} + <div className="num mt-1 flex flex-wrap items-center gap-x-3 text-[11px] text-[var(--color-dim)]"> + {p.channel && <span>{p.channel}</span>} + {m.generatedOn && <span>generated {m.generatedOn}</span>} + {typeof p.videosCited === "number" && <span>{p.videosCited} videos cited</span>} + {typeof p.enumeratedMatches === "number" && ( + <span>{p.enumeratedMatches} enumerated matches</span> + )} + </div> + </div> + {build.built && ( + <div className="text-right text-[11px] text-[var(--color-dim)]"> + <div className="font-mono text-[var(--color-text)]">{build.slug}.mp4</div> + <div className="num"> + {fmtBytes(build.finalSize)} · {fmtAgo(build.finalMtimeMs)} + </div> + {/* No window: the whole thing is the point of a deliverable link. */} + <Link + href={`/mix?body=${encodeURIComponent(build.finalPath)}&from=${encodeURIComponent(project.id)}`} + data-mix-deliverable="" + className="text-[var(--color-sel)] hover:underline" + > + open in mix + </Link> + </div> + )} + </div> + + <main className="deck-main flex-1 space-y-4 p-4"> + {/* --- what is wrong, first ------------------------------------- */} + {decisions.length > 0 && ( + <section data-project={project.id}> + <h2 className="micro mb-1.5"> + {decisions.filter((d) => d.severity !== "info").length} waiting of {decisions.length} + </h2> + <ul className="space-y-1"> + {decisions.map((d, i) => ( + <li + key={`${d.kind}-${d.target}-${i}`} + data-decision={d.kind} + data-severity={d.severity} + data-target={d.target} + className="flex flex-wrap items-baseline gap-2 rounded border border-[var(--color-line)] bg-[var(--color-panel)] px-3 py-1.5" + > + <span className={badgeVariants({ variant: TONE[d.severity], size: "sm" })}> + {d.severity} + </span> + <span className="font-mono text-[12px] text-[var(--color-text)]">{d.target}</span> + <span className="text-[12px] text-[var(--color-dim)]">{d.why}</span> + <span className="micro ml-auto">{d.kind}</span> + </li> + ))} + </ul> + </section> + )} + + {/* --- where the sources are read from --------------------------- */} + <section className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] px-3 py-2"> + <div className="micro mb-1">sources</div> + <div className="num text-[11px] text-[var(--color-dim)]"> + cues from <code className="font-mono">{channelsDir}</code> + {shadowExists && ( + <span className="ml-2 text-[var(--color-meter)]"> + (this project&rsquo;s own shadow tree) + </span> + )} + </div> + <div className="num mt-1 text-[11px] text-[var(--color-dim)]"> + QR codes resolve to{" "} + <code className="font-mono">{p.siteOrigin ?? "(nothing — siteOrigin is unset)"}</code> + </div> + </section> + + {/* --- building it ---------------------------------------------- */} + <ReportBuildChain + project={project.id} + entries={entries.map((e) => ({ id: e.id, kind: e.kind }))} + /> + + {/* --- the timeline --------------------------------------------- */} + <section> + <h2 className="micro mb-1.5"> + the cut — {entries.length} entries, in array order + </h2> + <ul className="space-y-1"> + {entries.map((e) => { + // Anything that is not a CLIP renders generically. The timeline's + // vocabulary is open -- one real manifest carries `scroll` and + // `chart` beside its cards -- and a page that only knows two words + // either crashes on the third or silently drops it. + if (e.kind !== "clip") { + return ( + <li + key={e.id} + data-entry={e.id} + data-kind={e.kind} + className="flex flex-wrap items-baseline gap-2 rounded border border-dashed border-[var(--color-line)] px-3 py-1.5 text-[12px]" + > + <span className="font-mono text-[var(--color-dim)]">{e.id}</span> + <Pill>{e.style ? `${e.kind} · ${e.style}` : e.kind}</Pill> + <span className="text-[var(--color-text)]"> + {e.heading ?? e.title ?? e.label ?? ""} + </span> + {e.seconds != null && <span className="num micro ml-auto">{e.seconds}s</span>} + </li> + ); + } + // `endsSentence === null` means the source has no punctuation to + // read, which is a different thing to say than "this cut is fine". + const midSentence = e.endsSentence === false && !e.lockEnd && !e.lock; + return ( + <li + key={e.id} + data-entry={e.id} + data-kind="clip" + data-cached={e.cached ? "1" : "0"} + data-mid-sentence={midSentence ? "1" : "0"} + className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] px-3 py-1.5" + > + <div className="flex flex-wrap items-baseline gap-2 text-[12px]"> + <span className="font-mono text-[var(--color-sel)]">{e.id}</span> + <span className="num font-mono text-[11px] text-[var(--color-dim)]"> + {e.video} {hms(e.start)}–{hms(e.end)} ({(e.end - e.start).toFixed(1)}s) + </span> + {e.lock && <Pill>locked</Pill>} + {!e.lock && e.lockStart && <Pill>start pinned</Pill>} + {!e.lock && e.lockEnd && <Pill>end pinned</Pill>} + {e.cached ? ( + <Pill>cached</Pill> + ) : ( + <Pill tone="open"> + not fetched + </Pill> + )} + {e.segment && <Pill>segment built</Pill>} + {!e.hasCues && ( + <Pill tone={TONE.blocking}>no cues</Pill> + )} + {midSentence && <Pill tone="open">ends mid-sentence</Pill>} + {e.noPunctuation && <Pill>source unpunctuated</Pill>} + {e.proposed && ( + <Pill tone="open"> + widener would move it + </Pill> + )} + <span className="ml-auto flex items-center gap-2"> + {mixHref(e) ? ( + <Link + href={mixHref(e)!} + data-mix-link={e.id} + className="text-[11px] text-[var(--color-sel)] hover:underline" + > + mix + </Link> + ) : ( + // Never a dead link that 400s: a clip with nothing + // fetched has nothing to open. + <span className="micro" data-mix-link-disabled={e.id}> + fetch it first + </span> + )} + <Link + href={`/browse/${project.id}/clip/${e.id}`} + data-bench-link={e.id} + className="text-[11px] text-[var(--color-sel)] hover:underline" + > + bench → + </Link> + <span className="micro">§{e.section ?? 0}</span> + </span> + </div> + {(showAll || midSentence || e.proposed) && e.quote && ( + <p className="mt-1 text-[11px] leading-snug text-[var(--color-dim)]"> + &ldquo;{String(e.quote).slice(0, 240)} + {String(e.quote).length > 240 ? "…" : ""}&rdquo; + </p> + )} + {midSentence && e.endCueText && ( + <p className="num mt-1 text-[11px] text-[var(--color-dirty)]"> + cut lands inside: &ldquo;…{String(e.endCueText).trim().slice(-64)}&rdquo; + </p> + )} + {e.proposed && ( + <p className="num mt-1 text-[11px] text-[var(--color-dirty)]"> + resolve-windows would make it {hms(e.proposed.start)}–{hms(e.proposed.end)} — + set <code className="font-mono">lock</code> if this window is deliberate + </p> + )} + </li> + ); + })} + </ul> + <div className="mt-2"> + <Link + href={`/browse/${project.id}${showAll ? "" : "?all=1"}`} + className="text-[11px] text-[var(--color-sel)] hover:underline" + > + {showAll ? "hide quotes" : "show every quote"} + </Link> + </div> + </section> + + {/* --- provenance, as written ------------------------------------ */} + <section> + <h2 className="micro mb-1.5">provenance</h2> + <dl className="space-y-1.5 rounded border border-[var(--color-line)] bg-[var(--color-panel)] px-3 py-2 text-[11px]"> + {Object.entries(p) + .filter(([, v]) => typeof v === "string" && v.length > 40) + .map(([k, v]) => ( + <div key={k}> + <dt className="micro">{k}</dt> + <dd className="text-[var(--color-dim)]">{String(v)}</dd> + </div> + ))} + </dl> + </section> + </main> + </div> + ); +} diff --git a/umtool/components/projects/SongProject.tsx b/umtool/components/projects/SongProject.tsx @@ -0,0 +1,255 @@ +import Link from "next/link"; +import { notFound } from "next/navigation"; +import BrowseHeader from "@/components/BrowseHeader"; +import VerdictChip from "@/components/VerdictChip"; +import SpecSheet from "@/components/SpecSheet"; +import NoteField from "@/components/NoteField"; +import CopyButton from "@/components/CopyButton"; +import { readSong, probeAll, thumbAliasesFor, type Cut, type Song } from "@/lib/browse"; +import ThumbBench from "@/components/ThumbBench"; +import { thumbView } from "@/lib/thumbs"; +import { fmtAgo, fmtBytes, fmtDur } from "@/lib/format"; +import { Markdown } from "@/lib/markdown"; +import { operationsFor, readSpec, validateSpec } from "@/lib/spec"; +import { readNotes, type NoteMap } from "@/lib/notes"; +import { TRIM_SETS } from "@/lib/trim"; + +// Moved out of app/browse/[song]/page.tsx unchanged, because [song] and +// [...path] cannot both be dynamic segments at the same level. The dispatcher +// in ProjectView.tsx hands it an id; nothing else about the page differs. + +export default async function SongProject({ id }: { id: string }) { + const song = await readSong(id); + if (!song) notFound(); + + // Durations are worth an ffprobe HERE but not on the index: this page is a + // handful of files and "which of these is the short cut" is exactly the + // question it answers. Memoised by mtime in lib/browse.ts. + const rels = [ + ...song.cuts.flatMap((c) => [c.shipped?.rel, ...c.variants.map((v) => v.rel)]), + ...song.unattributed.map((v) => v.rel), + ].filter((r): r is string => !!r); + const info = await probeAll(id, rels); + + const spec = await readSpec(id); + const problems = await validateSpec(spec); + const notes = await readNotes(id); + // Two small JSON reads, no probing -- the bench draws what the manifests say. + const thumbs = await thumbView(id, thumbAliasesFor(id)); + + return ( + <div className="flex h-full flex-col"> + <BrowseHeader + crumbs={[{ href: "/browse", label: "songs" }, { label: song.id }]} + note={`${song.present}/${song.cuts.length} cuts · ${song.variantCount} variants`} + /> + <div className="flex flex-wrap items-start gap-3 border-b border-[var(--color-line)] px-4 py-2"> + <div className="min-w-0 flex-1"> + {/* The one obvious place to write about the song, open by default -- + everything else on the page is collapsed until asked for. */} + <NoteField song={id} target="song" initial={notes.song ?? null} label="notes on this song" /> + </div> + <CopyButton + label="copy this song" + title="this song as markdown — spec, cuts, verdicts, notes and resolved marks" + url={`/api/browse/context?song=${encodeURIComponent(id)}`} + /> + </div> + <main className="deck-main flex-1"> + <div className="grid gap-4 p-4 xl:grid-cols-[minmax(0,2fr)_minmax(0,1fr)]"> + <div className="space-y-3"> + {song.cuts.map((cut) => ( + <CutCard key={cut.name} song={song} cut={cut} info={info} notes={notes} /> + ))} + + {song.unattributed.length > 0 && ( + <section className="rounded border border-[var(--color-dirty)]/40 bg-[var(--color-panel)] p-3"> + <div className="micro mb-2"> + matches no cut name — attribute by renaming, never by guessing + </div> + <ul className="space-y-1"> + {song.unattributed.map((v) => ( + <li key={v.rel} className="num text-[12px] text-[var(--color-dim)]"> + <span className="font-mono text-[var(--color-text)]">{v.rel}</span>{" "} + {fmtBytes(v.size)} + </li> + ))} + </ul> + </section> + )} + </div> + + <aside className="space-y-4"> + <ThumbBench song={song.id} view={thumbs} notes={notes} /> + <SpecSheet + song={song.id} + initial={spec} + problems={problems} + operations={operationsFor(spec, song)} + plans={song.plans.map((p) => p.name)} + trimSets={Object.values(TRIM_SETS).map((t) => ({ id: t.id, label: t.label }))} + /> + {song.readme && ( + <section className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] p-3"> + <Markdown text={song.readme} /> + </section> + )} + <section className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] p-3"> + <div className="micro mb-2">plans</div> + {song.plans.length === 0 ? ( + <p className="text-[12px] text-[var(--color-dim)]">no plan/ directory</p> + ) : ( + <ul className="space-y-0.5"> + {song.plans.map((p) => ( + <li key={p.name} className="num text-[11px] text-[var(--color-dim)]"> + <span className="flex gap-2"> + <span className="truncate font-mono text-[var(--color-text)]">{p.name}</span> + <span className="ml-auto shrink-0">{fmtBytes(p.size)}</span> + </span> + <NoteField + song={song.id} + target={`plan:${p.name}`} + initial={notes[`plan:${p.name}`] ?? null} + label="what this plan is" + rows={3} + /> + </li> + ))} + </ul> + )} + {song.hasClipsCsv && <div className="micro mt-2">clips.csv present</div>} + </section> + </aside> + </div> + </main> + </div> + ); +} + +function CutCard({ + song, + cut, + info, + notes, +}: { + song: Song; + cut: Cut; + info: Record<string, { duration: number; width: number; height: number }>; + notes: NoteMap; +}) { + const shipped = cut.shipped; + const live = cut.variants.filter((v) => !v.retired); + const retired = cut.variants.filter((v) => v.retired); + + return ( + <section + data-cut={cut.name} + data-present={shipped ? "1" : "0"} + className="rounded border border-[var(--color-line)] bg-[var(--color-panel)]" + > + <div className="flex items-start gap-3 p-3"> + {shipped ? ( + /* eslint-disable-next-line @next/next/no-img-element */ + <img + src={`/api/browse/poster?song=${encodeURIComponent(song.id)}&rel=${encodeURIComponent(shipped.rel)}&w=320`} + alt="" + width={160} + height={90} + className="w-40 shrink-0 rounded bg-[var(--color-panel-2)] object-cover" + /> + ) : ( + <div className="flex h-[90px] w-40 shrink-0 items-center justify-center rounded border border-dashed border-[var(--color-line)] text-[11px] text-[var(--color-dim)]"> + not built + </div> + )} + + <div className="min-w-0 flex-1 space-y-1"> + <div className="flex flex-wrap items-baseline gap-2"> + <Link + href={`/browse/${song.id}/${cut.name}`} + className="font-mono text-[13px] text-[var(--color-text)] hover:text-[var(--color-sel)]" + > + {cut.name} + </Link> + {shipped ? ( + <span className="num text-[11px] text-[var(--color-meter)]"> + {fmtDur(info[shipped.rel]?.duration ?? 0)} + </span> + ) : ( + <span className="text-[11px] text-[var(--color-dim)]">— a hole in the set</span> + )} + {shipped && info[shipped.rel] && ( + <span className="num text-[11px] text-[var(--color-dim)]"> + {info[shipped.rel].width}×{info[shipped.rel].height} + </span> + )} + <span className="num ml-auto text-[11px] text-[var(--color-dim)]"> + {shipped ? `${fmtBytes(shipped.size)} · ${fmtAgo(shipped.mtimeMs)}` : ""} + </span> + </div> + + {shipped && ( + <VerdictChip song={song.id} rel={shipped.rel} initial={shipped.verdict} /> + )} + + {/* Two different notes, deliberately. The `cut:` one is about the SLOT + -- it survives a promote and can be written about a cut that has + not been built. The `file:` one is about these bytes. */} + <NoteField + song={song.id} + target={`cut:${cut.name}`} + initial={notes[`cut:${cut.name}`] ?? null} + label={`notes on ${cut.name}`} + rows={3} + /> + {shipped && ( + <NoteField + song={song.id} + target={`file:${shipped.rel}`} + initial={notes[`file:${shipped.rel}`] ?? null} + label="notes on this file" + rows={3} + /> + )} + + {live.length > 0 && ( + <ul className="space-y-1 pt-1"> + {live.map((v) => ( + <li + key={v.rel} + data-variant={v.rel} + data-variant-tag={v.tag} + className="flex flex-wrap items-center gap-2 rounded bg-[var(--color-panel-2)] px-2 py-1" + > + <span className="font-mono text-[12px] text-[var(--color-text)]">{v.tag}</span> + <span className="num text-[11px] text-[var(--color-meter)]"> + {fmtDur(info[v.rel]?.duration ?? 0)} + </span> + <span className="num text-[11px] text-[var(--color-dim)]">{fmtBytes(v.size)}</span> + <div className="ml-auto"> + <VerdictChip song={song.id} rel={v.rel} initial={v.verdict} /> + </div> + <div className="w-full"> + <NoteField + song={song.id} + target={`file:${v.rel}`} + initial={notes[`file:${v.rel}`] ?? null} + label={`notes on ${v.tag}`} + rows={3} + /> + </div> + </li> + ))} + </ul> + )} + + {retired.length > 0 && ( + <div className="micro pt-1" data-retired={retired.length}> + {retired.length} retired: {retired.map((v) => v.tag).join(", ")} + </div> + )} + </div> + </div> + </section> + ); +} diff --git a/umtool/components/projects/SweepProject.tsx b/umtool/components/projects/SweepProject.tsx @@ -0,0 +1,73 @@ +import Link from "next/link"; +import { readdir, readFile, stat } from "node:fs/promises"; +import path from "node:path"; +import BrowseHeader from "@/components/BrowseHeader"; +import { Markdown } from "@/lib/markdown"; +import { fmtAgo } from "@/lib/format"; +import type { ProjectRef } from "@/lib/project-types"; + +// A report that is not yet a video. There is nothing to judge and nothing to +// build, so the page is the report itself plus the one action that matters: +// what it would take to make it a cut. + +const SWEEP_RE = /(^|[-_])sweep([-_]report)?\.md$|^sweep-report\.md$/i; + +export default async function SweepProject({ project }: { project: ProjectRef }) { + const names = await readdir(project.dir).catch(() => [] as string[]); + const reportName = names.find((n) => SWEEP_RE.test(n)); + const file = reportName ? path.join(project.dir, reportName) : null; + const [text, st] = await Promise.all([ + file ? readFile(file, "utf8").catch(() => null) : null, + file ? stat(file).catch(() => null) : null, + ]); + const citations = text ? (text.match(/\]\([^)]*[?&]v=/g) ?? []).length : 0; + + return ( + <div className="flex h-full flex-col"> + <BrowseHeader + crumbs={[{ href: "/browse", label: "projects" }, { label: project.id }]} + note={`${citations} citation${citations === 1 ? "" : "s"} · no manifest`} + /> + <main className="deck-main flex-1 space-y-4 p-4"> + <section className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] p-3"> + <h2 className="mb-1 text-[13px] text-[var(--color-text)]">A report, not yet a video</h2> + <p className="text-[12px] text-[var(--color-dim)]"> + Turning it into one means writing a{" "} + <code className="font-mono">video.manifest.json</code> beside it: one entry per + clip, each with a window in absolute source seconds taken from{" "} + <code className="font-mono">transcript.cues.json</code>. A report only records a + single start second per citation, so the windows cannot be recovered from it alone — + that matching is the work. + </p> + <p className="mt-1.5 text-[12px] text-[var(--color-dim)]"> + {citations > 0 + ? `${citations} citation${citations === 1 ? "" : "s"} to work from.` + : "No `?v=` citations found, so there is nothing to match windows against yet."}{" "} + See <code className="font-mono">umtool/docs/authoring.md</code>. + </p> + </section> + + {text ? ( + <section className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] p-3"> + <div className="mb-2 flex items-baseline gap-2"> + <span className="font-mono text-[12px] text-[var(--color-text)]">{reportName}</span> + {st && <span className="micro">{fmtAgo(st.mtimeMs)}</span>} + </div> + <div className="max-h-[60vh] overflow-y-auto"> + <Markdown text={text} /> + </div> + </section> + ) : ( + <p className="text-[12px] text-[var(--color-dim)]"> + No report file found under{" "} + <code className="font-mono">{project.id}</code>. + </p> + )} + + <Link href="/browse" className="text-[12px] text-[var(--color-sel)] hover:underline"> + ← every project + </Link> + </main> + </div> + ); +} diff --git a/umtool/components/ui/badge.tsx b/umtool/components/ui/badge.tsx @@ -0,0 +1,49 @@ +import { cva, type VariantProps } from "class-variance-authority"; +import { cn } from "@/lib/utils"; + +// --------------------------------------------------------------------------- +// The first shadcn thing to land, and deliberately the smallest one. +// +// It is a CLASS HELPER, not a component: `badgeVariants({variant})` returns a +// string, so a <Link> on a zero-client-JS page can wear it without becoming a +// client component. /browse and /browse/decisions are server-rendered with +// their filters as plain links, and a Radix component would quietly end that. +// +// The variants name PALETTE MEANINGS rather than shadcn's defaults, because the +// palette's rule is that colour means something: `on` is the interaction colour +// (the only one), and blocking/open/info are the three severities the decisions +// inbox already speaks in. There is no `primary`/`secondary` pair here because +// this app has no such distinction to make. +// --------------------------------------------------------------------------- + +export const badgeVariants = cva( + "inline-flex items-center rounded border px-1.5 py-0.5 font-mono text-[11px] transition-colors", + { + variants: { + variant: { + neutral: "border-border text-muted-foreground hover:text-foreground", + on: "border-primary text-primary", + blocking: "border-destructive text-destructive", + open: "border-[var(--color-dirty)] text-[var(--color-dirty)]", + info: "border-border text-muted-foreground", + meter: "border-[var(--color-meter)] text-[var(--color-meter)]", + }, + size: { + sm: "px-1.5 py-0.5 text-[10px] uppercase tracking-wider", + md: "", + }, + }, + defaultVariants: { variant: "neutral", size: "md" }, + }, +); + +export type BadgeVariants = VariantProps<typeof badgeVariants>; + +export function Badge({ + className, + variant, + size, + ...props +}: React.ComponentProps<"span"> & BadgeVariants) { + return <span className={cn(badgeVariants({ variant, size }), className)} {...props} />; +} diff --git a/umtool/components/ui/button.tsx b/umtool/components/ui/button.tsx @@ -0,0 +1,42 @@ +import { cva, type VariantProps } from "class-variance-authority"; +import { cn } from "@/lib/utils"; + +// Same posture as badge.tsx: the variants are a class helper first, so a <Link> +// can wear them without hydrating, and a <button> second. +// +// Hit sizes are the ones this app already uses. It is a bench somebody sits +// with for hours, so nothing here is smaller than what the existing controls +// are, and `sm` is the default because dense is the point. + +export const buttonVariants = cva( + "inline-flex items-center justify-center gap-1.5 rounded border transition-colors disabled:pointer-events-none disabled:opacity-50 focus-visible:outline-none focus-visible:border-primary", + { + variants: { + variant: { + outline: "border-border text-muted-foreground hover:text-foreground hover:bg-accent", + primary: + "border-primary text-primary hover:bg-[color-mix(in_srgb,var(--color-primary)_14%,transparent)]", + destructive: + "border-destructive text-destructive hover:bg-[color-mix(in_srgb,var(--color-destructive)_14%,transparent)]", + ghost: "border-transparent text-muted-foreground hover:text-foreground hover:bg-accent", + }, + size: { + sm: "px-2 py-1 text-[11px]", + md: "px-2.5 py-1 text-[12px]", + lg: "px-3 py-1.5 text-[13px]", + }, + }, + defaultVariants: { variant: "outline", size: "md" }, + }, +); + +export type ButtonVariants = VariantProps<typeof buttonVariants>; + +export function Button({ + className, + variant, + size, + ...props +}: React.ComponentProps<"button"> & ButtonVariants) { + return <button className={cn(buttonVariants({ variant, size }), className)} {...props} />; +} diff --git a/umtool/docs/README.md b/umtool/docs/README.md @@ -0,0 +1,91 @@ +# umtool docs + +umtool judges and drives the things this repo makes videos out of. There are two +kinds of work in it today and the tool treats them the same way: as **projects** +in a folder tree, each with a state, a set of open decisions, and a build. + +These sheets live in the repo because they describe code and have to move with it. +Notes about *one particular video* belong beside that video, in its own `README.md`. + +## You have been asked for an umtool video + +Work out which kind you are making first — the rest follows from it. + +| What you were handed | Kind | Start here | +|---|---|---| +| A cited sweep report (`*sweep-report.md`) and "make this a video" | `report-video` | [report-video.md](report-video.md), then [authoring.md](authoring.md) | +| A song's clips and "judge these" | `song` (template `um-song`) | `~/reports/quartering-uh-song/specs/` | +| A report with no manifest yet | `sweep-report` | [authoring.md](authoring.md) | + +The short path from a cited report to a built video: + +```sh +# 0. scaffold it (writes an EMPTY timeline and a citation checklist) +umtool new <slug> --from <sweep-report.md> + +# 1. write the timeline (authoring.md — this is the work, and nothing automates it) + +# 2. is every source still fetchable, and is every citation wired up? +umtool check <slug> + +# 3. widen windows to whole sentences (dry first, then apply) +node scripts/report-to-video/resolve-windows.mjs ~/reports/<slug>/video.manifest.json +node scripts/report-to-video/resolve-windows.mjs ~/reports/<slug>/video.manifest.json --write + +# 4. bench any clip whose edges you are unsure of +# /browse/<slug>/clip/<id> + +# 4b. if the manifest has a ledger, rule on every claim in it. `umtool check` +# BLOCKS until it is empty, because both totals lie on an unadjudicated one. +# /browse/<slug>/claim/<id> + +# 5. build: a fast pass to watch, then the real one. +# The BUTTON on the project page runs it -- cancellation, the per-step +# timeouts and the process-group kill live in the server's job runner. +# `umtool build` prints the same chain if you would rather paste it. +umtool build <slug> --preset fast +``` + +**A manifest may describe more than one cut.** `build-video.mjs --variant +sourced|full` selects between them; `sourced` is the default and still writes +`out/<slug>.mp4`, so the button and `umtool build` are unchanged. The working +files moved under `out/<variant>/` while `clips-raw` and `availability.json` +stayed at the root. See +[report-video.md](report-video.md#one-manifest-two-cuts). + +**Run step 2 before step 5, always.** It is a few seconds and it catches the two +defects that have already shipped in real videos: a manifest with no `siteOrigin` +(19 QR codes encoding `undefined/?v=…`) and one pointing at `http://localhost:3000` +(QR codes that resolve to nothing on anybody's phone). + +## The sheets + +| Sheet | What it covers | +|---|---| +| [projects.md](projects.md) | Kinds vs templates, marker files, ids, how to add a kind | +| [folders.md](folders.md) | The walk, `REPORTS_ROOT`, read roots vs write roots | +| [browse.md](browse.md) | The project index, the four filters, cards per kind | +| [report-video.md](report-video.md) | The manifest as an EDL, the three-stage window model, `lock` | +| [clip-bench.md](clip-bench.md) | Editing a clip's window against the waveform and the cues | +| [claim-bench.md](claim-bench.md) | Ruling on a `ledger[]` claim against ±90 s of context | +| [build.md](build.md) | The four-step chain, presets, cancellation, overwrite | +| [decisions.md](decisions.md) | What earns a severity, and how a kind contributes | +| [mix-from-a-project.md](mix-from-a-project.md) | Deep-linking a clip into `/mix` | +| [index.md](index.md) | The LMDB index, and why the filesystem stays the model | +| [cli.md](cli.md) | `umtool ls / show / check / build / window / …` | +| [authoring.md](authoring.md) | Writing a manifest from a sweep report | +| [e2e.md](e2e.md) | The fixture, the stubs, the global queue | +| [quirks.md](quirks.md) | Everything that cost time to find out | + +## The rules that outrank convenience + +1. **The filesystem is the model.** An index may cache what the tree says; if a + value exists *only* in the index, that is a bug. See [index.md](index.md). +2. **The index must not shell out.** Listing projects never probes, never runs + ffprobe, never runs yt-dlp. Measuring is what a project page and a job do. +3. **Judgements travel with the tree.** `verdicts.json`, `notes.json` and + `video.manifest.json` live *inside* the project directory, so copying the + directory copies the decisions. +4. **Never guess at something you cannot read.** A directory whose name does not + route, a project two kinds match, a citation with no cue file — each is + *reported*, never silently dropped or resolved by picking one. diff --git a/umtool/docs/authoring.md b/umtool/docs/authoring.md @@ -0,0 +1,125 @@ +# Authoring a report video + +From a cited sweep report to a built cut. Written for an agent; a person can +follow it too. + +## 0. Scaffold + +```sh +umtool new <slug> --from ~/reports/<sweep>/sweep-report.md +``` + +Writes the directory, a manifest skeleton with an **empty** timeline, and a +README listing every `?v=` citation as a checklist. + +## 1. Fill in provenance — `siteOrigin` FIRST + +```jsonc +"provenance": { + "siteOrigin": "https://jeralyzer.pages.dev", // the archive's REAL origin + "channelSlug": "the-quartering", // default channel for cue lookups + "channel": "TheQuartering" +} +``` + +**This is the field that shipped broken twice.** One manifest has no `siteOrigin` +at all — 19 QR codes encoding `undefined/?v=…` — and one has +`http://localhost:3000`, a finished video whose codes resolve to nothing on +anybody's phone. `umtool check` blocks until it is set to something real. + +## 2. Turn each citation into a WINDOW + +**This is the work, and it is the part nothing automates.** + +A report records **one** second per citation. A clip needs a start *and* an end, +and both come from the source's cue file: + +``` +transcripts/channels/<channel>/data/<video>/transcript.cues.json +``` + +For each citation: + +1. Open the cue file and find the cues around the cited second. +2. **Verify the quote is actually there, and that the speaker is who you think.** + Two standing traps: a first-person quote is often the host reading someone + else's tweet aloud or being sarcastic — check ±90 s of context; and ASR + garbles names (the corpus stores "Metokur" as "mr medicare"). +3. Take `start` from the first cue of the thought and `end` from the last. +4. Round to **2 dp**. + +```jsonc +{ "type": "clip", "id": "c04", "video": "uyz1_FIqIEk", + "start": 32980.24, "end": 32994.19, + "cite": 32989, + "quote": "the words this clip exists for", + "note": "why it is in the cut" } +``` + +**Array order is the cut.** There is no sort. + +Per-clip `channel` when the sources span mirrors. On Rumble, `video` must be the +**local directory slug**, not the site/MCP id — see [quirks.md](quirks.md). + +## 3. Check, before anything encodes + +```sh +umtool check <slug> +``` + +Fix everything blocking. It catches a dead origin, a missing `channelSlug`, +duplicate ids, a cue file that is not there, and a source the last preflight found +gone. + +## 4. Widen to sentences + +```sh +node scripts/report-to-video/resolve-windows.mjs <manifest> # dry +node scripts/report-to-video/resolve-windows.mjs <manifest> --write # apply +``` + +A cue boundary is a *line-wrap* boundary, so cutting there drops the lead-in that +makes a quote make sense. + +Then read the result. Where the widener made a clip worse — it swallowed +neighbouring audio, or it undid a deliberately short quote — set `lock` and say +why in the project README. Across the six real manifests, five are **100% locked**: +a human-chosen window usually *is* the truth. + +## 5. Bench each clip + +`/browse/<slug>/clip/<id>`. Look at the waveform and the cue rail, and act on +what it says: run a clip to the end of its sentence, or set `lockEnd` to +acknowledge that you meant to cut there. One 14-clip cut shipped with 8 clips +ending mid-thought. + +If it warns that the widener would revert your edit, take the offer to set the +matching lock. + +## 6. Fast pass, watch it, then final + +```sh +umtool build <slug> --preset fast # hard cuts, minutes +umtool build <slug> --preset final +``` + +Watch the fast pass end to end. It is the only way to find a clip that is +technically correct and editorially wrong. + +## The checklist + +- [ ] `siteOrigin` is the archive's real origin +- [ ] every `video` is the **local** directory slug +- [ ] every quote verified in context, not just found by search +- [ ] every window at 2 dp, `end` after `start` +- [ ] `umtool check` exits 0 +- [ ] windows resolved, or locked with a reason written down +- [ ] no clip ends mid-sentence unless `lockEnd` says so +- [ ] a fast pass watched all the way through + +## What a sweep will miss + +Worth knowing before you trust a citation list to be complete: a share-link +keyword is the **spine** of a sweep, not its coverage. Back it with substring, +clinical-vocabulary and symptom sweeps — one real cut's earliest and best clip +never says the search word at all. diff --git a/umtool/docs/browse.md b/umtool/docs/browse.md @@ -0,0 +1,88 @@ +# /browse — the project index + +Every project, of every kind, in one list. **Zero client JavaScript**: the filters +are links that change `searchParams`, and `?q=` is a plain GET form. Nothing here +hydrates, and `pnpm build` still reports the page as server-rendered. + +## The four filters + +| filter | param | values | +|---|---|---| +| kind / template | `?kind=`, `?template=` | registry ids | +| state | `?state=` | `draft` `windows` `fetched` `built` `shipped` `stale` | +| open decisions | `?open=blocking` \| `?open=1` | from the decision counts | +| text / recency | `?q=`, `?sort=name\|recent` | substring over the haystack | + +**Counts come from the unfiltered set.** A chip whose number changes when you +click a different chip moves under the cursor, and the whole point of a filter row +is to say how much is behind each one. An e2e spec asserts the chip's number +equals the number of cards rendered. + +`?q=` is a `<form method="get">` carrying the other filters as hidden inputs — the +`/browse/find` idiom — so searching does not throw away the filters you set, and +any filtered view is one pasteable URL. + +## The state vocabulary + +One vocabulary across every kind, not per-kind words, because the point of the +filter is to ask "what is half-done" without first asking "half-done at what". A +kind maps its own situation onto these; it does not invent a seventh. + +## The cards + +`components/projects/ProjectGrid.tsx` **contains no kind ids and no per-kind +branches**, and that is the design working rather than an omission: a kind's +summariser has already rendered its own facts (`19 clips · 4m46s · 17 sources`, +`3/4 cuts · 2 variants`) and its own flags. The grid lays out strings. + +A kind can attach `data-*` attributes to its card through `attrs`, which is how a +song keeps `data-missing` as an assertion surface without the grid knowing what a +cut is. + +The poster falls back: the deliverable → a built **segment** (which already +carries the chrome, so the card looks like the video mid-build) → a raw clip → +nothing. Served by `/api/browse/poster?project=…`, which takes **no +client-supplied `rel`** — the frame is the one the summariser chose. + +## Routing + +`app/browse/[...path]` resolves the **longest** path prefix that is a project and +hands the rest to that kind's view. `a/b` being a project must not stop `a/b/c` +from being one. + +- `[]` → the project page +- `["wide"]` on a song → the cut page, unchanged +- `["clip","c04"]` on a report video → the clip bench +- anything else → 404, rather than a page that silently drops half its URL + +The static tool pages (`/browse/decisions`, `find`, `sources`, `faces`, `trim`, +`at`) still win their routes, and a spec asserts each is 200. + +## Cost + +Listing never probes and never shells out — the rule `listSongs()` already +followed, extended to every kind. A summary is memoised against a signature of +mtimes and sizes, so it survives for as long as the project has not changed and is +discarded the moment it has. + +The `open` filter needs decision counts for every project, which means running +each kind's reducer. That is memoised the same way; the expensive input is cue +files, and their *derived* answers are cached against the file's own mtime. + +## Discovered by getting it wrong once + +**Tailwind's source detection is turned OFF, and the three real directories are +named in `app/globals.css`.** It has to be. Auto-detection honours `.gitignore` +but scans everything else, so a scratch `NEXT_DIST_DIR` got read, its binary +turbopack cache yielded a garbage class candidate, and every page 500d on a CSS +parse error with nothing wrong in the CSS. + +Then it happened a second time from **this file**: an earlier draft quoted the +corrupt candidate to explain the first failure, Tailwind scanned the markdown, +and the trap re-created itself. Hence `source(none)` plus explicit `@source` — +a doc, a fixture, a test artefact or a stray dist dir can no longer poison the +stylesheet at all. If you add a directory that holds class names, name it there. + +**`lmdb` must be in `serverExternalPackages`.** Bundled, Turbopack tries to +resolve its optional `moduleRequire('cbor-x')` and fails the whole module graph — +so every page importing `lib/projects` 500s naming a package nothing here uses. diff --git a/umtool/docs/build.md b/umtool/docs/build.md @@ -0,0 +1,109 @@ +# Building + +From the project page, or `umtool build <project>` to see the chain. + +## Four steps + +| # | step | why it is separate | +|---|---|---| +| 1 | **check every source is still fetchable** | `yt-dlp --simulate`, no bytes. See below. | +| 2 | **resolve windows (DRY)** | Applying is a separate, explicit action. | +| 3 | **build** | `--progress ndjson --continue-on-error` | +| 4 | **verify the file that came out** | A build can exit 0 and be wrong. | + +**Availability is a STEP, not a preamble somebody remembers to run.** It is the +one fact about a manifest that goes stale in *both* directions — a source can die +after the manifest is written, and one annotated "gone" can come back. It costs +seconds. Without it a dead source is discovered twenty minutes and a dozen +paid-for fetches into the build. + +**Resolve runs dry.** A widener silently rewriting windows somebody just set in +the bench is exactly the surprise `lock` exists to prevent, so the chain never +passes `--write` and applying is a second action. + +**Verify exists because success is not self-evident.** A concat that produced a +zero-length file, a chapter pass that dropped markers, a timeline that lost a clip +because `--continue-on-error` let it — each looks like success at the terminal and +like a finished video in a directory listing. `verify-build.mjs` checks duration > +0, chapters == timeline entries, and a length floor. + +## Presets + +- **preview one clip** — `--only <id> --no-xfade`, for after moving an edge. +- **fast pass** — hard cuts over the whole timeline. Minutes, not tens of minutes. + What you watch to check the argument. +- **final** — crossfades and chapters. The deliverable. + +## Timeouts + +Per step, not one number. The default is 15 minutes and exists to catch the +accidental hour-long job; a 19-clip crossfaded build legitimately runs 20 to 40, +so the build step asks for `max(15 min, clips × 90 s)`. Raising the default to fit +the build would remove the guard from everything else. + +## Cancelling is safe, and resuming is free + +Cancel kills the **process group**. `build-video.mjs` shells out through +`execFile`, so the thing actually burning CPU or holding a download open is a +*grandchild* — `child.kill()` reaps the node process and leaves it running, which +is the same failure the diarize backfill had. + +Every artefact is content-addressed: a fetched window by its window, a segment by +its clip id. Re-running skips whatever finished. **A cancelled build is a paused +one.** + +## Overwriting + +`build-video.mjs` always passes `-y`. An output **newer than its manifest** is +refused (409, `needsReplace`); with `replace=1` it is moved aside as +`out/<slug>.<YYYYMMDD-HHMM>.mp4` — the stamp shape `promote` already uses for a +demoted cut. A deliverable that cost an hour of network fetches is never destroyed +to make a new one. + +## One job, process-wide + +Two builds writing one `out/segments/` would interleave, and two in different +projects would still fight over yt-dlp's rate limits and the CPU. A second start +is a 409 naming what is running. + +## Progress + +`--progress ndjson` emits one JSON object per line: `start`, `card`, `clip`, +`fetch`, `snap`, `segment`, `entry-failed`, `concat`, `chapters`, `note`, `done`, +`error`. The UI renders one box per timeline entry from them, because "step 3 of +4, running" is not progress when step 3 is the twenty-minute one. + +The event set is exactly what was already being printed. Making it a *format* +switch is what stops a wording change from breaking the driver. + +## `--continue-on-error` + +A dead source at clip 14 of 19 otherwise throws away thirteen fetches already paid +for. With it, everything buildable is built — and the run then **refuses to +concatenate** and exits non-zero. A finished file quietly missing a citation looks +complete, which is worse than no file. + +## Why it is spawned, not imported + +A 40-minute chain of yt-dlp and ffmpeg inside a request handler has no +cancellation story, its `execFile` buffers live in the server's heap, and a +runaway grandchild outlives the request that started it. `buildVideo()` is +exported anyway, and `widen()` *is* imported — the bench needs the CLI's own +function, or the two would disagree about where a clip ends. + +`umtool build` **prints** the chain rather than running it, for the same reason: +cancellation, the timeouts and the group kill live in the server's job runner, and +a second runner would be a second, worse set of those. + +## Everything here is testable offline + +The e2e fixture writes stub `YTDLP_BIN` and `QRENCODE_BIN`. The stub reports one +id removed the way a deleted upload is, which gives `source-unavailable` a true +answer. A full 4-clip build — cache reuse, three stub fetches, QR overlay, concat, +chapters, verify — runs in **3 seconds with no network**. + +## See also + +[quirks.md](quirks.md) for the VP9 trap, `--ignore-config`, exit 101, the Rumble +HLS retry and the relative silence threshold — every one of which will bite a +build and none of which is guessable. diff --git a/umtool/docs/claim-bench.md b/umtool/docs/claim-bench.md @@ -0,0 +1,119 @@ +# The claim bench + +`/browse/<project>/claim/<id>`. + +The clip bench asks *"where exactly does this cut?"*. This one asks *"what did he +mean by that?"* — and the two want opposite things. A clip wants sample-accurate +edges; a claim wants **room**. + +It exists because a `ledger[]` entry used to carry an undocumented interpretation +in its `company` field, and four different hazards were riding on it. + +## Why the quote is not enough + +The standing corpus rule: a first-person quote is routinely the host **reading +someone else's words**, or being sarcastic, and neither is visible inside the +quote. One claim in the employee-count corpus is a *guest's* payroll recorded as +the subject's, and it reads identically until you listen either side of it. + +So the page opens on ±90 s of cues with the cited one marked **inside** the +paragraph. That mark is the whole point: it shows the quote had someone else +talking round it. + +`CLAIM_CONTEXT_PAD = 90`, overridable with `?pad=` up to 600. + +## The six fields + +| field | vocabulary | what it settles | +|---|---|---| +| `scope` | `media` `coffee` `publica` `all` | which payroll the number is about | +| `scopeBasis` | free text, **required non-blank** | the phrase from the context that settles it | +| `scopeConfidence` | `clear` `read` `unresolved` | did the quote settle it, the context, or nothing | +| `population` | `employees` `full-time` `salaried` `contractor` `1099` `people` | the denominator | +| `valueKind` | `uttered` `derived` `synthetic` | is the number his, our sum, or our midpoint | +| `flags` | array, **always written** even when empty | anything a predicate cannot compute | + +Two of these are load-bearing in a way that is easy to miss: + +- **`scopeBasis` is refused when blank.** An adjudication with no basis is an + opinion, and a reviewer cannot check an opinion against the audio. +- **`flags: []` is written, not omitted.** An absent array reads as "nobody has + looked", which is exactly the state an adjudication is supposed to leave behind. + +**`scopeConfidence: "unresolved"` is a legitimate outcome**, not a failure to +finish. It feeds *neither* total. Forcing a reading on a genuinely ambiguous +sentence is the failure mode this whole surface exists to prevent. + +## The seventh field, which is optional + +`roles` records **who** he named, when he enumerated them rather than counting +them. + +```jsonc +"roles": [{ "role": "video editor", "count": 2, "verbatim": "two video editors" }] +``` + +The bench edits it as one line per role — `count | role | his words` — because +most claims have none and the handful that do are a two-line list; a repeater +widget would be more chrome than the field it edits. The preview under the box +shows what the rail will draw (`2 editors · 1 designer`), and a line that will not +parse disables the save rather than writing half a roster. + +**`verbatim` is required and is the point of the field.** The count and the role +name are our reading; without his own words beside them nobody can check the +reading against the audio — the same rule `scopeBasis` exists for. + +**It is deliberately NOT one of the six.** Most claims are a number and nothing +else, and gating the inbox on a field only a handful of entries can ever carry +would leave it permanently red. `adjudicationGaps()` ignores it; `rolesGaps()` +checks it when it is there, and a bad roster is a **400**. + +Why it is worth collecting at all: in the employee-count corpus the roster is the +control. Five times he names who works for him and five times it is two video +editors and a graphics designer; the totals he attaches are three, then four, then +ten. The rail draws the roster under the tally precisely so it can be seen +standing still while the numbers above it move. + +## The vocabularies are imported, never restated + +`SCOPES` / `SCOPE_CONFIDENCE` / `POPULATIONS` / `VALUE_KINDS` come from +`report-to-video/ledger-totals`, into both the page and `updateClaim()`. A page +offering a seventh population the arithmetic has never heard of is the silent +divergence the shared module exists to stop. A value outside the vocabulary is a +**400**, never a coercion. + +## Writing + +Same contract as the clip bench: `updateClaim()` re-reads inside a process-wide +lock, writes tmp+rename, preserves the CLI's `JSON.stringify(m, null, 2)` +formatting, and **guards on the manifest's own mtime**. A stale token is a `409`, +never a silent overwrite — somebody may have run `resolve-windows --write` in +between, and losing that is losing human judgement. + +It is deliberately *not* `updateClip()` with more fields. A clip edit moves a +window; a claim edit records a ruling on what a sentence meant. Sharing a function +would mean one of them could quietly write the other's fields. + +## Audio, when there is any + +A claim is a **moment**, not a window, so its playable candidates are the cached +`out/clips-raw` files that *contain* its `cite`. Most claims have one already, +because the build over-fetches around every clip. + +When none does, the page offers a fetch that runs the pipeline's own +`--fetch-only`, which now accepts a **ledger id** as well as a timeline id: it +synthesises a hair-wide entry around `cite` and lets `--pad` do the rest. The file +lands in `clips-raw` under the build's own naming and a later build reuses it. + +That path needs `channel` / `video` / `cite` on the ledger entry. They were +recoverable for the employee-count corpus from the markdown report's own links — +but 20 of 50 needed the **Rumble site-id → URL-slug** remap first (see +[report-video.md](report-video.md)). + +## Sign-off is "the inbox is empty" + +`claim-unadjudicated` is **blocking** and is emitted one row per entry. Work it to +empty; see [decisions.md](decisions.md). + +Never auto-apply an adjudication. Every ruling carries its `scopeBasis` precisely +so the next person can check it against the audio in one click. diff --git a/umtool/docs/cli.md b/umtool/docs/cli.md @@ -0,0 +1,89 @@ +# umtool, from a terminal + +``` +pnpm --filter umtool exec umtool <command> +node umtool/bin/umtool.mjs <command> +``` + +The audience is an AI assistant working in this repo, which is why every command +takes `--json` and why `check` exits non-zero. + +It reads the **same** `lib/projects/*.mjs` the app does, so `umtool ls` and +`/browse` cannot disagree about what a project is, and `umtool check` and the +decisions inbox cannot disagree about what is wrong with one. + +## Commands + +| | | +|---|---| +| `ls [--kind --template --state --open --blocking --q --sort --json]` | the index, as text or JSON | +| `show <project> [--json]` | one project: summary, the cut, per-clip status, decisions | +| `check [<project>] [--json]` | **exit 1 on anything blocking** | +| `decisions [--json]` | the inbox | +| `folders [--json]` · `kinds [--json]` | the tree, the registry | +| `window <project> <clip> [--start S] [--end E] [--lock] [--lock-end] [--no-lock-end] [--note …]` | edit a window | +| `build <project> [--preset preview\|fast\|final] [--only ID]` | **prints** the chain | +| `index [--rebuild] [--prune] [--since MS] [--json]` | the cache | +| `new <slug> [--kind report-video] [--from <sweep-report.md>]` | scaffold | + +A project argument is an exact id, a directory, or a **unique** basename. Two +projects answering to one name is reported, never resolved by picking one. + +## Environment + +`REPORTS_DIR`, `SONG_REPORTS_DIR`, `SONG_DIR`, `CHANNELS_DIR`, `UMTOOL_INDEX_DIR` +— which is how it is tested against the e2e fixture. + +## `check` is the one to run before every build + +``` +$ umtool check +BLOCKING ferret-rescue manifest-invalid provenance.siteOrigin + `http://localhost:3000` — every QR in this cut resolves to nothing on anyone else's phone +BLOCKING quartering-employee-count manifest-invalid provenance.siteOrigin + missing — every QR in this cut encodes `undefined/?v=…` +OPEN quartering-walmart-shelves stale-build out/quartering-walmart-shelves.mp4 +12 project(s), 2 blocking, 1 open +$ echo $? +1 +``` + +Those are the two defects that shipped in finished videos. It is a few seconds in +front of a twenty-minute build, and it exits non-zero so a script can gate on it. + +## Two deliberate limits + +**`build` prints, it does not run.** Cancellation, per-step timeouts and the +process-group kill live in the server's job runner; a second runner here would be +a second, worse set of those. Use the button on the project page, or paste the +printed commands. + +**`check` cannot compute the song reducer.** Nine decision kinds are TypeScript +beside `readSong()`, the loudness cache and the accepted cover set. It reports how +many projects it only checked the routing of, and points at `/browse/decisions`. + +## `window` goes through the same writer the bench does + +2 dp, the CLI's own formatting, tmp+rename, one `.bak`. A second implementation is +how the two would start disagreeing about where a clip ends. + +``` +$ umtool window ferret-rescue c01 --start 43.12 --end 61.48 --lock-end +c01: 43.12–61.48 -> 43.12–61.48 + lockEnd +``` + +`--no-lock-end` **removes** the key rather than writing `false`: these manifests +are read by humans, and `"lockEnd": false` reads like a decision. + +## `new` writes an EMPTY timeline, on purpose + +It would be easy to derive first-draft clips from a report's citations — the shape +is regular. It is not done, and that is the honest position rather than a missing +feature: a report records **one** second per citation, a window needs a start and +an end taken from `transcript.cues.json`, and matching a quote to its cues is the +actual work. A generated timeline of guessed windows would look finished and be +wrong, and every clip would have to be opened anyway. + +So it writes what can be known, lists the citations it found as a **checklist**, +and leaves `siteOrigin` empty so `check` blocks until somebody sets it. diff --git a/umtool/docs/clip-bench.md b/umtool/docs/clip-bench.md @@ -0,0 +1,93 @@ +# The clip bench + +`/browse/<project>/clip/<id>`. + +"How much context does this clip need" used to be a loop of hand-editing JSON, +re-running two CLIs and watching an mp4. This is that loop in one place. + +## Absolute source seconds, everywhere + +The manifest's numbers, the cue file's, the QR's. The cached file's own start +(`fetchStart`) is the **only** relative number in the component, and it exists +solely to set `video.currentTime`. The moment those two are allowed to mix is the +moment a window is off by the pad and nobody can see why. + +## The preview is the cached file, served whole + +`out/clips-raw/<video>_<a>-<b>.mp4`, with byte ranges, and all windowing happens +in the browser. No ffmpeg per drag. + +**A 206 is not optional** — without one the `<video>` element will not seek in a +stream it did not fully download, and that is the entire interaction. The file is +immutable (its window is in its name), so it is cached hard and `analyseMedia`'s +envelope can never miss twice for the same window. + +`file` must be a member of the server's own scan of that clip's cached windows. +Never a path from the client. + +## A drag never downloads + +Dragging past the cached window **clamps** and offers a button. A handle that +silently starts a 12-second network fetch is a handle you stop trusting. + +The fetch runs the pipeline's own `--fetch-only` path, so the file lands named the +way the build expects, with the same format pin and the same Rumble HLS retry — +and containing-window reuse then makes that generous fetch **be** the build's +cache rather than a second one. + +Past the source's own duration the handle stops for good. + +## Three things the JSON cannot show you + +**Ends mid-sentence**, quoting the cue the cut lands inside — because "ends +mid-sentence" alone does not tell you what you are cutting off. One 14-clip cut +shipped with 8 clips ending mid-thought. `lockEnd` is the acknowledgement and +silences it. + +**This source has no punctuation**, when the ASR emitted no terminators at all. +Then *every* clip "ends mid-sentence" and the fact says nothing about the cut, so +the answer is `null` — cannot be known — rather than a confident `false`. Set +those edges by ear and lock them. + +**What the widener would do**, computed in-process because `widen()` is pure once +the cues are read. Moving an edge somewhere `resolve-windows` would not produce +means the next `--write` reverts it, so the bench offers to set the matching lock. +That is why five of six real manifests are 100% locked. "Run the widener and see" +stops being a leap of faith. + +## The cue rail + +Every cue in view, positioned by time. Inside the selection in full contrast, +outside dimmed — so "what am I cutting off" is *read*, not inferred from a +waveform. A `¶` marks a cue that closes a sentence, using the same regex +`resolve-windows.mjs` uses (imported, not re-written, so the rail and the widener +cannot disagree). Clicking a cue snaps the nearer edge to it. + +## Keyboard + +`[` `]` move the start · `,` `.` move the end · shift for 0.5 s instead of 0.05 · +`space` auditions the selection · `R` resets to the saved window. + +Audition happens on pointer-**up**, never during a drag — the Deck's rule, because +a sound restarting on every `pointermove` is unusable. + +## Saving + +`PUT /api/report/window` with `{project, clip, start, end, lock…, token}`. It +never sends a path and it cannot ask for an entry to move. + +See [report-video.md](report-video.md) for the four rules the writer keeps (2 dp, +the CLI's formatting, tmp+rename under a lock, an mtime token). A stale token is a +**409 with both values**, never a silent overwrite: the other writer is usually +somebody's judgement. + +## Discovered by getting it wrong once + +**The page and the inbox must answer the same question the same way.** `umtool +show` pilled four ferret-rescue clips "ends mid-sentence" while the decisions +inbox stayed silent about them, because the inbox had a punctuation gate and the +detail did not. The inbox was right. + +**The bench's prediction is testable, and is tested.** It says 3.00–6.00 widens to +3.00–9.00; saving 3.00–9.00 makes `resolve-windows` a no-op on that clip. That +round trip is the whole argument for the bench. diff --git a/umtool/docs/decisions.md b/umtool/docs/decisions.md @@ -0,0 +1,106 @@ +# Decisions + +One worklist, every project, every kind: `/browse/decisions`, or +`umtool decisions`. + +## Severity is EARNED + +| | | +|---|---| +| `blocking` | something downstream would **lie or die** if you acted on it | +| `open` | a real decision nobody has made | +| `info` | a true fact that is not a decision | + +An inbox that marks four missing cuts per song as blocking is an inbox nobody +opens twice. A song that never had a vertical is not waiting on you. + +## It is a REDUCER, and it must never measure + +Every call it makes is one a project page already makes. An inbox that shells out +to ffmpeg once per rendition, or to yt-dlp once per source, is an inbox that takes +a minute to open — which is the one thing it cannot afford to be. + +So loudness is read from the **cache**, and availability is read from whatever the +preflight last **wrote** (`out/availability.json`), never measured here. "Nobody +has ever run one" is itself something it can say. + +## The kinds + +Each registry entry owns its own vocabulary (`decisionKinds`), and the union is +assembled rather than hand-written — a closed union in one shared file would mean +every future kind editing it to say a word only it uses. + +**report-video** + +| kind | severity | trigger | +|---|---|---| +| `manifest-invalid` | blocking | missing or `localhost` `siteOrigin`, missing `channelSlug`, duplicate ids, `section` out of range | +| `clip-no-cues` | blocking | no `transcript.cues.json` — the build dies there | +| `clip-unfetchable` | blocking | the last preflight says the source is gone | +| `clip-mid-sentence` | open | the cut lands inside a cue that does not close a sentence, and `lockEnd` is unset | +| `window-overlap` | open/info | two clips from one source overlap | +| `no-punctuation` | info | a source's ASR has no terminators — one row per project | +| `claim-unadjudicated` | blocking | a `ledger[]` entry missing any of the six adjudication fields | +| `claim-incoherent` | open | a fired coherence predicate, in plain words | +| `stale-build` | open | the output is older than the manifest | +| `unbuilt` | info | never built. A normal state, not a decision | + +`claim-unadjudicated` **earns** blocking: both the stated and the implied total +lie if you act on an unadjudicated ledger, and they lie quietly, in a chart, with +somebody's name on it. It is also the one kind that is deliberately **one row per +entry** rather than collapsed — fifty unpunctuated sources are the same true +thing said fifty times, but fifty unadjudicated claims are fifty *different* +decisions, and the inbox being empty is the sign-off. + +`claim-incoherent` is `open` rather than blocking because the decision is +**editorial**: is the flag right, and does it belong on screen? Neither answer +stops a build. + +Both come from `ledger-totals.mjs` in `report-to-video`, which is pure arithmetic +over JSON already parsed into memory — so the reducer stays a reducer. Audio is +fetched only when a claim page is opened, one claim at a time. + +**song** — the nine the existing reducer emits: `unjudged-variant`, +`missing-cut`, `no-recipe`, `stale-recipe`, `spec-problem`, `no-plan`, +`unattributed`, `thumb-unaccepted`, `loudness`. + +**routing**, kind-independent — `shadowed-name` (blocking), `unroutable-name` +(info), `ambiguous-project` (blocking). + +## Adding one + +Extend the kind's `decisionKinds` and emit it from its `decisions(ctx, summary)`. +Nothing outside `lib/projects/` needs to change. + +## What the CLI can and cannot do + +`umtool check` computes every report-video decision and every routing one, and +exits 1 on anything blocking — which is what makes it usable as a gate before a +build. It **cannot** compute the song reducer: that is TypeScript beside +`readSong()`, the loudness cache and the accepted cover set, and a second +implementation is the thing this design exists to avoid having two of. It says +how many projects it only checked the routing of. + +## Discovered by getting it wrong once + +**Forty true rows are worse than one.** The first real run emitted a +`no-punctuation` row per *source* — forty-odd across six projects, all correct, +burying two blocking rows off the top of the list. One per project now. + +**A reducer that says the opposite of the card next to it.** The first +adjudicated build shipped a rail tally reading "Spans everything **18**" directly +beside a card reading "He never said eighteen". The adjudication had set +`valueKind: "derived"` and nothing downstream was reading it: the tally still took +the latest value of any kind, and the ledger's `label` still said "the only +explicit sum in the corpus". A field that changes what something *means* has to be +chased through every renderer that prints it, and the only way this surfaced was +looking at a frame. + +**Do not assume where a project reads its cues from.** The first run confidently +reported three sources of `quartering-flagging-takedowns` as having no cue file. +They cite deleted YouTube uploads, are cut from live Rumble mirrors, and build +against a **shadow `CHANNELS_DIR`** the project's own `make-shadow-channels.sh` +writes. A project now says which directory it reads — `provenance.channelsDir`, or +the `.shadow-channels` convention that already existed — and neither is a guess: +both are things the project wrote down. When the builder is present but unrun, the +decision says to run it rather than declaring the sources gone. diff --git a/umtool/docs/e2e.md b/umtool/docs/e2e.md @@ -0,0 +1,81 @@ +# The e2e suite + +```sh +pnpm --filter umtool e2e # everything +pnpm --filter umtool e2e projects.spec.ts # one file +``` + +A **"waiting for the e2e queue"** banner is normal, not a hang: the queue is +machine-global and one suite runs at a time. + +## The fixture + +`e2e/fixtures/make-fixture.mjs`, rebuilt on every run. The suite **never** runs +against the real song directory or the real reports tree — those hold thousands of +real human verdicts and six finished videos, and a spec that judged a clip or +rendered over a deliverable would be indistinguishable from a person doing it. + +**Every fixture item has a true answer.** That is the principle; the specifics: + +| | | +|---|---| +| `bg.mp4` | 220 Hz then 3000 Hz at the same loudness — only the brightness curve can see the change. A cue at 3.00 s. | +| `song.mp4` | 2 s of silence then a tone. `firstSound` at 2.00 s. | +| `vid1` cues | punctuated, with a run-on cue at 3–6 s | +| `vid2` cues | **no terminator anywhere** — the real degradation in this corpus | +| `gone1` cues | readable, but the stub yt-dlp reports the upload removed | +| `vid1_0.00-9.00.mp4` | tone/silence/tone with silences centred on 3.0 s and 6.0 s (verified in the file: 2.90–3.11, 5.92–6.11) | + +## The projects, and why each exists + +| | | +|---|---| +| `report-fixture` | read-only. 4 clips: c01 ends mid-sentence, c04 does too but sets `lockEnd`, c03's source has no punctuation | +| `bench-fixture` | the clip bench **writes** | +| `build-fixture` | the build **writes** | +| `gone-fixture` | its source is gone — the preflight must block it | +| `no-origin-fixture` / `localhost-fixture` | the two defects that shipped | +| `bike-fixture` | the third kind | +| `find/` | shadowed by a tool page | +| `deep/nested/solo-fixture` | a pass-through chain, for the collapse | + +**Three copies of one manifest is not duplication.** Sharing one project between +the read-only specs and the writing ones made the suite pass or fail depending on +which file playwright ran first — and the failure named the wrong thing entirely. + +## Stubs + +`YTDLP_BIN` and `QRENCODE_BIN` point at node scripts the fixture writes, so the +whole build chain runs **offline and deterministically**. They are node, not bash: +the yt-dlp stub does fractional arithmetic on `--download-sections *FROM-TO`, and +doing that in bash means awk, which means three layers of quoting inside a +generated file. It got mangled once. + +## Writing a spec here + +- Assert **relationships, not magic numbers**. "the chip's number equals the + number of cards" survives a new fixture project; `toHaveCount(6)` does not. +- If your spec writes, give it its own project. +- A project id is a **path**: `[data-project$='/alpha']`, not `[data-song=alpha]`. +- Some assertions read the **source**, not a page — that a kind id is not + special-cased outside the registry, that `RESERVED_BROWSE` matches the real + directory listing. Those are the ones that fail when the design is broken in a + way no rendering can show. + +## Known flake + +`undo.spec.ts:93` reds under load and passes in isolation. It is the documented +waveform-drag capture race, not a regression. + +## Gotchas + +**Run e2e in dev mode** (the default). `E2E_MODE=start` serves the last build, +which is stale for uncommitted source changes. + +**`pnpm build` is a separate check.** The suite runs in dev, which never +prerenders — a layout or client-component change can pass every spec and 500 in +production. Two real bugs in this feature were found that way and one only by a +page render. + +**Kill stray dev servers by port**, not with `pkill -f`: the pattern matches your +own shell's command line. diff --git a/umtool/docs/folders.md b/umtool/docs/folders.md @@ -0,0 +1,90 @@ +# Folders, and the roots + +## `REPORTS_ROOT` + +The tree every project hangs off. `REPORTS_DIR ?? ~/reports`, and in e2e it +defaults to `dirname(SONG_REPORTS_DIR)` so a fixture stays confined without a new +environment variable. + +`SONG_REPORTS` (the um-song deliverables) keeps its exact previous default and is +now a *subdirectory* of `REPORTS_ROOT` rather than the widest root there is. + +## The walk + +Two rules do almost all the work. + +**A PROJECT IS A LEAF.** Detection stops the descent. That is what keeps `out/` +— 1,210 files and 3.1 GB across `~/reports` — out of the walk entirely. Nothing +in the index ever sees a clip, a segment, a card PNG or a variant. + +**A FOLDER WITH NO PROJECT BENEATH IT DOES NOT EXIST.** That silently drops the +~40 loose test directories under `quartering-uh-song` — `alarm-tests`, +`chop-tests`, `run-visual-tests`, `sfx`, `pipeline` — with no denylist to +maintain and nothing to update when the 41st appears. + +Also: dotfiles, `node_modules`, `out`, `variants`, `plan`, `thumbs` and `data` +are never descended into; depth is capped at 4; symlinked directories *are* +followed, but every real path is visited once so a link to an ancestor terminates +instead of spinning. + +Measured on the real tree: **12 projects in 25 ms**, and the 3.1 GB never touched. + +## Collapsing + +A folder with no projects and exactly one child collapses **for display**: +`quartering-uh-song / videos` is one heading. + +**The URL is never collapsed.** `/browse/quartering-uh-song/videos/yoshi` stays +the one true address. A URL has to mean the same thing in six weeks, and a +display convenience does not get to decide what a link is. + +## Read roots vs write roots + +`resolveInRoots()` guarded what may be **opened** and what may be **rendered to**. +They were one list — so widening the read root to reach report videos would in +the same stroke have made every report's `out/` a legal render target. A 46 MB +deliverable that cost an hour of network fetches, one typo in `/api/mix/render` +away from being overwritten. + +``` +READ_ROOTS SONG_REPORTS, REPORTS_ROOT, SONG_DATA, SONG_SCRATCH +WRITE_ROOTS SONG_REPORTS, SONG_SCRATCH +``` + +Reports became readable and mixable. **Nothing new became writable.** A mix of a +report clip still lands in `SONG_REPORTS`, and a hand-typed path outside the +write set is refused exactly as before. + +`MIX_ROOTS` overrides the read set; `MIX_WRITE_ROOTS` overrides the write set. + +**`SONG_REPORTS` stays first in the read list.** It is a subdirectory of +`REPORTS_ROOT`, so whichever comes first decides every relative label — and +putting `REPORTS_ROOT` first would silently rewrite every existing +`videos/<song>/wide.mp4` into `quartering-uh-song/videos/<song>/wide.mp4`. +`labelFor` and `resolveInRoots` read the same ordered list, which is what keeps a +label a round trip. + +## Media listing + +`listMedia()` walks the **project** roots two levels deep — a report's deliverable +is at `<project>/out/<slug>.mp4` and a song's cut at `videos/<song>/<cut>.mp4`, +and neither was visible before. `SONG_DATA` and `SONG_SCRATCH` stay at one level: +a second level there is thousands of stats of clip fragments to find nothing +anybody would load. + +`segments`, `clips-raw`, `cards` and `qr` are excluded by name, or reaching one +level deeper would put ~260 intermediates into a picker that is already saturated. + +## Discovered by getting it wrong once + +**Relative paths bind to the first root, without stating.** `resolveInRoots` does +not touch the disk, so a relative path resolves against the first root it *could* +live under whether or not it is there. Survivable only because every path that +crosses the wire from a picker or a project link is **absolute** — a relative one +is a display label being handed back, and those came from `labelFor` against the +same ordered list. Keep it that way. + +**The picker's cap was saturated, and depth alone did not fix it.** Measured: +four of the six report deliverables still fell off the end of a 600-entry +newest-first list. Coverage had to become a property of the *enumeration* — each +project is asked for its own files, with its own small cap — not of the limit. diff --git a/umtool/docs/index.md b/umtool/docs/index.md @@ -0,0 +1,75 @@ +# The index + +`CACHE_DIR/index/projects.mdb`, LMDB, entirely optional. + +## Honest sizing + +At twelve projects this saves **50 to 150 ms** per load. It is not a speed fix +today and is not presented as one. What it buys: + +- **`--since`** — an agent asking what changed is a range read, not a diff of two + full scans. +- **recency as a range read**, for when the tree is 500 projects. The key is + `[MAX - mtimeMs, id]`, so ascending *is* newest-first. +- **decision counts without running the reducer** — the cost that grows fastest, + being the only part of a project read that touches megabytes of cue files. + +## The rule + +> **If a value exists only in the index, that is a bug.** + +It is in the code, because it is what keeps `lib/browse.ts`'s "No database. The +filesystem is the model." true. + +**FS first, index after, best effort**, in a swallowed try/catch. A crash between +the two leaves a signature that no longer matches, which the next read repairs. A +stale index self-heals and the user sees nothing but latency. + +Index-**first** could claim something the filesystem does not say. That is the one +failure this refuses. + +## Freshness signs INPUTS + +`sha1(schema + kind signature + mtimes + sizes)`. Never the produced record, which +would be circular; never bytes, which the export build already learned about. + +The schema folds into every signature so a bump invalidates everything — +deliberately **not** a generation counter, which would invalidate every project +whenever any one of them changed. + +Every read verifies. A record whose signature no longer matches is discarded, not +migrated. + +## Degradation + +A missing or unopenable store returns a **no-op** whose `get()` is `null` and +whose `put()` does nothing — copied in posture from +`common/lib/channelSignature.ts`. A fresh checkout, a deleted cache and a machine +without the native module all take that path, and everything still works. + +## Observing it + +It is deliberately almost invisible, so its health has to be surfaced on purpose: + +- the footer note on `/browse` — `index: 9/11 fresh` +- the `x-index` header on `/api/browse/projects` +- `umtool index` — records, schema, when it was built +- `umtool index --rebuild` / `--prune` / `--since <ms>` + +Three specs hold it to its contract: the header reports what it served, deleting +the `.mdb` produces byte-identical page data, and a manifest edited behind its +back is re-read rather than served stale. + +## Discovered by getting it wrong once + +**`lmdb` has to be a direct dependency of umtool.** The plan said to import it +through `common`, which already depends on it — but under pnpm's strict resolution +it does not resolve from umtool at all, and the CLI (plain node, no bundler) +cannot import a bridge written in TypeScript. pnpm dedupes it to the same store +entry anyway. + +**`lmdb` has to be in `serverExternalPackages`.** Bundled, Turbopack tries to +resolve its `moduleRequire('cbor-x')` — an *optional* dependency it only reaches +for an encoding nothing here uses — and fails the whole module graph. Every page +importing `lib/projects` then 500s naming a package that is not involved. Caught by +e2e, not by `tsc` and not by a build run before the wiring. diff --git a/umtool/docs/mix-from-a-project.md b/umtool/docs/mix-from-a-project.md @@ -0,0 +1,84 @@ +# Reaching /mix from a project + +Six of seven projects used to resolve to `null` in the mix bench: `MEDIA_ROOTS` +was the um-song subtree, so no report video's media was openable at all. + +## The link + +``` +/mix?body=<absolute>&start=<s>&end=<s>&from=<projectId>&clip=<clipId> +``` + +- **`body` is absolute**, like the picker's own `<option value>`. That is what + dodges the relative-binding hazard: a relative path is tried against each root + in order and never stats, so it binds to the first root it *could* live under. +- **`from` and `clip` are provenance** — the back-crumb and the header line. They + are deliberately not used to resolve media; that would be a second, divergent + resolver for what the first one already did. +- A deliverable link carries no window: the whole thing is the point. + +Resolved **server-side** in `app/mix/page.tsx`, which is why this needs no +`useSearchParams` and no Suspense boundary — that page already ran on the server, +it simply never read its own `searchParams`. + +## What a clip opens, and why + +The **widest cached raw window**, not the built segment. + +1. It exists as soon as the clip has been fetched once; segments only exist after + a build. +2. It has **no chrome burned in** — which is what a mix is looking at. +3. It is the file the bench already has peaks for. + +The built segment is the fallback, and the link says which it is. A clip with +neither renders **no link at all** rather than a dead one that 400s. + +## The prefill + +`start = clip.start - fetchStart`, `end = clip.end - fetchStart`, computed +server-side because the arithmetic is exact and known there; making the client do +it would be a second place to get it wrong. + +The bench then says *"this window came from c01 — the file itself runs +0.00–9.00s"*, so the material outside the window does not look unreachable. +Dragging past the end is harmless: `resolveMix` clamps `end` to the body's +duration, and `normaliseSpec` clamps `start` to ≥ 0. + +## Precedence + +**preset > the saved pair > blank.** A link that named a file and a window must +not lose to whatever the bench was last pointed at. The per-pair knobs — handover, +fade, gain, duck — *are* adopted, because those are things learned about that +pair. + +Arriving via a link does **not** write a session. A session is saved only on a +real render, so drive-by navigation cannot change what the next bare `/mix` visit +opens. + +## Refuse, never clamp + +A `body` outside the roots, or one with no audio track, produces a bench with no +preset and a **visible reason**. Silently opening a different file than the link +named is the one outcome worse than an error, and `lib/mix.ts` already takes this +line for the same reason. + +## The picker + +Grouped by project via `<optgroup>` — zero client JS, still a native select — with +a one-line text filter above it. + +Coverage is a property of the **enumeration**, not of the cap: each project is +asked for its own files with its own small cap, so one busy project cannot push +every other off the end. Measured before this: four of six report deliverables +fell off a 600-entry newest-first list, and per-song cuts never appeared at all. + +A report project whose only media is `out/clips-raw` correctly offers **nothing** — +those are intermediates, excluded by name. + +## Where a mix lands + +`SONG_REPORTS`, still. Reports became **readable**, not writable — see +[folders.md](folders.md). A mix of a report clip therefore lands in the um-song +deliverables directory, which is not ideal, and the trade was deliberate: the +alternative made every report's `out/` a legal render target, one typo away from +overwriting a 46 MB deliverable that cost an hour of fetches. diff --git a/umtool/docs/projects.md b/umtool/docs/projects.md @@ -0,0 +1,115 @@ +# Projects + +A **project** is a directory under `REPORTS_ROOT` that holds one piece of work. +umtool finds them by walking the tree; nothing registers itself and nothing had +to move on disk for this to exist. + +## Kind vs template + +A **kind** is what a thing *is*. A **template** is which configuration of that +kind it is. `um-song` is not a kind — it is the one template of the `song` kind +that exists so far, and the distinction is the whole point: the next thing this +tool has to hold (a supercut, a cover set, a vertical short) will be a new +template of an existing kind at least as often as a new kind. + +Three kinds ship: + +| kind | template | marker | what it is | +|---|---|---|---| +| `report-video` | `cited-timeline` | `video.manifest.json` | a sweep report said in the sources' own voices | +| `song` | `um-song` | `spec.json`, `verdicts.json`, or any cut | filler sounds playing a game tune | +| `sweep-report` | `sweep` | a `*sweep*.md` **and no manifest** | a report that is not yet a video | + +`sweep-report` earns its place on day one because `~/reports/hasan-bike` is one, +and because a kind with no decisions, no build and no rich read is the cheapest +possible proof that the registry is extensible. + +## Detection + +`project.json` (`{kind, template}`) wins outright, so a directory can always +declare itself. Otherwise every kind's `detect()` runs against the directory's +entry NAMES — no reads, no stats. + +**Two matches is an error, never a guess.** A directory that is two kinds is a +bug, and picking one would hide it; it gets an `ambiguous-project` decision +instead. + +## Identity: an id is a PATH + +A project's id is its POSIX path relative to `REPORTS_ROOT` — +`ferret-rescue`, or `quartering-uh-song/videos/yoshi`. Not the basename: bare +names collide across folders (a second `pokemon` is a matter of time) and the +path is what makes a link stable. + +A **bare basename still works as a URL**, because `/browse/yoshi` and +`/browse/yoshi/wide` are the addresses that exist in every decision href, every +spec, and whatever anybody has open. It resolves **only when unique** — two +projects sharing a name is precisely why an id is a path, so that case reports +the collision and offers both canonical URLs rather than picking one. + +## Two ways a project cannot be opened + +Both used to fail **silently**, which is worse than either. + +**`shadowed`** — the first path segment is a static page under `app/browse/` +(`decisions`, `faces`, `find`, `sources`, `trim`, `at`). A static segment beats a +dynamic one, so `/browse/find` renders the phrase console no matter what is on +disk. The project is still listed, with a `blocking` decision, and its link goes +to `/browse/at?path=…`. + +**`unroutable`** — the name fails `isSegment()` (a space is enough). It has no +address of its own; it is listed with an `info` decision and reached the same way. + +`RESERVED_BROWSE` is asserted by an e2e spec to equal the real directory listing +of `app/browse/`, so a tenth tool page cannot quietly make a project unreachable. + +## Adding a kind + +**A registry entry and one view. Nothing else.** + +1. Add an entry to `lib/projects/kinds.mjs`: `id`, `template`, `label`, `badge`, + `detect(names)`, `stages`, `decisionKinds`, and optionally `summarise`, + `signature`, `decisions`, `views`. +2. Add a component to `components/projects/` and a case in `ProjectView.tsx`. + +`e2e/projects.spec.ts` enforces this three ways, and they are the assertions that +fail when somebody special-cases a kind in a page — which no page test can see: + +- **no kind id is special-cased outside `lib/projects/` and + `components/projects/`** — a grep for lines carrying both a kind id and the + word `kind`; +- **`RESERVED_BROWSE` equals the real static pages**; +- **a kind injected through `UMTOOL_EXTRA_KINDS` reaches the index, the chips and + the CLI with no code edit at all**. + +## Where the code is + +| file | what | +|---|---| +| `lib/projects/kinds.mjs` | the registry. Plain ESM, no TypeScript | +| `lib/projects/walk.mjs` | the folder walk, routing, collapsing | +| `lib/projects/{report,song,sweep}.mjs` | per-kind reading | +| `lib/projects/core.mjs` | the CLI's assembled view | +| `lib/projects.ts` | the app's façade: caching, dispatch, the index | +| `lib/project-types.ts` | types only, **no `node:` import ever** | + +## Discovered by getting it wrong once + +**`lib/projects/kinds.mjs` is server-only**, despite being plain ESM. It imports +the per-kind modules and those read the disk, so importing it from a client +component drags `node:fs` into the browser bundle — which this repo has already +been bitten by: it passed `tsc --noEmit` and then 500'd every page. A client +component takes `lib/project-types.ts` and gets the rest as props. **`pnpm build`, +not typecheck, is what catches a regression here.** + +**Watch the import cycle.** `songIds()` lives in its own module +(`lib/projects/song-ids.mjs`) because putting it in `song.mjs` closes +`song → walk → kinds → song`. Plain node survives that; Turbopack evaluates +`kinds.mjs` while `song.mjs` is still initialising and every song page 500s with +"Cannot access 'CUT_NAMES' before initialization". `pnpm build` does not see it +either — nothing prerenders. A page render does. + +**Do not write a root path out by hand.** The song-decision dispatch had the +production path hard-coded, so under the e2e fixture — whose songs live elsewhere +— *no song decisions were produced at all* and the inbox looked clean because it +was empty. diff --git a/umtool/docs/quirks.md b/umtool/docs/quirks.md @@ -0,0 +1,227 @@ +# Quirks + +Things that cost time to find out. Each one is here because it was discovered by +getting it wrong, and none of them is guessable from the code. + +## Fetching source clips + +**yt-dlp picks VP9 + Opus at these heights unless you pin the format.** Since +`--force-keyframes-at-cuts` re-encodes, that means `libvpx-vp9`: 27 seconds to cut +a 5-second clip. It also writes `.webm` and appends that to `-o`, so the file you +asked for is not the file on disk. Pin `bv*[vcodec^=avc1][height<=N]+ba[acodec^=mp4a]` +and `--merge-output-format mp4`. + +**`--ignore-config` is not optional.** The operator's own yt-dlp config redirects +output and attaches thumbnail and metadata post-processors. Without it the clips +land somewhere else entirely and the build reports success having produced nothing +where it was looking. + +**yt-dlp exit 101 is success.** It is the clean early stop (`break-on-existing`, +`--max-downloads`). Treat it as success, as the rest of the repo does. + +**Rumble HLS needs `-extension_picky 0`, as a RETRY and never as a default.** +Rumble serves HLS whose segments are named `.tar`, which ffmpeg 8 rejects outright +("URL … is not in allowed_segment_extensions"), killing the fetch with exit 183 — +and Rumble ships no progressive fallback, so every Rumble clip is unbuildable +without it. But the option lives on the **HLS demuxer**: pass it against a +progressive URL (YouTube's googlevideo mp4) and ffmpeg aborts with "Option +extension_picky not found". Adding it unconditionally trades a Rumble failure for +a YouTube one. + +**`--force-keyframes-at-cuts` matters because the clip IS the citation.** Without +it the cut snaps to the nearest preceding keyframe, which can be seconds early. +Fine for scrubbing; not fine when someone is checking your quote. + +## Cutting + +**The silence threshold has to be relative to the clip, not absolute.** These are +game streams: the gaps between words are full of game audio and music. Measured on +a typical clip — mean volume −21 dB, **0** silences found at −32 dB, **25** at +−26 dB. Measure with `volumedetect` first and cut a few dB under the clip's own +mean. + +**Snapping only proves the edges are quiet.** It finds gaps in audio, which is +usually a word boundary and is not guaranteed to be. A speaker who does not pause +gets the unsnapped cut. + +**ASR cue boundaries are line-wrap boundaries, not sentence boundaries.** They land +mid-sentence routinely and mid-word often. That is the whole reason +`resolve-windows.mjs` exists — and the reason a clip bench that shows you the cues +is worth more than one that only shows you a waveform. + +**Some uploads carry no punctuation at all.** Sentence widening then has nothing to +find and silently does nothing. The clip bench says so explicitly rather than +leaving you to wonder why "extend to sentence end" is inert; set the edge by ear +and lock it. + +## Manifests + +**`resolve-windows.mjs` is a fixed point, and that was not free.** The manifest +stores times at 2 dp, so a value read back can sit a hair below the cue end it came +from — which lands the end lookup on the *previous* cue, runs the forward search on +to the *next* sentence, and grows the same clip a little on every run. Hence `EPS` +in the lookup and the 0.05 s deadband on applying a change. **Anything that writes a +window must round to 2 dp**, or it reintroduces exactly this bug. + +**`writeJsonAtomic` would vandalise a manifest.** It writes `JSON.stringify(v, null, 1)`; +`resolve-windows.mjs` writes `null, 2` plus a trailing newline. Saving one window +edit through the default writer reformats 600 lines and makes the diff unreadable. +Manifest writes pass `{space: 2, newline: true}`. + +**`lock: true` is the norm, not the exception.** Measured across the six real +manifests: ferret-rescue locks 1 of 10, every other manifest locks **100%**. A +human-chosen window usually *is* the truth, and widening it would undo an editorial +decision — cutting a quote short is a choice, and a single ASR cue often carries a +whole paragraph. So `lock` is a first-class explained control in the bench, not an +advanced toggle, and moving an edge to somewhere `widen()` would not produce offers +to set the matching lock. + +**`siteOrigin` is unvalidated and has already shipped broken twice.** +`quartering-employee-count` has none — 19 QR codes encoding `undefined/?v=…` — and +`ferret-rescue` has `http://localhost:3000`, a shipped video whose codes resolve to +nothing on anyone's phone. `umtool check` exists largely for this. + +**A clip may name its own `channel`.** The same streamer's VODs are mirrored across +several archived channels, and cue files are keyed by channel, so one manifest-wide +slug cannot find them all. + +**Building the Rumble site-id -> directory-slug map takes three seconds.** Every +`transcript.cues.json` carries its site id in the first ~400 bytes, so you never +have to parse the whole file — read the head of each and index it. On 7,925 +Rumble videos that is 3.6 s, and it is what turns a report's share links (which +carry the *site* id) into manifest `video` fields (which must be the *slug*): + +```py +for d in os.listdir("."): # transcripts/channels/<chan>/data + head = open(f"{d}/transcript.cues.json", "rb").read(400).decode("utf8", "replace") + m = re.search(r'"id"\s*:\s*"([^"]+)"', head) + if m: out[m.group(1)] = d # site id -> directory slug +``` + +**Rumble ids: the manifest's `video` must be the local directory slug.** Cue files +live under the URL slug, not the MCP video id. A manifest using the id finds +nothing — and finds it twenty minutes into a build, unless `umtool check` ran first. + +## Rendering + +**ImageMagick's `-size` leaks into Pango.** It applies to the *next* image +operation, so a stale `-size` silently changes the text raster. + +**drawtext does not wrap, and commas are structural in a filtergraph.** Wrap to a +character budget, write the text to a file and use `textfile=`, so nothing needs +shell or filter escaping. Single-quote any expression containing a comma. + +**Stream titles need emoji and `!command` suffixes stripped** or they render as +tofu in the attribution line. + +**Segments are encoded to identical parameters on purpose**, so the final concat is +a stream copy. Mismatched streams are the usual reason a naive concat produces a +broken or audio-desynced file. + +**A QR must be fully opaque and must keep its quiet zone.** A translucent QR will +not scan, and the white border is part of the symbol, not decoration. + +**A long URL cannot be a small QR, and the failure is silent.** The +employee-count share link carries 23 channel filters and is ~1.4 k characters: +that is a version-40 symbol, **177 modules**, which inside a 132 px tile is +0.7 px per module. It renders, it looks like a QR, and nothing on Earth scans it. +Keep a short `provenance.qrLink` for card tiles (~200 chars → 63 modules → 2.1 px +per module, which `zbarimg` reads off the raster) and **scan a rendered frame** +rather than trusting that a code appeared. + +**Resize a QR with `-filter point`.** Any resampling filter blurs the module +edges, and a blurred QR stops scanning. The geometry has to be decided before the +code is generated, because the tile it sits in is sized in the rail's arithmetic. + +**An opaque curtain parks OUTSIDE the window it covers, which may be on top of +something.** The rail's log curtain ends one window-height below the log — which +was empty ground until the QR tile moved into the rail's foot. The fix is +ordering, not geometry: the tile overlays after the curtain. + +**ffmetadata is line-based and `=`, `;`, `#`, `\` are structural.** A chapter title +carrying any of them has to be escaped or the file silently mis-parses. + +## Chrome rendered in a browser (`chromeEngine: "hyperframes"`) + +**A bare `local()` `@font-face` silently falls back in the render browser.** It +resolves fine on a desktop, so a snapshot looks right and every metric in the +rendered band is wrong. Copy the real `.ttf` in beside the composition and +`url()` it. + +**Naming a real family anywhere in the fallback stack makes the compiler go and +FETCH it from Google Fonts.** That is a network dependency at render time *and* a +different cut of the face from the local file the ffmpeg cards use — so the two +halves of the same frame disagree about metrics. Give the embedded face a private +family name (`'Band'`) and let the stack fall through to `sans-serif`. + +**Animate the reveal with ONE clip-path, not a `stroke-dashoffset` per path.** The +obvious build offsets each series' dash. It looks right for the strokes and wrong +for everything else: a filled region (a gap band between two series) has no stroke +to offset, so it appears whole the instant it fades in and the chart shows an +answer the playhead has not reached. + +**PNG regions overlay BEFORE the rail chain, not after.** The rail chain ends in +`format=yuv420p`, and overlaying an alpha sequence onto yuv420p is the same +alpha-subsampling trap the rail already documents, one layer later. `format=yuv444` +and `shortest=1` on every overlay; one `format=yuv420p` at the very end. + +**A finite image sequence needs no `-t`.** Unlike `-loop 1` it ends by itself, so +the deadlock five chained loops hit cannot happen — but `shortest=1` is still +required, because a secondary longer than the main extends the output. + +**Reserving the band's height is not the same as `render.footerHeight`.** Three +renderers were reading the manifest's 100 while the band owned 200, so the closing +chart drew its footnotes underneath it and the ledger scroll cropped 100 px short. +One `reservedFooterHeight()` helper now serves all of them — and `railGeometry` +was the fourth, found later: the rail column ran 100 px past the band's own top +edge, about two rows of its log window. + +**Cost, measured.** 1500×200 alpha, 30 fps: ~30 ms/frame, ~32 KB/frame. The full +374 s band is 11,460 frames, 348 s of wall clock and 372 MB, on a box already +running something else. + +## Rail strips and rolling counters + +**A slab that slides moves text that did not change.** The tally used to be four +rows walked by one `crop`, so a coffee figure changing dragged "The Quartering" +up the screen with it. Static parts (swatch, company label) belong in the chrome; +only the cell that changes may move. + +**A crop can only walk, so the DIRECTION of a roll is a property of the strip's +layout.** Lay the pair as `[old, new]` and the window walks down (content moves +up, a rise); lay it as `[new, old]` and it walks up (a fall). There is no +"direction" term in the ramp at all. + +**Two rows of a rolling strip must hold identical content wherever the crop +steps.** Between transitions the window repositions instantly to the next pair's +first row; that step is invisible only because the row it leaves and the row it +arrives at are the same picture. It also means a delta chip has to ride on both +rows — blanking it at the step is what makes an invisible reposition visible. + +**A manifest value beats a default, which is obvious and still cost twenty +minutes.** Changing `rowHeight`'s default in `railGeometry` changed nothing, +because the manifest set it explicitly. Print the geometry rather than reasoning +about which value won. + +## The tool itself + +**`ffmpeg` inside a `while read` loop eats the loop's stdin** and the loop stops +early, silently, having "passed". Use `ffmpeg -nostdin`. (`ffprobe` has no such +flag and errors if given one.) + + + +**`node -e console.log(<number>)` emits ANSI escapes on a TTY**, which corrupt +ffmpeg filtergraphs and shell tests silently. + +**`pnpm lint` in `editor/` always fails** — there is no eslint config there. Use +`pnpm exec tsc --noEmit`. + +**A value import that drags `node:fs` into a client component 500s every page** and +passes typecheck. `pnpm build`, not just `tsc --noEmit`, is what catches it — which +is why the registry's *types* live in `lib/project-types.ts`, separate from the +`.mjs` that reads the disk. + +**e2e is serialized machine-wide.** A "waiting for the e2e queue" banner is normal, +not a hang; the serial suite is long. A run that wins the lock but finds its ports +bound aborts and names the offending pid. diff --git a/umtool/docs/report-video.md b/umtool/docs/report-video.md @@ -0,0 +1,263 @@ +# The `report-video` kind + +A report video is a **cited timeline**: a sweep report's findings, said by the +sources in their own voice, in order, each clip carrying a burned-in attribution +and a QR that resolves to that exact moment in the archive's own viewer. The point +is that a viewer does not have to take the edit on trust. + +The pipeline lives at `scripts/report-to-video/` and its own +[README](../../scripts/report-to-video/README.md) is the source of truth for how it +renders. This sheet covers what that README cannot know: what umtool reads, what +umtool **writes**, and the rules a UI has to honour so the CLI and the app can never +disagree. + +## The manifest is an EDL, and it is the project + +`<project>/video.manifest.json` is both the marker file for the kind and the edit +decision list. Nothing else needs to exist for a directory to be a report video. + +```jsonc +{ + "schemaVersion": 1, + "slug": "quartering-gout", // out/<slug>.mp4 is the deliverable + "title": "…", "subtitle": "…", "generatedOn": "2026-08-12", + "provenance": { … }, // how the sweep was done, and its caveats + "render": { … }, // resolution, fonts, palette, the cut knobs + "timelineNodes": [ … ], // the footer's progress track, if any + "timeline": [ … ] // the ordered cut. Array order IS the cut. +} +``` + +**Ordering is array order.** There is no `index` field and no sort. A reorder is a +move within the array, and it has to recompute `sectionEnter` — the flag that says +"this clip is the first of its section", which is what makes the footer marker +travel. umtool never reorders as a side effect of a window edit. + +### A clip entry + +```jsonc +{ "type": "clip", "id": "c04", "video": "uyz1_FIqIEk", + "channel": "hasanabi-vods3", // optional: which archived channel's cues + "start": 32980.24, "end": 32994.19, // absolute source seconds, 2 dp + "cite": 32989, // the second shown in the attribution line + "citeUrl": "https://…", // optional: overrides the derived QR target + "quote": "…", // the words this clip exists for + "note": "…", // why it is in the cut (editorial, for humans) + "chapter": "…", // chapter title; falls back to date + title + "section": 2, "sectionEnter": true, + "lock": true, "lockStart": true, "lockEnd": true } +``` + +A card entry is `{"type":"card", "id", "style", "seconds", …}` — see the pipeline +README for the styles. Cards have no window and no source. + +**The entry vocabulary is OPEN, and code must treat it that way.** A CLIP is +`type === "clip"`; everything else is a non-clip entry with no window and no +source, and it is *not* necessarily a card. One real manifest +(`quartering-employee-count`) carries `scroll` and `chart` entries beside its +ten cards. Anything that branches on card-or-clip will send `undefined` into a +path join the first time it meets a third type — which is exactly what 500'd the +project page and crashed `umtool show`, on the one manifest of six that has any. +The fixture now carries an entry of a type nothing in the code knows about, so +this cannot come back. + +## The three-stage window model + +A clip's window passes through three different notions of "where the cut is", and +confusing them is the source of most window bugs. + +| Stage | Where it comes from | Whose problem | +|---|---|---| +| **1. Cue span** | `transcript.cues.json` — the cues covering the quote | The sweep. A report only records a single start second, so a window cannot be recovered from the report alone. | +| **2. Sentence window** | `resolve-windows.mjs` walks outward to a cue ending in `.?!` | Meaning. A cue boundary is a *line-wrap* boundary; cutting there drops the lead-in that makes a quote make sense. | +| **3. Audio cut** | `build-video.mjs` snaps to a silence found in the fetched audio | Sound. A sentence boundary is still not a *speech* boundary; snapping is what stops clips cutting through a word. | + +Only stage 2 is stored. Stage 1 is recoverable from the cue file, and stage 3 is +recomputed on every build from the audio — which is why a build is reproducible +from the manifest alone, and why the clip bench edits stage 2 and shows you stage 1. + +**Everything is in absolute source seconds**, from the manifest through the bench +to the API. The fetched file's own start (`fetchStart`) is the only place a relative +number appears, and it is always derived, never stored. + +## Writing a window: the four rules + +umtool is a *second* writer of a file the CLI also writes. So: + +1. **Round to 2 dp.** Non-negotiable. `resolve-windows.mjs` is a fixed point and + both its `EPS` lookup tolerance and its 0.05 s deadband assume 2 dp storage. + Writing 4 dp makes the widener grow the clip on every subsequent run. +2. **Re-read before writing, under `withStateLock`, and write atomically.** The + file has two writers; a lost update here is lost human judgement. +3. **Preserve the CLI's formatting** — `JSON.stringify(m, null, 2) + "\n"`. The + default `writeJsonAtomic` uses indent 1, which turns a two-number edit into a + 600-line diff. +4. **Guard with an mtime token.** A `PUT` carrying a stale token is a 409, not a + silent overwrite — an agent or a CLI run may have written in between. + +A `.bak` of the last hand-authored state is kept (rate-limited, not a rolling +stack); `ferret-rescue/video.manifest.json.bak` already set that precedent. + +## `lock`, `lockStart`, `lockEnd` + +These are **editorial acknowledgements**, not advanced options, and the data says +so: across the six real manifests ferret-rescue locks 1 clip of 10 and *every other +manifest locks 100%*. + +- **`lock`** — resolve-windows must not touch this clip at all. The two cases: + the author deliberately cut a quote short (a single ASR cue often carries a whole + paragraph, so trimming to the sentence that matters is a real decision that + widening would undo), or the lead-in would drag in seconds of some *other* audio. + A live case from `quartering-gout`: two clips whose source cue file has two cues + spanning almost the whole runtime, where an unlocked widening pass would expand a + 35-second window to a 2,488-second one. +- **`lockStart` / `lockEnd`** — pin one edge exactly. Needed because sentence + detection is only as good as the ASR's punctuation, and some uploads have none. + +`lockEnd` is also how the "ends mid-sentence" warning is *acknowledged*: setting it +is the author saying "I meant to cut here", and it suppresses the warning. This +matters — one 14-clip cut shipped with 8 clips ending mid-thought. + +**The rule that prevents silent loss:** if you move an edge in the bench to a value +`widen()` would not produce, the bench offers to set the matching lock, defaulted +on. Without it, the next `resolve-windows --write` reverts your edit. + +## Provenance, and the two defects that shipped + +`provenance` is prose written by the sweep, and most of it is for humans. Three +fields are load-bearing: + +- **`siteOrigin`** — the archive origin every QR is built from. **Nothing validated + this**, and two real videos shipped broken: `quartering-employee-count` has no + `siteOrigin` at all (19 QR codes encoding `undefined/?v=…`) and `ferret-rescue` + has `http://localhost:3000` (codes that resolve to nothing on a phone). This is + the single best reason to run `umtool check` before every build. +- **`channelSlug`** — the default archived channel for cue lookups, overridable per + clip. +- **`availabilityCheckedOn`** — when sources were last confirmed fetchable. Stale in + both directions; `check-availability.mjs` refreshes it. + +## The Rumble two-ids rule + +A Rumble video has **two** ids: the site/MCP `video_id` is the *embed* id, while the +local cue directory is named for the **URL slug**. The manifest's `video` field must +be the local slug; the report cites the site id. A manifest that uses the site id +finds no cue file — and finds that out twenty minutes into a build, unless +`umtool check` ran first. `citeUrl` exists for the mirror case: a clip taken from a +copy whose archived transcript is broken should point the QR at the copy that reads. + +## Deleted sources + +A source can be deleted from YouTube after the manifest is written, and clips are a +**live network fetch** — there is no local media. The archive's own availability +data is a *snapshot*, so it goes stale in both directions. Options, in order of +preference: cut the same moment from a live mirror (via a `CHANNELS_DIR` shadow +directory — but check the archive's alignment statement first, because mirrors are +not assumed to share a clock), or convert the clip to a quote card. + +## The ledger, and why it must be adjudicated + +`ledger[]` is the claim set the rail and the closing cards read. Beside the +descriptive fields (`date`, `value`, `display`, `label`, `quote`, `src`, +`entryId`) every entry carries **six adjudication fields** — `scope`, +`scopeBasis`, `scopeConfidence`, `population`, `valueKind`, `flags` — plus the +`channel` / `video` / `cite` triple that lets a claim page fetch its own audio. + +They exist because `company` was an **undocumented interpretation** and four +hazards were riding on it: + +| hazard | what it looked like | +|---|---| +| scope ambiguity | *"I have 10 employees, my coffee company employees…"* — all-companies or coffee-only, depending on where the comma falls | +| derived, not stated | *"10 at coffee brand coffee, I've got eight staff for the live stream"* recorded as **18**, a figure nobody utters | +| population drift | "full-time salaried", "employees" and "all basically contractors" compared as one series | +| synthetic values | 10.5 for *"about 10 people, 11 people"* — a midpoint we invented and attributed to him | + +**The rule that follows:** a *stated* total may contain only **a figure he utters +as a single number for a named scope**. Sums and midpoints are ours and belong to +the *implied* series, which says so on screen. + +`scripts/report-to-video/ledger-totals.mjs` is the single implementation of that +arithmetic — the umtool reducer, the chart band and the closing card all import +it, so they cannot disagree. It **refuses to run on an unadjudicated ledger** +(`strict: true`); the inbox passes `strict: false` so it can show coherence rows +while the rest is still being worked. + +**Six** named predicates compute incoherence rather than asserting it: +`contradicts_component`, `same_day_conflict`, `self_negating`, +`population_mismatch`, `not_his_number`, `status_flip`. Deliberately **not** a +predicate: a large rise or fall between claims — fluctuation is usually the +*subject*, and flagging it would be putting a thumb on the scale. + +A seventh **optional** field, `roles`, records who he named when he enumerated +rather than counted. It is not part of the gate — see [claim-bench.md](claim-bench.md). + +Rule on them at [`/browse/<project>/claim/<id>`](claim-bench.md). + +## One manifest, two cuts + +`timeline` entries may carry `"variant": "full"`, and cards may carry +`variants: { sourced: { … } }` field overrides. `selectVariant()` in +`build-video.mjs` applies both once, immediately after the manifest is read, and +then filters `ledger` to the claims whose `entryId` survived. + +What umtool has to know about it: + +- **`out/<slug>.mp4` is still the `sourced` deliverable**, which is why + `buildStateOf()` keeps working unchanged. `full` writes `out/<slug>-full.mp4` + beside it, and registering that as a second output is a follow-up. +- **`out/clips-raw` and `out/availability.json` stay at the root** and are shared; + `cards/`, `segments/`, `qr/`, `chrome/` and `schedule.json` moved under + `out/<variant>/`. Anything reading a raw window (the clip bench, the claim + bench) is unaffected; anything reading a segment needs a variant. +- **The driver passes `--out <project>/out`**, which is the ROOT, so it is correct + as written and builds the default `sourced` cut. +- A **`ledger`** timeline entry is a new type, and the vocabulary is open, so + `clipsOf`/`cardsOf`/the "other entries" bucket already handle it. A window + cannot be written to one — `updateClip()` refuses any non-clip. + +## Deleted sources are a probe result, not an annotation + +`check-availability.mjs` probes every **ledger** source as well as every clip +source, and `out/availability.json` records the verdict with the date it ran. The +`full` cut prints that verdict on screen (`source deleted` / `source unreachable` +/ `not clipped`), so a hand-written "source since removed" in a card kicker is now +a bug: several unclipped claims come from videos the cut clips elsewhere, i.e. +demonstrably live. + +**Severity follows the CLIPS, not the source.** Widening the probe to ledger +sources immediately put six permanent `blocking` rows in the inbox reading +*"0 clip(s) cite it; the build dies here"* — a sentence that refutes itself. A +dead source that no clip cites is exactly the case the `full` cut quotes on a +card **because** it is gone, so it is `info`: named, dated, and not in anybody's +way. It becomes blocking the moment a clip cites it. + +## A field that changes meaning has to be chased through every renderer + +`valueKind: "derived"` was set on the 18, and the first build still shipped a rail +tally reading **"Spans everything 18"** directly beside a card reading *"He never +said eighteen"*. Three separate readers were still taking the old meaning: the +tally took the latest value of any kind, the ledger's own `label` still said "the +only explicit sum in the corpus", and the closing chart plotted off a legacy +`plotted` flag. + +Nothing caught it. Not the tests, not `verify-build`, not `umtool check` — every +one of them was true about a video that contradicted itself on screen. **Looking +at a frame** was the only thing that found it. + +## Discovered by getting it wrong once + +- **A build that loses a clip is worse than a build that fails.** `--continue-on-error` + finishes everything buildable and then *refuses to concatenate*, because a + finished file quietly missing a citation looks complete. +- **The cache is content-addressed by window, and that was being wasted.** + Before containing-window reuse, every window edit was a fresh download: + `ferret-rescue/out/clips-raw` holds 31 files for 10 clips, one source four times + over overlapping windows. +- **Mirrors are not assumed to share a clock.** Timestamps mapped from one mirror to + another have to be verified against the mirror's own cue text, not assumed. +- **Every entry is a chapter, so counts must agree.** `verify-build.mjs` compares + chapter count to *timeline* length, not clip count — a mismatch means the file + and the timeline disagree about what is in it. +- **`--chapters-only` is cheap and separate.** Retitling chapters does not need a + re-encode; the per-clip segments on disk are all the offsets need. diff --git a/umtool/e2e/browse.spec.ts b/umtool/e2e/browse.spec.ts @@ -13,16 +13,18 @@ const dur = (s: number) => `${Math.floor(s / 60)}:${String(s % 60).padStart(2, " test("the index lists both songs, and says which cuts are missing", async ({ page }) => { await page.goto("/browse"); - await expect(page.locator("[data-song=alpha]")).toBeVisible(); - await expect(page.locator("[data-song=beta]")).toBeVisible(); + // A project id is a PATH now, so the suffix is what identifies a song + // regardless of how deep the fixture nests it. + await expect(page.locator("[data-project$='/alpha']")).toBeVisible(); + await expect(page.locator("[data-project$='/beta']")).toBeVisible(); // beta has wide and wide-short only. A cut list derived from the directory // would make it look complete; the list is fixed precisely so it cannot. - await expect(page.locator("[data-song=beta] [data-missing]")).toHaveAttribute( + await expect(page.locator("[data-project$='/beta'] [data-missing]")).toHaveAttribute( "data-missing", "vertical,vertical-short", ); - await expect(page.locator("[data-song=alpha] [data-missing]")).toHaveCount(0); + await expect(page.locator("[data-project$='/alpha'] [data-missing]")).toHaveCount(0); }); // THE assertion this whole feature turns on. `wide-short-nokit` must attribute diff --git a/umtool/e2e/build.spec.ts b/umtool/e2e/build.spec.ts @@ -0,0 +1,202 @@ +import { test, expect } from "@playwright/test"; +import { execFileSync } from "node:child_process"; +import { existsSync, readdirSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +// --------------------------------------------------------------------------- +// The build driver. +// +// Every one of these runs OFFLINE. The fixture writes stub YTDLP_BIN and +// QRENCODE_BIN and playwright.config points the server at them, so the chain -- +// preflight, resolve, build, verify -- exercises its real code with no network +// and a deterministic answer. The stub reports `gone1` removed the way a deleted +// upload is, which is what gives the preflight's blocking path a true answer +// rather than a plausible one. +// +// build-fixture is its own project. The bench specs write windows and the index +// specs assert them; a build stamps deliverables aside and gets refused when one +// is newer than its manifest. Sharing would make each suite's result depend on +// the other's order. +// --------------------------------------------------------------------------- + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const UMTOOL = path.join(HERE, ".."); +const FIXTURE = path.join(UMTOOL, ".e2e-song"); +const PROJECT = "reports/build-fixture"; +const OUT = path.join(FIXTURE, "reports", "build-fixture", "out"); +const FINAL = path.join(OUT, "build-fixture.mp4"); + +type Job = { id: string; state: string; error: string | null }; + +async function waitFor( + request: { get: (u: string) => Promise<{ json: () => Promise<unknown> }> }, + id: string, + ms = 120_000, +) { + const until = Date.now() + ms; + while (Date.now() < until) { + const j = (await (await request.get(`/api/report/build?job=${id}`)).json()) as Job & { + events: { ev: string; id?: string }[]; + }; + if (j.state !== "running") return j; + await new Promise((r) => setTimeout(r, 400)); + } + throw new Error("the job never finished"); +} + +test("a dry run prints the exact chain, and runs nothing", async ({ request }) => { + const r = await request.post("/api/report/build?dry=1", { + data: { project: PROJECT, preset: "fast" }, + }); + const j = (await r.json()) as { dry: boolean; steps: { label: string; argv: string[]; timeoutMs: number }[] }; + expect(j.dry).toBe(true); + + // Preflight is a STEP, not a preamble: it is the one fact that goes stale in + // both directions, and without it a dead source is found twenty minutes and a + // dozen paid-for fetches in. + expect(j.steps[0].argv.join(" ")).toContain("check-availability.mjs"); + expect(j.steps[1].argv.join(" ")).toContain("resolve-windows.mjs"); + // Dry: the widener is never handed --write from here. Applying is explicit. + expect(j.steps[1].argv).not.toContain("--write"); + expect(j.steps[2].argv.join(" ")).toContain("build-video.mjs"); + expect(j.steps[2].argv).toContain("--continue-on-error"); + expect(j.steps.at(-1)!.argv.join(" ")).toContain("verify-build.mjs"); + + // The build asks for more than the 15-minute default, which exists to catch + // the accidental hour-long job and would SIGKILL a real 19-clip run. + expect(j.steps[2].timeoutMs).toBeGreaterThanOrEqual(15 * 60_000); +}); + +test("a build runs end to end, offline, and the file is verified", async ({ request }) => { + const start = await request.post("/api/report/build", { + data: { project: PROJECT, preset: "fast" }, + }); + expect(start.ok()).toBeTruthy(); + const { job } = (await start.json()) as { job: Job }; + + const done = await waitFor(request, job.id); + expect(done.state, done.error ?? "").toBe("done"); + expect(existsSync(FINAL)).toBe(true); + + // Per-clip progress, from the pipeline's own NDJSON. "Step 3 of 4, running" + // is not progress when step 3 is the twenty-minute one. + const events = (done as unknown as { events: { ev: string; id?: string }[] }).events ?? []; + expect(events.some((e) => e.ev === "segment" && e.id === "c01")).toBe(true); + expect(events.some((e) => e.ev === "segment" && e.id === "c02")).toBe(true); + expect(events.some((e) => e.ev === "done")).toBe(true); + + // c01's window is already cached, so the build must REUSE it rather than ask + // the stub for it -- the same containment rule the bench's wide fetch relies on. + const fetches = events.filter((e) => e.ev === "fetch"); + expect(fetches.find((e) => e.id === "c01")).toMatchObject({ cached: true }); +}); + +test("an existing deliverable is not destroyed to make a new one", async ({ request }) => { + // The previous test built it, so out/ is now newer than the manifest. + expect(existsSync(FINAL)).toBe(true); + + const refused = await request.post("/api/report/build", { + data: { project: PROJECT, preset: "fast" }, + }); + expect(refused.status()).toBe(409); + const j = (await refused.json()) as { needsReplace: boolean; error: string }; + expect(j.needsReplace).toBe(true); + expect(j.error).toContain("newer than the manifest"); + + // With replace=1 the old file is STAMPED ASIDE, never overwritten -- it cost + // an hour of network fetches in the real case. + const ok = await request.post("/api/report/build?replace=1", { + data: { project: PROJECT, preset: "fast" }, + }); + expect(ok.ok()).toBeTruthy(); + const stamped = readdirSync(OUT).filter((n) => /^build-fixture\.\d{8}-\d{4}\.mp4$/.test(n)); + expect(stamped.length).toBeGreaterThan(0); + + const { job } = (await ok.json()) as { job: Job }; + const done = await waitFor(request, job.id); + expect(done.state, done.error ?? "").toBe("done"); +}); + +test("a source that is gone blocks the build at step 1", async ({ request }) => { + const start = await request.post("/api/report/build", { + data: { project: "reports/gone-fixture", preset: "fast" }, + }); + expect(start.ok()).toBeTruthy(); + const { job } = (await start.json()) as { job: Job }; + + const done = await waitFor(request, job.id); + // Failed at the preflight, having encoded nothing. + expect(done.state).toBe("failed"); + expect(existsSync(path.join(FIXTURE, "reports", "gone-fixture", "out", "gone-fixture.mp4"))).toBe( + false, + ); + + // And it wrote down what it found, so the inbox can report it without running + // yt-dlp itself. + const avail = path.join(FIXTURE, "reports", "gone-fixture", "out", "availability.json"); + expect(existsSync(avail)).toBe(true); +}); + +test("a second build is refused while one is running, and cancel leaves no orphan", async ({ + request, +}) => { + const first = await request.post("/api/report/build?replace=1", { + data: { project: PROJECT, preset: "final" }, + }); + expect(first.ok()).toBeTruthy(); + const { job } = (await first.json()) as { job: Job }; + + // One job, process-wide. Two builds would interleave in one out/segments, and + // two in different projects would still fight over yt-dlp and the CPU. + const second = await request.post("/api/report/build", { + data: { project: "reports/report-fixture", preset: "fast" }, + }); + expect(second.status()).toBe(409); + + await request.post(`/api/report/build?cancel=${job.id}`); + const done = await waitFor(request, job.id, 60_000); + expect(done.state).toBe("failed"); + + // The GRANDCHILDREN are the point. build-video shells out, so child.kill() + // reaps the node process and leaves yt-dlp and ffmpeg running -- the same + // failure the diarize backfill had. Killing the process GROUP is what stops it. + // + // Scoped to THIS suite's own children, two ways, and both are needed. + // + // It runs pgrep without a shell: going through `bash -lc` puts the pattern + // into bash's own command line and pgrep matches the shell that is asking. + // And it keeps only lines that are an actual `node …/build-video.mjs` + // invocation against THIS FIXTURE -- because this is a shared machine, and a + // concurrent agent running its own builds (or merely waiting on one with + // `pgrep -f build-video.mjs` in its command line) is not an orphan of ours. + // Asserting the machine holds no build-video process at all is an assertion + // about somebody else's work. + // + // Polled rather than asserted once: the kill is SIGTERM to the group and + // SIGKILL five seconds later, so "gone" is a state it reaches rather than one + // it is in the instant the job reports failed. + const ours = () => { + let out = ""; + try { + out = execFileSync("pgrep", ["-af", "build-video.mjs"]).toString(); + } catch { + return ""; // pgrep exits 1 when nothing matches, which is the good case + } + return out + .split("\n") + .filter(Boolean) + .filter((line) => { + const cmd = line.slice(line.indexOf(" ") + 1); + return /^\S*node\b/.test(cmd) && cmd.includes(FIXTURE); + }) + .join("\n"); + }; + let strays = ""; + for (let i = 0; i < 30; i += 1) { + strays = ours(); + if (!strays) break; + await new Promise((r) => setTimeout(r, 400)); + } + expect(strays, `left running: ${strays}`).toBe(""); +}); diff --git a/umtool/e2e/claim-bench.spec.ts b/umtool/e2e/claim-bench.spec.ts @@ -0,0 +1,182 @@ +import { test, expect } from "@playwright/test"; +import { readFileSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +// --------------------------------------------------------------------------- +// The claim bench, and the two decision kinds that drive it. +// +// The fixture ledger is built so every answer here is exact: +// +// k01 is fully adjudicated and pinned to c01 — the settled case. +// k02 is adjudicated but `valueKind: "derived"`, so `not_his_number` must +// fire on it and it must stay OUT of the stated series. +// k03 carries none of the six fields, so it is the one blocking row. +// +// Every test in this file writes, so it uses bench-fixture rather than +// report-fixture — the same split the clip bench already keeps. +// --------------------------------------------------------------------------- + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const UMTOOL = path.join(HERE, ".."); +const FIXTURE = path.join(UMTOOL, ".e2e-song"); +const PROJECT = "reports/bench-fixture"; +const MANIFEST = path.join(FIXTURE, "reports", "bench-fixture", "video.manifest.json"); +const claimPage = (id: string) => `/browse/${PROJECT}/claim/${id}`; + +type Claim = { + id: string; + scope?: string; + scopeBasis?: string; + scopeConfidence?: string; + population?: string; + valueKind?: string; + flags?: string[]; + roles?: Array<{ role: string; count: number; verbatim: string }>; +}; + +const readClaim = (id: string): Claim => { + const m = JSON.parse(readFileSync(MANIFEST, "utf8")) as { ledger: Claim[] }; + return m.ledger.find((c) => c.id === id)!; +}; + +const tokenFor = async ( + request: { get: (u: string) => Promise<{ json: () => Promise<unknown> }> }, + claim: string, +) => { + const r = await request.get(`/api/report/claim?project=${encodeURIComponent(PROJECT)}&claim=${claim}`); + return (await r.json()) as { token: string; gaps: string[]; cues: unknown[]; at: number }; +}; + +test("an unadjudicated claim is one BLOCKING row, naming the fields it lacks", async ({ page }) => { + await page.goto("/browse/decisions"); + const row = page.getByText(/no ruling on/).first(); + await expect(row).toBeVisible(); + // All six, because nobody has touched k03. + await expect(row).toContainText("scope"); + await expect(row).toContainText("population"); + await expect(row).toContainText("valueKind"); +}); + +test("the claim page opens on the quote inside its context, not on the quote alone", async ({ page }) => { + await page.goto(claimPage("k01")); + await expect(page.locator("[data-claim='k01']")).toBeVisible(); + const ctx = page.getByTestId("claim-context"); + await expect(ctx).toBeVisible(); + // The cited cue is marked INSIDE the paragraph — that mark is the whole + // point, because it is what shows the quote had someone else talking round it. + await expect(ctx.locator("[data-cited='1']").first()).toBeVisible(); +}); + +test("an adjudicated claim shows its ruling, an unadjudicated one shows the gap", async ({ page }) => { + await page.goto(claimPage("k01")); + await expect(page.getByText("adjudicated", { exact: true })).toBeVisible(); + await expect(page.getByTestId("scope-media")).toHaveAttribute("aria-pressed", "true"); + await expect(page.getByTestId("valuekind-uttered")).toHaveAttribute("aria-pressed", "true"); + + await page.goto(claimPage("k03")); + await expect(page.getByText(/unadjudicated — missing/)).toBeVisible(); + await expect(page.getByTestId("scope-media")).toHaveAttribute("aria-pressed", "false"); +}); + +test("a ruling round-trips to the ledger", async ({ page }) => { + await page.goto(claimPage("k03")); + await page.getByTestId("scope-coffee").click(); + await page.getByTestId("confidence-read").click(); + await page.getByTestId("population-contractor").click(); + await page.getByTestId("valuekind-synthetic").click(); + await page.getByTestId("claim-basis").fill("the phrase that settles it"); + await page.getByTestId("claim-flags").fill("a hand-written flag"); + await page.getByTestId("claim-save").click(); + + await expect(page.getByTestId("claim-note")).toContainText("saved"); + + const c = readClaim("k03"); + expect(c.scope).toBe("coffee"); + expect(c.scopeConfidence).toBe("read"); + expect(c.population).toBe("contractor"); + expect(c.valueKind).toBe("synthetic"); + expect(c.scopeBasis).toBe("the phrase that settles it"); + expect(c.flags).toEqual(["a hand-written flag"]); +}); + +test("a blank scopeBasis is refused — an adjudication with no basis is an opinion", async ({ request }) => { + const { token } = await tokenFor(request, "k01"); + const r = await request.put("/api/report/claim", { + data: { project: PROJECT, claim: "k01", token, scopeBasis: " " }, + }); + expect(r.status()).toBe(400); + expect(await r.text()).toContain("scopeBasis"); +}); + +test("a value outside the vocabulary is refused, not coerced", async ({ request }) => { + const { token } = await tokenFor(request, "k01"); + const r = await request.put("/api/report/claim", { + data: { project: PROJECT, claim: "k01", token, population: "staff" }, + }); + expect(r.status()).toBe(400); + expect(await r.text()).toContain("population"); + // …and the stored ruling is untouched. + expect(readClaim("k01").population).toBe("employees"); +}); + +test("a stale token is a 409, never a silent overwrite", async ({ request }) => { + const r = await request.put("/api/report/claim", { + data: { project: PROJECT, claim: "k01", token: "0", scope: "coffee" }, + }); + expect(r.status()).toBe(409); + expect(readClaim("k01").scope).toBe("media"); +}); + +test("a fired predicate is an OPEN row, in plain words", async ({ page }) => { + await page.goto("/browse/decisions"); + // k02 is `derived`, so the sum is ours and the inbox has to say so. + await expect(page.getByText(/our sum, not his figure/)).toBeVisible(); +}); + +test("the context read is ±90s around the cite, not the clip's window", async ({ request }) => { + const d = await tokenFor(request, "k01"); + expect(d.at).toBe(3); + expect(Array.isArray(d.cues)).toBe(true); + expect(d.gaps).toEqual([]); +}); + +// --------------------------------------------------------------------------- +// The roster — the seventh field, and the only optional one. +// --------------------------------------------------------------------------- + +test("a roster round-trips, and is not part of the gate", async ({ page }) => { + await page.goto(claimPage("k01")); + await page.getByTestId("claim-roles").fill( + "2 | video editor | two video editors\n1 | graphics designer | a graphics designer", + ); + // The preview is what the rail will draw, so it is worth seeing before saving. + await expect(page.getByText("reads as “2 editors · 1 designer”")).toBeVisible(); + await page.getByTestId("claim-save").click(); + await expect(page.getByTestId("claim-note")).toContainText("saved"); + + expect(readClaim("k01").roles).toEqual([ + { count: 2, role: "video editor", verbatim: "two video editors" }, + { count: 1, role: "graphics designer", verbatim: "a graphics designer" }, + ]); + // …and the claim was already adjudicated without one, which is the point: + // `roles` is not one of the six. + await expect(page.getByText("adjudicated", { exact: true })).toBeVisible(); +}); + +test("a roster line with no verbatim cannot be saved", async ({ page }) => { + await page.goto(claimPage("k02")); + await page.getByTestId("claim-roles").fill("2 | video editor"); + await expect(page.getByText(/needs count \| role \| his words/)).toBeVisible(); + await expect(page.getByTestId("claim-save")).toBeDisabled(); +}); + +test("a malformed roster is refused by the API too, not just by the page", async ({ request }) => { + const { token } = await tokenFor(request, "k02"); + const r = await request.put("/api/report/claim", { + data: { project: PROJECT, claim: "k02", token, roles: [{ role: "", count: 2, verbatim: "x" }] }, + }); + expect(r.status()).toBe(400); + expect(await r.text()).toContain("roles"); + expect(readClaim("k02").roles).toBeUndefined(); +}); diff --git a/umtool/e2e/clip-bench.spec.ts b/umtool/e2e/clip-bench.spec.ts @@ -0,0 +1,209 @@ +import { test, expect } from "@playwright/test"; +import { execFileSync } from "node:child_process"; +import { readFileSync, existsSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +// --------------------------------------------------------------------------- +// The clip bench. +// +// The fixture is built so every answer here is exact rather than plausible: +// +// vid1 is PUNCTUATED and carries a run-on cue at 3-6s. So c01 (3.00-6.00) +// ends mid-sentence, and widen() must carry its end to 9.00 -- the next cue +// that closes one. c02 (9.00-12.00) ends on a full stop. c04 ends inside a +// run-on cue too but sets lockEnd, which is the author saying "I meant to cut +// here" and must silence the warning. +// +// vid2 has no terminator anywhere, which is the real degradation in this +// corpus. c03's edges cannot be judged against sentences at all, and the +// bench has to SAY that rather than quietly claiming the cut is fine. +// +// vid1_0.00-9.00.mp4 is tone / silence / tone / silence / tone with the +// silences centred on 3.0s and 6.0s. Verified in the file itself: +// silencedetect reports 2.90-3.11 and 5.92-6.11. +// --------------------------------------------------------------------------- + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const UMTOOL = path.join(HERE, ".."); +const FIXTURE = path.join(UMTOOL, ".e2e-song"); +// bench-fixture, NOT report-fixture. Every test in this file WRITES, and the +// index and decision specs assert what report-fixture's windows are -- sharing +// one project made the suite pass or fail on which spec file ran first. +const PROJECT = "reports/bench-fixture"; +const MANIFEST = path.join(FIXTURE, "reports", "bench-fixture", "video.manifest.json"); +const bench = (clip: string) => `/browse/${PROJECT}/clip/${clip}`; + +const readClip = (id: string) => { + const m = JSON.parse(readFileSync(MANIFEST, "utf8")) as { + timeline: { id: string; start: number; end: number; lockEnd?: boolean }[]; + }; + return m.timeline.find((e) => e.id === id)!; +}; + +const token = async (request: { get: (u: string) => Promise<{ json: () => Promise<unknown> }> }, clip: string) => { + const r = await request.get(`/api/report/clip?project=${encodeURIComponent(PROJECT)}&clip=${clip}`); + return (await r.json()) as { token: string }; +}; + +test("the bench opens on the cached window, with the clip inside it", async ({ page }) => { + await page.goto(bench("c01")); + await expect(page.locator("[data-bench=c01]")).toBeVisible(); + await expect(page.getByTestId("waveform")).toBeVisible(); + // The <video> is the cached source window served whole, not an ffmpeg slice + // per drag -- that is what makes dragging free. + await expect(page.getByTestId("clip-video")).toHaveAttribute("src", /api\/report\/raw/); +}); + +test("a clip that ends mid-sentence says so; lockEnd silences it", async ({ page }) => { + await page.goto(bench("c01")); + await expect(page.locator("[data-warn=mid-sentence]")).toBeVisible(); + // The cut lands inside this cue, and quoting it is the point -- "ends + // mid-sentence" alone does not tell you what you are cutting off. + await expect(page.locator("[data-warn=mid-sentence]")).toContainText("and because"); + + await page.goto(bench("c04")); + await expect(page.locator("[data-warn=mid-sentence]")).toHaveCount(0); +}); + +test("an unpunctuated source is admitted rather than judged", async ({ page }) => { + await page.goto(bench("c03")); + // Not "this cut is fine": the question cannot be answered from this source. + await expect(page.locator("[data-warn=no-punctuation]")).toBeVisible(); + await expect(page.locator("[data-warn=mid-sentence]")).toHaveCount(0); +}); + +test("the bench predicts what the widener would do before you run it", async ({ page }) => { + await page.goto(bench("c01")); + const warn = page.locator("[data-warn=would-revert]"); + await expect(warn).toBeVisible(); + // 3.00-6.00 widens to 3.00-9.00: the next cue that closes a sentence. Knowing + // that BEFORE `resolve-windows --write` is what stops the widener silently + // reverting an edit. + await expect(warn).toContainText("0:09.00"); +}); + +test("the cue rail marks the cues that close a sentence", async ({ page }) => { + await page.goto(bench("c01")); + // vid1: four of its first five cues end in a full stop; the run-on one does not. + await expect(page.locator("[data-cue][data-ends-sentence='1']").first()).toBeVisible(); + const runOn = page.locator("[data-cue='3'][data-ends-sentence='0']"); + await expect(runOn).toBeVisible(); +}); + +test("a saved window is stored at 2 dp, and the CLI's formatting survives", async ({ request }) => { + const { token: t } = await token(request, "c01"); + const res = await request.put("/api/report/window", { + data: { project: PROJECT, clip: "c01", start: 3.123456, end: 6.987654, token: t }, + }); + expect(res.ok()).toBeTruthy(); + + // 2 dp is NOT cosmetic: resolve-windows.mjs is a fixed point, and both its EPS + // lookup and its 0.05s deadband assume it. 4 dp makes the widener grow the + // same clip on every run. + const e = readClip("c01"); + expect(e.start).toBe(3.12); + expect(e.end).toBe(6.99); + + // Indent 2 plus a trailing newline, which is what the CLI writes. The default + // writeJsonAtomic uses indent 1 and would turn this into a 600-line diff. + const raw = readFileSync(MANIFEST, "utf8"); + expect(raw).toContain('\n "schemaVersion": 1,'); + expect(raw.endsWith("}\n")).toBe(true); + + // One copy of the last hand-authored state, made on the first write. + expect(existsSync(`${MANIFEST}.bak`)).toBe(true); +}); + +test("a stale token is refused rather than allowed to overwrite", async ({ request }) => { + const { token: t } = await token(request, "c02"); + const first = await request.put("/api/report/window", { + data: { project: PROJECT, clip: "c02", start: 9, end: 11.5, token: t }, + }); + expect(first.ok()).toBeTruthy(); + + // The same token again: somebody (resolve-windows --write, an agent, another + // tab) wrote in between, and what they wrote is a judgement. + const second = await request.put("/api/report/window", { + data: { project: PROJECT, clip: "c02", start: 9, end: 10, token: t }, + }); + expect(second.status()).toBe(409); + const j = (await second.json()) as { stale: boolean }; + expect(j.stale).toBe(true); + // And it did NOT write. + expect(readClip("c02").end).toBe(11.5); +}); + +test("the edit the bench predicts is a FIXED POINT for the widener", async ({ request }) => { + const { token: t } = await token(request, "c01"); + // The value the bench said resolve-windows would produce. + await request.put("/api/report/window", { + data: { project: PROJECT, clip: "c01", start: 3.0, end: 9.0, token: t }, + }); + + const out = execFileSync( + "node", + [path.join(UMTOOL, "..", "scripts", "report-to-video", "resolve-windows.mjs"), MANIFEST], + { encoding: "utf8", env: { ...process.env, CHANNELS_DIR: path.join(FIXTURE, "channels") } }, + ); + // Setting an edge where a sentence actually ends means the next --write is a + // no-op on that clip. That round-trip is the whole argument for the bench. + const line = out.split("\n").find((l) => l.startsWith("c01"))!; + expect(line).toContain("3.0–9.0 -> 3.0–9.0"); +}); + +test("the raw window serves byte ranges, and refuses a file that is not this clip's", async ({ + request, +}) => { + const q = `project=${encodeURIComponent(PROJECT)}&clip=c01`; + const full = await request.get(`/api/report/raw?${q}`); + expect(full.status()).toBe(200); + expect(full.headers()["accept-ranges"]).toBe("bytes"); + // The absolute source second the file starts at, so the client converts + // without a second request. + expect(full.headers()["x-fetch-start"]).toBe("0"); + + const part = await request.get(`/api/report/raw?${q}`, { headers: { Range: "bytes=0-1023" } }); + // Without a 206 the <video> element will not seek in a stream it did not + // fully download, which is the whole interaction. + expect(part.status()).toBe(206); + expect(part.headers()["content-range"]).toMatch(/^bytes 0-1023\/\d+$/); + + // `file` must be a member of the server's own scan for THIS clip's video. + expect((await request.get(`/api/report/raw?${q}&file=vid2_0.00-9.00.mp4`)).status()).toBe(404); + expect((await request.get(`/api/report/raw?${q}&file=../../../etc/passwd`)).status()).toBe(404); +}); + +test("peaks come back in absolute source seconds, symmetric, with the silences in them", async ({ + request, +}) => { + const r = await request.get( + `/api/report/peaks?project=${encodeURIComponent(PROJECT)}&clip=c01&n=900`, + ); + const j = (await r.json()) as { from: number; fetchStart: number; min: number[]; max: number[] }; + // Absolute, not relative to the fetched file. + expect(j.from).toBe(0); + expect(j.fetchStart).toBe(0); + // analyseMedia's envelope is positive-only; mirroring it is what makes the + // canvas draw a waveform rather than a row of upward spikes. + expect(j.min[10]).toBe(-j.max[10]); + + // The synthesised silence at 3.0s. 900 buckets over 9s is one per 10ms. + const at = (t: number) => j.max[Math.round((t / 9) * j.max.length)]; + expect(at(3.0)).toBe(0); + expect(at(6.0)).toBe(0); + expect(at(1.0)).toBeGreaterThan(0.05); +}); + +test("a bench save shows up on the project page", async ({ page, request }) => { + const { token: t } = await token(request, "c01"); + await request.put("/api/report/window", { + data: { project: PROJECT, clip: "c01", start: 3, end: 9, lockEnd: true, token: t }, + }); + + await page.goto(`/browse/${PROJECT}`); + const row = page.locator("[data-entry=c01]"); + await expect(row).toContainText("end pinned"); + // lockEnd is the acknowledgement, so the row's warning goes with it. + await expect(row).toHaveAttribute("data-mid-sentence", "0"); +}); diff --git a/umtool/e2e/deck.spec.ts b/umtool/e2e/deck.spec.ts @@ -39,8 +39,10 @@ test("/browse/decisions is not swallowed by the [song] route", async ({ page }) const res = await page.goto("/browse/decisions"); expect(res?.status()).toBe(200); await expect(page.getByRole("navigation", { name: "Breadcrumb" })).toContainText("decisions"); - // If [song] had caught it, this would be a 404 for a song called "decisions". - await expect(page.locator("[data-project=deck]")).toBeVisible(); + // If the catch-all had swallowed it, this would be a 404 for a project called + // "decisions". A decision's `project` is a PATH now, so the suffix identifies + // the song without hard-coding how deep the fixture nests it. + await expect(page.locator("[data-project$='/deck']")).toBeVisible(); }); test("deck's open decisions are exactly the three that were seeded", async ({ request }) => { diff --git a/umtool/e2e/fixtures/make-fixture.mjs b/umtool/e2e/fixtures/make-fixture.mjs @@ -263,14 +263,32 @@ if (copied.length) { // beta two cuts and nothing else -- the hole in the cut set, which a // scan-derived cut list would render as complete const VIDEOS = path.join(reports, "videos"); -/** A silent video of a known length, at a colour that identifies it on sight. */ +/** + * A silent video of a known length, at a colour that identifies it on sight. + * + * CBR is not cosmetic. listMedia() drops anything under 256 KB as "a fragment, + * a probe or a one-note extraction", and a flat colour encodes to about eight + * kilobytes -- so an unpadded fixture cut is invisible to the mix picker, and + * the grouped-picker spec sees an empty list while the code under it is + * perfectly correct. Real deliverables are 26 to 157 MB. + * + * `-b:v` alone does nothing here: libx264 defaults to CRF and ignores it on + * content this compressible. Constant bitrate forces the padding, and leaves + * the duration exactly as asked. + */ +const BULK = [ + "-c:v", "libx264", + "-b:v", "1500k", "-minrate", "1500k", "-maxrate", "1500k", "-bufsize", "3000k", + "-x264-params", "nal-hrd=cbr:force-cfr=1", +]; + const clip = (file, seconds, colour) => { mkdirSync(path.dirname(file), { recursive: true }); ff([ "-f", "lavfi", "-i", `color=c=${colour}:size=320x180:rate=15:duration=${seconds}`, "-f", "lavfi", "-i", `anullsrc=r=48000:cl=stereo:d=${seconds}`, "-t", String(seconds), - "-c:v", "libx264", "-pix_fmt", "yuv420p", "-c:a", "aac", "-ar", "48000", + ...BULK, "-pix_fmt", "yuv420p", "-c:a", "aac", "-ar", "48000", file, ]); }; @@ -586,6 +604,386 @@ writeFileSync( ), ); + +// -- PROJECTS: report videos, a sweep report, and the two routing traps -------- +// +// REPORTS_ROOT defaults to dirname(SONG_REPORTS_DIR), so everything the project +// walk sees is inside this fixture. Each of these exists to give one finding a +// TRUE ANSWER rather than a plausible one: +// +// report-fixture a good manifest whose clips have known cue text, so +// "ends mid-sentence" and "the widener would move this" +// are checkable rather than believable +// no-origin-fixture no siteOrigin at all -- the defect that shipped 19 dead +// QR codes in a real cut +// localhost-fixture siteOrigin http://localhost:3000 -- the defect that +// shipped a real video whose codes resolve on nobody's phone +// bike-fixture a sweep report with no manifest: the third kind, and the +// proof that adding one costs a registry entry and a view +// find/ a project named for a TOOL PAGE. It can never win the +// route, and before this it failed silently +// deep/nested/solo a pass-through folder chain, so collapsing has an answer +const CHANNELS = path.join(dest, "channels"); + +// Two sources, deliberately different in the one way that matters to widening. +// +// vid1 is PUNCTUATED and carries a run-on cue at 3-6s, so a clip ending at 6.0 +// ends mid-sentence and widen() must walk it out to 9.0 (the next cue that +// closes one). vid2 has NO terminator anywhere, which is the real degradation +// this corpus has -- widening cannot help there and the tool has to say so +// instead of silently doing nothing. +const CUES = { + vid1: [ + [0, 3, "This is a complete sentence."], + [3, 6, "And this one runs on and because"], + [6, 9, "of that it finishes here."], + [9, 12, "Another whole sentence entirely."], + [12, 15, "A fourth one, done."], + [15, 18, "trailing off and then"], + [18, 21, "it lands at last."], + ], + // Cited by gone-fixture. Its cue file exists (so the manifest is readable) but + // the stub yt-dlp reports it removed, which is what gives `source-unavailable` + // a true answer rather than a plausible one. + gone1: [ + [0, 3, "This upload has since been deleted."], + [3, 6, "But its transcript is still in the archive."], + ], + vid2: [ + [0, 3, "no punctuation anywhere in this upload"], + [3, 6, "the asr never emitted a full stop"], + [6, 9, "so every cue just runs into the next"], + [9, 12, "and widening has nothing to find"], + [12, 15, "which is a thing to say out loud"], + ], +}; +for (const [vid, rows] of Object.entries(CUES)) { + const dir = path.join(CHANNELS, "testchan", "data", vid); + mkdirSync(dir, { recursive: true }); + writeFileSync( + path.join(dir, "transcript.cues.json"), + JSON.stringify( + { + title: `Fixture source ${vid}`, + uploadDate: "20250101", + webpageUrl: `https://example.invalid/watch?v=${vid}`, + duration: rows[rows.length - 1][1], + cues: rows.map(([start, end, text]) => ({ start, end, text })), + }, + null, + 1, + ), + ); +} + +// A font the header's drawtext can actually load, or no header. +// +// build-video.mjs draws the citation line with `fontfile='<render.fontRegular>'` +// and an empty one is a filtergraph error, not a missing label. Rather than +// assume a font, look for one and honestly set headerHeight: 0 when there is +// none -- which is itself a documented manifest configuration ("a cut whose +// sources are listed elsewhere does not need its own attribution burnt in"). +const FONT_CANDIDATES = [ + "/usr/share/fonts/TTF/FiraSans-Regular.ttf", + "/usr/share/fonts/TTF/DejaVuSans.ttf", + "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf", + "/usr/share/fonts/liberation/LiberationSans-Regular.ttf", +]; +const FONT = FONT_CANDIDATES.find((f) => existsSync(f)) ?? null; + +const manifest = (slug, title, provenance, timeline, ledger = null) => ({ + schemaVersion: 1, + slug, + title, + subtitle: "a fixture", + generatedOn: "2026-01-01", + provenance: { channelSlug: "testchan", ...provenance }, + render: { + width: 640, + height: 360, + fps: 15, + audioRate: 48000, + audioChannels: 2, + maxHeightSource: 360, + fetchPad: 3, + snapWindow: 1.6, + transition: 0.2, + crf: 30, + preset: "ultrafast", + // No timelineNodes, so no footer -- which is what keeps ImageMagick out of + // the fixture build entirely. + footerHeight: 0, + headerHeight: FONT ? 24 : 0, + ...(FONT ? { fontRegular: FONT, fontBold: FONT } : {}), + palette: { bg: "#12100c", fg: "#f6f1e6", muted: "#a2957f", accent: "#c8752a", amber: "#ffc860" }, + }, + timelineNodes: [], + timeline, + ...(ledger ? { ledger } : {}), +}); + +const writeProject = (rel, doc, root = reports) => { + const dir = path.join(root, rel); + mkdirSync(dir, { recursive: true }); + writeFileSync(path.join(dir, "video.manifest.json"), JSON.stringify(doc, null, 2) + "\n"); + return dir; +}; + +const REPORT = writeProject( + "report-fixture", + manifest("report-fixture", "The Report Fixture", { siteOrigin: "https://archive.example" }, [ + // Ends inside the run-on cue, and is not locked -> exactly one + // clip-mid-sentence decision in the whole fixture, and widen() would move + // its end from 6.00 to 9.00. + { type: "clip", id: "c01", video: "vid1", start: 3.0, end: 6.0, cite: 3, section: 0, quote: "and because" }, + // Ends on a full stop -> clean, and widen() is a no-op. + { type: "clip", id: "c02", video: "vid1", start: 9.0, end: 12.0, cite: 9, section: 0, quote: "another whole sentence" }, + // An unpunctuated source -> feeds the no-punctuation row, never the + // mid-sentence one. + { type: "clip", id: "c03", video: "vid2", start: 1.0, end: 4.0, cite: 1, section: 0, quote: "no punctuation" }, + // Also ends mid-cue, but lockEnd ACKNOWLEDGES it, so it must stay silent. + { type: "clip", id: "c04", video: "vid1", start: 15.0, end: 18.0, cite: 15, section: 0, lockEnd: true, quote: "trailing off" }, + // A card, and an entry of a type NOTHING IN THE CODE KNOWS ABOUT. + // + // The timeline's vocabulary is open: quartering-employee-count carries + // `scroll` and `chart` entries beside its cards, and code that treated + // anything-not-a-card as a clip sent `undefined` into path.join() and 500'd + // the whole project page. Every other real manifest is clips only, which is + // exactly why that survived testing. `zz-unknown` is here so it cannot again. + { type: "card", id: "k01", style: "chapter", seconds: 3, heading: "A card" }, + { type: "zz-unknown", id: "z01", seconds: 5, heading: "An entry type from the future" }, + ]), +); + +writeProject( + "no-origin-fixture", + manifest("no-origin-fixture", "No Origin", {}, [ + { type: "clip", id: "c01", video: "vid1", start: 0, end: 3, cite: 0, section: 0, lock: true, quote: "q" }, + ]), +); + +writeProject( + "localhost-fixture", + manifest("localhost-fixture", "Localhost Origin", { siteOrigin: "http://localhost:3000" }, [ + { type: "clip", id: "c01", video: "vid1", start: 0, end: 3, cite: 0, section: 0, lock: true, quote: "q" }, + ]), +); + +// At the REPORTS_ROOT itself, not under reports/ -- shadowing is about the FIRST +// path segment, because that is the one a static route under app/browse/ wins. +// `reports/find` is perfectly routable; `find` can never be. +writeProject( + "find", + manifest("find", "Shadowed By A Tool Page", { siteOrigin: "https://archive.example" }, [ + { type: "clip", id: "c01", video: "vid1", start: 0, end: 3, cite: 0, section: 0, lock: true, quote: "q" }, + ]), + dest, +); + +writeProject( + path.join("deep", "nested", "solo-fixture"), + manifest("solo-fixture", "Down A Pass-Through Chain", { siteOrigin: "https://archive.example" }, [ + { type: "clip", id: "c01", video: "vid1", start: 0, end: 3, cite: 0, section: 0, lock: true, quote: "q" }, + ]), +); + +// A SECOND copy of the same project, for the specs that WRITE. +// +// The clip bench saves windows and sets locks; the index and decision specs +// assert what report-fixture's windows are. One fixture for both means the +// suite passes or fails depending on which file playwright happened to run +// first -- which it did, once, and the failure named the wrong thing entirely. +// So the read-only assertions get report-fixture and every mutation gets this. +const BENCH = writeProject( + "bench-fixture", + manifest("bench-fixture", "The Bench Fixture", { siteOrigin: "https://archive.example" }, [ + { type: "clip", id: "c01", video: "vid1", start: 3.0, end: 6.0, cite: 3, section: 0, quote: "and because" }, + { type: "clip", id: "c02", video: "vid1", start: 9.0, end: 12.0, cite: 9, section: 0, quote: "another whole sentence" }, + { type: "clip", id: "c03", video: "vid2", start: 1.0, end: 4.0, cite: 1, section: 0, quote: "no punctuation" }, + { type: "clip", id: "c04", video: "vid1", start: 15.0, end: 18.0, cite: 15, section: 0, lockEnd: true, quote: "trailing off" }, + ], + // A three-row ledger, so the claim bench and the two claim decision kinds + // have something exact to assert against: + // + // k01 is fully adjudicated and pinned to a clip -- the settled case. + // k02 is adjudicated but DERIVED, so `not_his_number` must fire on it and + // it must stay out of the stated series. + // k03 carries none of the six fields, so it is the blocking row the inbox + // has to show and the page has to offer controls for. + [ + { id: "k01", date: "2022-01-01", company: "media", value: 4, display: "4", + label: "counts four", quote: "and because", src: "vid1 @ 0:03", + channel: "testchan", video: "vid1", cite: 3, entryId: "c01", + scope: "media", scopeBasis: "names the channel", scopeConfidence: "clear", + population: "employees", valueKind: "uttered", flags: [] }, + { id: "k02", date: "2022-02-01", company: "all", value: 9, display: "9", + label: "4 + 5, summed by us", quote: "another whole sentence", src: "vid1 @ 0:09", + channel: "testchan", video: "vid1", cite: 9, entryId: "c02", + scope: "all", scopeBasis: "no company named", scopeConfidence: "read", + population: "employees", valueKind: "derived", flags: [] }, + { id: "k03", date: "2022-03-01", company: "media", value: 5, display: "5", + label: "nobody has ruled on this one", quote: "no punctuation", src: "vid2 @ 0:01", + channel: "testchan", video: "vid2", cite: 1 }, + ]), +); + +// -- STUB BINARIES, so a build is offline and deterministic -------------------- +// +// The pipeline shells out to yt-dlp for the availability preflight and for every +// fetch. Neither belongs in a test: the first needs the network and the second +// needs somebody else's server to still be serving. YTDLP_BIN and QRENCODE_BIN +// already exist as overrides for exactly this, so the fixture provides both. +// +// They are NODE scripts, not shell. The yt-dlp stub has to parse +// `--download-sections *FROM-TO` and do fractional arithmetic on it, and doing +// that in bash means awk, which means three layers of quoting inside a +// generated file. It got mangled once; this cannot. +// +// The stub gives `source-unavailable` a TRUE answer: any URL naming `gone1` +// fails the way a removed upload does, so a manifest citing it is genuinely +// blocked rather than assumed to be. +const BIN = path.join(dest, "bin"); +mkdirSync(BIN, { recursive: true }); + +writeFileSync( + path.join(BIN, "yt-dlp"), + `#!/usr/bin/env node +// Fixture stub for yt-dlp. Deterministic, offline. +import { spawnSync } from "node:child_process"; +import { mkdirSync } from "node:fs"; +import path from "node:path"; + +const argv = process.argv.slice(2); +const all = argv.join(" "); + +if (all.includes("gone1")) { + process.stderr.write("ERROR: [youtube] gone1: Video unavailable. This video has been removed by the uploader\\n"); + process.exit(1); +} +if (argv.includes("--simulate")) process.exit(0); + +const out = argv[argv.indexOf("-o") + 1]; +if (!out || argv.indexOf("-o") < 0) { + process.stderr.write("stub: no -o\\n"); + process.exit(2); +} +const sec = argv[argv.indexOf("--download-sections") + 1] ?? "*0-5"; +const [from, to] = sec.replace(/^\\*/, "").split("-").map(Number); +const dur = Math.max(1, (to || 5) - (from || 0)); + +mkdirSync(path.dirname(out), { recursive: true }); +const r = spawnSync( + "ffmpeg", + ["-nostdin", "-v", "error", "-y", + "-f", "lavfi", "-i", \`color=c=darkgreen:size=320x180:rate=15:duration=\${dur}\`, + "-f", "lavfi", "-i", \`sine=frequency=440:duration=\${dur}\`, + "-t", String(dur), + "-c:v", "libx264", "-pix_fmt", "yuv420p", "-c:a", "aac", "-ar", "48000", "-ac", "2", out], + { stdio: "inherit" }, +); +process.exit(r.status ?? 1); +`, + { mode: 0o755 }, +); + +writeFileSync( + path.join(BIN, "qrencode"), + `#!/usr/bin/env node +// Fixture stub for qrencode: a real code is not needed to prove one was overlaid. +import { spawnSync } from "node:child_process"; +import { mkdirSync } from "node:fs"; +import path from "node:path"; + +const argv = process.argv.slice(2); +const i = argv.indexOf("-o"); +if (i < 0) process.exit(2); +const out = argv[i + 1]; +mkdirSync(path.dirname(out), { recursive: true }); +const r = spawnSync( + "ffmpeg", + ["-nostdin", "-v", "error", "-y", "-f", "lavfi", "-i", "color=c=white:size=64x64", "-frames:v", "1", out], + { stdio: "inherit" }, +); +process.exit(r.status ?? 1); +`, + { mode: 0o755 }, +); + +// A THIRD copy, for the build specs. +// +// Same reason bench-fixture exists: a build writes out/, stamps deliverables +// aside and is refused when one is newer than its manifest. Sharing a project +// with the bench specs would make each suite's result depend on the other's +// order, which has already cost one confusing red. +const BUILD = writeProject( + "build-fixture", + manifest("build-fixture", "The Build Fixture", { siteOrigin: "https://archive.example" }, [ + { type: "clip", id: "c01", video: "vid1", start: 3.0, end: 6.0, cite: 3, section: 0, lock: true, quote: "and because" }, + { type: "clip", id: "c02", video: "vid1", start: 9.0, end: 12.0, cite: 9, section: 0, lock: true, quote: "another whole sentence" }, + ]), +); + +// A cut whose source is GONE. The preflight must block it, and must block it +// before anything encodes -- which is the whole reason it is step 1 rather than +// a preamble somebody remembers to run. +writeProject( + "gone-fixture", + manifest("gone-fixture", "A Source That Is Gone", { siteOrigin: "https://archive.example" }, [ + { type: "clip", id: "c01", video: "gone1", start: 0, end: 3, cite: 0, section: 0, lock: true, quote: "deleted" }, + ]), +); + +mkdirSync(path.join(reports, "bike-fixture"), { recursive: true }); +writeFileSync( + path.join(reports, "bike-fixture", "sweep-report.md"), + [ + "# The Bike Fixture", + "", + "A cited report that nobody has turned into a video yet.", + "", + '> "the first citation"', + "— [source @ 0:03](https://archive.example/?v=testchan%2Fvid1&t=3)", + "", + '> "the second citation"', + "— [source @ 0:09](https://archive.example/?v=testchan%2Fvid1&t=9)", + "", + ].join("\n"), +); + +// One CACHED SOURCE WINDOW, so the bench and a --skip-fetch build have real +// material without a network. It covers exactly the window c01 would fetch +// (start-3 to end+3 = 0.00-9.00), and it is tone / silence / tone / silence / +// tone with the silences centred on 3.0s and 6.0s -- the two cut points -- so +// snapping has an exact answer instead of a plausible one. +mkdirSync(path.join(REPORT, "out", "clips-raw"), { recursive: true }); +ff([ + "-f", "lavfi", "-i", "color=c=darkgreen:size=320x180:rate=15:duration=9", + "-f", "lavfi", "-i", + "sine=frequency=440:duration=9,volume=enable='between(t,2.9,3.1)+between(t,5.9,6.1)':volume=0", + "-map", "0:v", "-map", "1:a", "-t", "9", + // Past listMedia's 256 KB floor, so the mix picker can see it. See clip(). + ...BULK, "-pix_fmt", "yuv420p", "-c:a", "aac", "-ar", "48000", "-ac", "2", + path.join(REPORT, "out", "clips-raw", "vid1_0.00-9.00.mp4"), +]); +// One report project ships a DELIVERABLE, so "a report video's finished file +// appears in the mix picker" is testable without depending on the build specs +// having run first. The others deliberately have none: a project whose only +// media is out/clips-raw has nothing to offer a picker, because those are +// intermediates and are excluded by name. +mkdirSync(path.join(reports, "no-origin-fixture", "out"), { recursive: true }); + +for (const dir of [BENCH, BUILD]) { + mkdirSync(path.join(dir, "out", "clips-raw"), { recursive: true }); + copyFileSync( + path.join(REPORT, "out", "clips-raw", "vid1_0.00-9.00.mp4"), + path.join(dir, "out", "clips-raw", "vid1_0.00-9.00.mp4"), + ); +} +copyFileSync( + path.join(REPORT, "out", "clips-raw", "vid1_0.00-9.00.mp4"), + path.join(reports, "no-origin-fixture", "out", "no-origin-fixture.mp4"), +); + console.log(`fixture at ${dest}`); if (planned) console.log(` planned clip (used in a build): ${planned}`); console.log(` videos/: alpha (4 cuts, 3 variants), beta (2 cuts), deck (1 cut, 2 variants)`); @@ -600,4 +998,9 @@ console.log(` flagged source: ${flagged ? flagged.video : "none — no asr/"}`) console.log(` SONG_CODE_DIR=${path.join(dest, "code")}`); console.log(` SONG_DIR=${path.join(dest, "data")}`); console.log(` SONG_REPORTS_DIR=${reports}`); +console.log(` YTDLP_BIN=${path.join(BIN, "yt-dlp")} QRENCODE_BIN=${path.join(BIN, "qrencode")}`); +console.log(` CHANNELS_DIR=${CHANNELS} (testchan/vid1 punctuated, vid2 not)`); +console.log(` projects: report-fixture (4 clips, 1 mid-sentence), no-origin-fixture,`); +console.log(` localhost-fixture, bike-fixture (sweep), find/ (shadowed),`); +console.log(` deep/nested/solo-fixture (collapse case), bench-fixture (writable)`); console.log(` ${taken} candidate files copied, 2 mix tracks synthesised`); diff --git a/umtool/e2e/mix.spec.ts b/umtool/e2e/mix.spec.ts @@ -128,3 +128,82 @@ test("choosing a song seeds the handover where the song starts", async ({ page } const handover = page.locator('input[type="number"]').first(); await expect.poll(async () => Number(await handover.inputValue())).toBeGreaterThan(1.9); }); + +// --------------------------------------------------------------------------- +// Reaching the bench from a project. +// +// Six of seven projects used to resolve to null here: MEDIA_ROOTS was the +// um-song subtree, so no report video's media was openable at all. These assert +// that the way in is a link, and that a bad one is REFUSED rather than quietly +// opening something else. +// --------------------------------------------------------------------------- + +test("the picker is grouped by project, and every project's media is in it", async ({ request }) => { + const j = (await (await request.get("/api/mix/files?group=project")).json()) as { + groups: { project: string; label: string; files: { label: string }[] }[]; + other: { label: string }[]; + }; + + // Coverage is a property of the ENUMERATION, not of the cap. A flat + // newest-first list with a 600-entry limit dropped four of six real report + // deliverables off the end, and per-song cuts never appeared at all. + const ids = j.groups.map((g) => g.project); + // A song's cuts, three levels down, which listMedia never reached before. + expect(ids).toContain("reports/videos/alpha"); + // A report video's DELIVERABLE. + expect(ids).toContain("reports/no-origin-fixture"); + + // And the intermediates stay out. out/ holds one deliverable and 40 to 60 + // working files, so reaching one level deeper without excluding these by name + // would put ~260 of them in a picker that is already saturated. A report + // project whose only media is clips-raw correctly offers NOTHING. + const all = [...j.groups.flatMap((g) => g.files), ...j.other]; + expect(all.filter((f) => f.label.includes("segments/"))).toHaveLength(0); + expect(all.filter((f) => f.label.includes("cards/"))).toHaveLength(0); + expect(all.filter((f) => f.label.includes("clips-raw/"))).toHaveLength(0); +}); + +test("a clip row links into the bench with the window already set", async ({ page }) => { + await page.goto("/browse/reports/report-fixture"); + const link = page.locator("[data-mix-link=c01]"); + await expect(link).toBeVisible(); + + const href = (await link.getAttribute("href"))!; + // The window minus the cached file's own start. c01 is 3.00-6.00 and the file + // begins at 0.00, so the arithmetic is visible and exact. + expect(href).toContain("start=3.00"); + expect(href).toContain("end=6.00"); + // ABSOLUTE, like the picker's own <option value>. A relative path is tried + // against each root in order and never stats, so it binds to the first root it + // COULD live under whether or not it is there. + expect(decodeURIComponent(href)).toContain("/out/clips-raw/vid1_0.00-9.00.mp4"); + + await link.click(); + await expect(page).toHaveURL(/\/mix\?/); + await expect(page.locator("[data-mix-refused]")).toHaveCount(0); + // The crumb back to where the link came from. + await expect(page.locator("[data-mix-from]")).toHaveText(/report-fixture/); + // And the fact that the FILE is wider than the clip, so the material outside + // the window does not look unreachable. + await expect(page.locator("[data-mix-preset=c01]")).toContainText("0.00"); +}); + +test("a deep link outside the roots is REFUSED, not clamped to something else", async ({ page }) => { + for (const bad of ["/etc/passwd", "../../../../etc/hosts", "/tmp/nope.mp4"]) { + await page.goto(`/mix?body=${encodeURIComponent(bad)}`); + // Silently opening a DIFFERENT file than the link named is the one outcome + // worse than an error. + await expect(page.locator("[data-mix-refused]")).toBeVisible(); + await expect(page.locator("[data-mix-preset]")).toHaveCount(0); + } +}); + +test("a clip with nothing fetched offers no link at all, rather than a dead one", async ({ + page, +}) => { + await page.goto("/browse/reports/report-fixture"); + // c02's window has no cached file covering it, and a 400 on click would be + // worse than saying so. + await expect(page.locator("[data-mix-link-disabled=c02]")).toBeVisible(); + await expect(page.locator("[data-mix-link=c02]")).toHaveCount(0); +}); diff --git a/umtool/e2e/projects.spec.ts b/umtool/e2e/projects.spec.ts @@ -0,0 +1,495 @@ +import { test, expect } from "@playwright/test"; +import { execFileSync } from "node:child_process"; +import { readFileSync, readdirSync, rmSync, writeFileSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +// --------------------------------------------------------------------------- +// The registry, and the claim that adding a kind is cheap. +// +// Most of this suite is the usual sort of test: render a page, assert what is +// on it. Three of them are not -- they read the repo's own source at test time, +// because the thing being asserted is a PROPERTY OF THE CODE ("a kind id never +// appears outside the registry") rather than of any page. Those are the ones +// that fail when somebody special-cases a kind in a page, which is the exact +// regression this design exists to prevent and the one a page test cannot see. +// +// The fixture (make-fixture.mjs) holds, deliberately: +// report-fixture 4 clips; c01 ends mid-sentence, c04 is lockEnd and must +// stay silent, c03's source has no punctuation at all +// no-origin-fixture no siteOrigin -> blocking +// localhost-fixture localhost origin -> blocking +// bike-fixture a sweep report with no manifest -- the third kind +// find/ shadowed by the /browse/find tool page +// deep/nested/solo a pass-through chain, for the collapse +// --------------------------------------------------------------------------- + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const UMTOOL = path.join(HERE, ".."); + +test("the index lists every kind, and says which state each project is in", async ({ page }) => { + await page.goto("/browse"); + + // The chip's number must EQUAL the number of cards, because the counts come + // from the unfiltered set. Asserting that relationship rather than a magic + // number is what stops every new fixture project from editing this spec. + const cards = await page.locator("[data-kind='report-video']").count(); + expect(cards).toBeGreaterThan(1); + await expect(page.getByRole("link", { name: /^report video \d+$/ })).toHaveText( + `report video ${cards}`, + ); + + // NOT an exact count: browse.spec.ts creates a song through /api/browse/init, + // so the number here depends on what else has run. What matters is that the + // kind is present and that the fixture's own songs are in it. + await expect(page.locator("[data-kind='song']").first()).toBeVisible(); + await expect(page.locator("[data-project='reports/videos/alpha']")).toHaveAttribute( + "data-kind", + "song", + ); + // The third kind, which exists to prove a kind with no decisions, no build + // and no rich read still gets a card, a chip and a page. + await expect(page.locator("[data-kind='sweep-report']")).toHaveCount(1); + + await expect(page.locator("[data-project='reports/report-fixture']")).toHaveAttribute( + "data-state", + // It has a cached source window and no output: not just "windows written". + "fetched", + ); + await expect(page.locator("[data-project='reports/bike-fixture']")).toHaveAttribute( + "data-state", + "draft", + ); +}); + +test("a kind chip filters, and the counts do not move when it does", async ({ page }) => { + await page.goto("/browse"); + const chip = page.getByRole("link", { name: /^report video \d+$/ }); + const before = await chip.textContent(); + + await chip.click(); + await expect(page).toHaveURL(/kind=report-video/); + await expect(page.locator("[data-kind='song']")).toHaveCount(0); + await expect(page.locator("[data-kind='report-video']").first()).toBeVisible(); + + // Counts come from the UNFILTERED set on purpose: a chip whose number changes + // when you click a different chip moves under the cursor. + await expect(page.getByRole("link", { name: /^report video \d+$/ })).toHaveText(before ?? ""); + await expect(page.getByRole("link", { name: /^report video \d+$/ })).toHaveAttribute( + "aria-current", + "true", + ); +}); + +test("?q= survives the filters it was typed under, and is one pasteable URL", async ({ page }) => { + await page.goto("/browse?kind=report-video"); + await page.getByLabel("filter projects").fill("localhost"); + await page.getByRole("button", { name: "filter" }).click(); + + await expect(page).toHaveURL(/kind=report-video/); + await expect(page).toHaveURL(/q=localhost/); + await expect(page.locator("[data-project]")).toHaveCount(1); + await expect(page.locator("[data-project='reports/localhost-fixture']")).toBeVisible(); +}); + +test("the two siteOrigin defects are BLOCKING, and nothing else is", async ({ page }) => { + await page.goto("/browse/decisions?severity=blocking"); + + // These are the two that shipped in real videos: one manifest with no + // siteOrigin (19 QR codes reading `undefined/?v=…`) and one pointing at + // localhost (codes that resolve to nothing on a phone). + const origins = page.locator("[data-decision='manifest-invalid'][data-target='provenance.siteOrigin']"); + await expect(origins).toHaveCount(2); + + await expect( + page.locator("[data-project='reports/no-origin-fixture'] [data-decision='manifest-invalid']"), + ).toContainText("undefined"); + await expect( + page.locator("[data-project='reports/localhost-fixture'] [data-decision='manifest-invalid']"), + ).toContainText("localhost"); +}); + +test("a clip that ends mid-sentence is reported once, and lockEnd acknowledges it", async ({ + page, +}) => { + await page.goto("/browse/decisions?kind=clip-mid-sentence"); + + // Scoped to one project rather than counted globally: the bench specs get + // their own writable copy of this manifest, so a global count would be + // asserting how many fixtures exist rather than what the rule does. + // + // c01 ends inside "And this one runs on and because". c04 ends inside + // "trailing off and then" but sets lockEnd -- which is the author saying "I + // meant to cut here", so it must NOT appear. + const rows = page.locator( + "[data-project='reports/report-fixture'] [data-decision='clip-mid-sentence']", + ); + await expect(rows).toHaveCount(1); + await expect(rows).toHaveAttribute("data-target", "c01"); +}); + +test("an unpunctuated source is said ONCE, not once per clip", async ({ page }) => { + await page.goto("/browse/decisions?kind=no-punctuation"); + // vid2 has no terminator anywhere, and one clip cites it. Six real projects + // produced forty-odd of these rows before they were collapsed, which is an + // inbox whose blocking rows have scrolled off the top -- so the assertion is + // ONE row per project, not one per source. + await expect( + page.locator("[data-project='reports/report-fixture'] [data-decision='no-punctuation']"), + ).toHaveCount(1); +}); + +test("a project named for a tool page is BLOCKING, and the tool page still wins", async ({ + page, +}) => { + await page.goto("/browse"); + const card = page.locator("[data-project='find']"); + await expect(card).toHaveAttribute("data-routing", "shadowed"); + // Its link goes to the escape hatch, not to an address that renders something + // else. Before this it linked to /browse/find and failed silently. + await expect(card).toHaveAttribute("href", /\/browse\/at\?path=find/); + + await page.goto("/browse/find"); + // The phrase console, not the project. + await expect(page.locator("[data-project='find']")).toHaveCount(0); + + await page.goto("/browse/at?path=find"); + // The banner says it and so does the decision row -- both are correct, so the + // assertion takes the first rather than pretending only one exists. + await expect(page.getByText(/is a tool page/).first()).toBeVisible(); + await expect(page.getByText("Shadowed By A Tool Page")).toBeVisible(); +}); + +test("a pass-through folder chain collapses for DISPLAY and never in the URL", async ({ page }) => { + await page.goto("/browse"); + // `deep` holds no projects and one child, so the heading reads as one label. + await expect(page.locator("[data-folder='reports/deep/nested']")).toContainText("deep / nested"); + + // The URL is not collapsed, and every level of it resolves. + await page.goto("/browse/reports/deep/nested/solo-fixture"); + await expect(page.getByText("Down A Pass-Through Chain")).toBeVisible(); + await page.goto("/browse/reports/deep/nested"); + await expect(page.locator("[data-project='reports/deep/nested/solo-fixture']")).toBeVisible(); +}); + +test("the song URLs that already existed still mean the same thing", async ({ page }) => { + // A project id is a path now, but /browse/alpha and /browse/alpha/wide are + // the URLs in every decision href, every spec, and whatever anybody has open. + await page.goto("/browse/alpha"); + await expect(page.locator("[data-cut=wide]")).toBeVisible(); + await page.goto("/browse/alpha/wide"); + await expect(page).toHaveURL(/\/browse\/alpha\/wide/); + // The cut page's own breadcrumb, which is the thing that proves the VIEW + // resolved rather than the project page having been served for both URLs. + await expect(page.getByText(/^every note \(\d+\)$/)).toBeVisible(); + + // And the canonical path works too. + await page.goto("/browse/reports/videos/alpha"); + await expect(page.locator("[data-cut=wide]")).toBeVisible(); +}); + +test("every tool page under /browse still wins its route", async ({ page }) => { + for (const p of ["/browse/decisions", "/browse/find", "/browse/sources", "/browse/faces"]) { + const res = await page.goto(p); + expect(res?.status(), `${p} should still be 200`).toBe(200); + } +}); + +// --------------------------------------------------------------------------- +// The three source-level assertions. +// --------------------------------------------------------------------------- + +test("no page or lib outside the registry special-cases a kind", () => { + const ids = JSON.parse( + execFileSync( + "node", + ["-e", "import('./lib/projects/kinds.mjs').then(m=>console.log(JSON.stringify(m.PROJECT_KINDS.map(k=>k.id))))"], + { cwd: UMTOOL, encoding: "utf8" }, + ).trim(), + ) as string[]; + + // What is being caught is a BRANCH ON A KIND -- `if (p.kind === "song")`, + // `kind: "report-video"`, a lookup keyed by one -- not the mere appearance of + // the word. That distinction has to be drawn: `song` is also a query + // parameter name in eight routes and a directory name in three modules, and a + // bare grep for it reports eleven files that are entirely correct. + // + // So a line is an offender when it carries a kind id AS A STRING and mentions + // `kind` on the same line. It is a heuristic and worth saying so: a + // sufficiently indirect special-case (assigning the id to a const first) would + // slip past. It catches the shape people actually write. + const offenders: string[] = []; + for (const id of ids) { + let out = ""; + try { + out = execFileSync( + "grep", + [ + "-rn", + "--include=*.ts", + "--include=*.tsx", + "--include=*.mjs", + "-e", + `"${id}"`, + "-e", + `'${id}'`, + "app", + "lib", + "components", + ], + { cwd: UMTOOL, encoding: "utf8" }, + ); + } catch { + out = ""; // grep exits 1 when it finds nothing, which is the good case + } + for (const line of out.split("\n").filter(Boolean)) { + const file = line.split(":")[0]; + if (file.startsWith("lib/projects/") || file.startsWith("components/projects/")) continue; + if (!/kind/i.test(line.slice(file.length))) continue; + offenders.push(line); + } + } + + // This is the mechanical form of "adding a kind costs a registry entry and one + // view". An `if (kind === "report-video")` in a page lands here. + expect(offenders, `kind ids leaked outside the registry:\n${offenders.join("\n")}`).toEqual([]); +}); + +test("the reserved names are exactly the static pages under app/browse", () => { + const real = readdirSync(path.join(UMTOOL, "app", "browse"), { withFileTypes: true }) + .filter((e) => e.isDirectory() && !e.name.startsWith("[")) + .map((e) => e.name) + .sort(); + + const declared = JSON.parse( + execFileSync( + "node", + ["-e", "import('./lib/projects/kinds.mjs').then(m=>console.log(JSON.stringify(m.RESERVED_BROWSE)))"], + { cwd: UMTOOL, encoding: "utf8" }, + ).trim(), + ) as string[]; + + // A tenth tool page must not silently make a project unreachable. + expect([...declared].sort()).toEqual(real); +}); + +test("a kind the registry has never seen appears everywhere, with no code edit", () => { + // The extensibility claim, tested rather than asserted. UMTOOL_EXTRA_KINDS is + // read only by kinds.mjs; if a new kind needs an edit anywhere else, this + // fails. + const extra = JSON.stringify([ + { id: "fixture-kind", template: "fixture", label: "fixture kind", badge: "fix", marker: "FIXTURE.marker" }, + ]); + const out = execFileSync( + "node", + [ + "-e", + "import('./lib/projects/kinds.mjs').then(m=>console.log(JSON.stringify({" + + "ids:m.PROJECT_KINDS.map(k=>k.id)," + + "meta:m.KIND_META().map(k=>k.badge)," + + "detected:m.detectKind(new Set(['FIXTURE.marker']))})))", + ], + { cwd: UMTOOL, encoding: "utf8", env: { ...process.env, UMTOOL_EXTRA_KINDS: extra } }, + ); + const j = JSON.parse(out.trim()); + + expect(j.ids).toContain("fixture-kind"); + expect(j.meta).toContain("fix"); + expect(j.detected).toEqual({ kind: "fixture-kind", template: "fixture" }); +}); + +// --------------------------------------------------------------------------- +// The CLI, against the same fixture. +// +// It shares lib/projects/*.mjs with the app, so these are not really testing a +// second implementation -- they are testing that there ISN'T one. +// --------------------------------------------------------------------------- + +// Same path playwright.config.ts builds it at. +const FIXTURE = path.join(UMTOOL, ".e2e-song"); +const cliEnv = { + ...process.env, + SONG_REPORTS_DIR: path.join(FIXTURE, "reports"), + SONG_DIR: path.join(FIXTURE, "data"), + CHANNELS_DIR: path.join(FIXTURE, "channels"), +}; +const umtool = (args: string[]) => + execFileSync("node", ["bin/umtool.mjs", ...args], { cwd: UMTOOL, encoding: "utf8", env: cliEnv }); + +test("umtool ls sees the same projects the index does", () => { + const rows = JSON.parse(umtool(["ls", "--json"])) as { id: string; kind: string }[]; + const ids = rows.map((r) => r.id); + + expect(ids).toContain("reports/report-fixture"); + expect(ids).toContain("reports/bike-fixture"); + // Including the one that has no URL of its own -- it is listed, never dropped. + expect(ids).toContain("find"); + expect(rows.find((r) => r.id === "reports/bike-fixture")?.kind).toBe("sweep-report"); +}); + +test("umtool check exits 1 on a manifest that would ship dead QR codes", () => { + let code = 0; + let stdout = ""; + try { + stdout = umtool(["check", "no-origin-fixture"]); + } catch (e) { + const err = e as { status: number; stdout: string }; + code = err.status; + stdout = err.stdout; + } + // A build script can gate on this, which is the whole reason it exists. + expect(code).toBe(1); + expect(stdout).toContain("BLOCKING"); + expect(stdout).toContain("siteOrigin"); + + // And a manifest with a real origin does not. + const ok = umtool(["check", "report-fixture", "--json"]); + expect(JSON.parse(ok).ok).toBe(true); +}); + +test("umtool show reports the same window facts the page draws", () => { + const j = JSON.parse(umtool(["show", "report-fixture", "--json"])); + const byId = Object.fromEntries(j.entries.map((e: { id: string }) => [e.id, e])); + + // c01 ends inside a run-on cue, and the widener would carry it to the next + // sentence end at 9.0. Knowing that BEFORE running --write is the point. + expect(byId.c01.endsSentence).toBe(false); + expect(byId.c01.proposed).toEqual({ start: 3, end: 9 }); + // c02 ends on a full stop, so there is nothing to propose. + expect(byId.c02.endsSentence).toBe(true); + expect(byId.c02.proposed).toBeNull(); + // c03's source has no punctuation at all: the question cannot be answered, and + // `null` says so rather than a confident `false`. + expect(byId.c03.endsSentence).toBeNull(); + expect(byId.c03.noPunctuation).toBe(true); + + // The cached window is the one the build would fetch, found by containment. + expect(byId.c01.cached.name).toBe("vid1_0.00-9.00.mp4"); + expect(byId.c02.cached).toBeNull(); +}); + +test("umtool refuses a name that means two projects rather than picking one", () => { + // `alpha` is unique here, so it resolves -- the refusal path is exercised by + // asking for something that is not there at all, which must also not guess. + expect(JSON.parse(umtool(["show", "alpha", "--json"])).kind).toBe("song"); + let code = 0; + try { + umtool(["show", "definitely-not-a-project"]); + } catch (e) { + code = (e as { status: number }).status; + } + expect(code).toBe(2); +}); + +// --------------------------------------------------------------------------- +// The index. +// +// Its whole contract is that it changes NOTHING except latency. These assert +// that by breaking it in the two ways it can be broken and checking the pages +// still say the same thing. +// --------------------------------------------------------------------------- + +test("the index is observable, and reports how much of a load it served", async ({ request }) => { + // First load populates it; the second should be served from it. + await request.get("/api/browse/projects"); + const r = await request.get("/api/browse/projects"); + const health = r.headers()["x-index"]; + expect(health).toBeTruthy(); + if (health !== "off") { + const [fresh, total] = health.split(" ")[0].split("/").map(Number); + expect(total).toBeGreaterThan(0); + expect(fresh).toBe(total); + } +}); + +test("deleting the index changes nothing but latency", async ({ request }) => { + const before = (await (await request.get("/api/browse/projects")).json()) as { + projects: { id: string; state: string; title: string }[]; + }; + + // CACHE_DIR is documented as derived output, safe to delete at any time. This + // is that promise, tested. + rmSync(path.join(FIXTURE, "data", ".cache", "umtool", "index"), { + recursive: true, + force: true, + }); + + const after = (await (await request.get("/api/browse/projects")).json()) as typeof before; + expect(after.projects.map((p) => `${p.id}:${p.state}:${p.title}`)).toEqual( + before.projects.map((p) => `${p.id}:${p.state}:${p.title}`), + ); +}); + +test("a project that changed on disk is re-read, not served stale", async ({ request }) => { + const idOf = (j: { projects: { id: string; facts: string[] }[] }, id: string) => + j.projects.find((p) => p.id === id)!; + + const before = (await (await request.get("/api/browse/projects")).json()) as { + projects: { id: string; facts: string[] }[]; + }; + const wasClips = idOf(before, "reports/gone-fixture").facts.find((f) => f.endsWith("clip")); + expect(wasClips).toBe("1 clip"); + + // Add a clip behind the index's back. Signatures are over INPUTS -- the + // manifest's own mtime and size -- so this must invalidate the record. + const file = path.join(FIXTURE, "reports", "gone-fixture", "video.manifest.json"); + const m = JSON.parse(readFileSync(file, "utf8")) as { timeline: unknown[] }; + m.timeline.push({ + type: "clip", id: "c02", video: "gone1", start: 3, end: 6, cite: 3, section: 0, + lock: true, quote: "second", + }); + writeFileSync(file, JSON.stringify(m, null, 2) + "\n"); + + const after = (await (await request.get("/api/browse/projects")).json()) as typeof before; + expect(idOf(after, "reports/gone-fixture").facts).toContain("2 clips"); +}); + +test("umtool new scaffolds a project that check immediately blocks", () => { + const dir = path.join(FIXTURE, "scaffold-root"); + rmSync(dir, { recursive: true, force: true }); + const env = { ...cliEnv, REPORTS_DIR: dir }; + const run = (args: string[]) => + execFileSync("node", ["bin/umtool.mjs", ...args], { cwd: UMTOOL, encoding: "utf8", env }); + + run(["new", "scaffolded", "--json"]); + const m = JSON.parse( + readFileSync(path.join(dir, "scaffolded", "video.manifest.json"), "utf8"), + ) as { timeline: unknown[]; provenance: { siteOrigin: string } }; + + // An EMPTY timeline on purpose. A report records ONE second per citation; a + // window needs a start and an end from the cue file, and generating guesses + // would look finished and be wrong. + expect(m.timeline).toHaveLength(0); + // And an empty siteOrigin, so the field that shipped broken twice cannot be + // left plausible-looking. + expect(m.provenance.siteOrigin).toBe(""); + + let code = 0; + try { + run(["check", "scaffolded"]); + } catch (e) { + code = (e as { status: number }).status; + } + expect(code).toBe(1); + rmSync(dir, { recursive: true, force: true }); +}); + +test("a timeline entry of an unknown type renders, rather than crashing the page", async ({ + page, +}) => { + // The manifest's vocabulary is OPEN. A real one carries `scroll` and `chart` + // beside its cards; code that assumed card-or-clip put `undefined` into + // path.join() and 500'd the project page. The rule is that a CLIP is + // `type === "clip"` and everything else renders generically. + const res = await page.goto("/browse/reports/report-fixture"); + expect(res?.status()).toBe(200); + + await expect(page.locator("[data-entry=z01]")).toHaveAttribute("data-kind", "zz-unknown"); + await expect(page.locator("[data-entry=z01]")).toContainText("An entry type from the future"); + await expect(page.locator("[data-entry=k01]")).toHaveAttribute("data-kind", "card"); + + // And it is COUNTED, not silently dropped: a card saying "4 clips · 1 card" + // about a 6-entry timeline would be lying by omission. + await expect(page.locator("[data-entry]")).toHaveCount(6); + await expect(page.locator("[data-entry=c01]")).toHaveAttribute("data-kind", "clip"); +}); diff --git a/umtool/lib/browse.ts b/umtool/lib/browse.ts @@ -3,6 +3,10 @@ import path from "node:path"; import { SONG_REPORTS, labelFor, resolveInRoots } from "./paths"; import { probeMedia, type MediaInfo } from "./media"; import { readJson } from "./state"; +import { CUT_NAMES as CUT_NAMES_RAW } from "./projects/song.mjs"; +import { songIdsUnder } from "./projects/song-ids.mjs"; +import { REPORTS_ROOT } from "./paths"; +import type { CutName } from "./project-types"; import { acceptedFor, readThumbAccepted, type ThumbDoc } from "./thumbs"; // --------------------------------------------------------------------------- @@ -29,8 +33,11 @@ export const BROWSE_ROOT = path.join(SONG_REPORTS, "videos"); // mortal-kombat and mario-rpg have no vertical.mp4, and a scan-derived list // would render those songs as complete. A hole in the deliverable set is // information -- it is the thing the judging loop is meant to surface. -export const CUT_NAMES = ["wide", "wide-short", "vertical", "vertical-short"] as const; -export type CutName = (typeof CUT_NAMES)[number]; +// +// The list itself moved to lib/projects/song.mjs so `umtool ls` shares it; this +// re-export is what every existing caller still imports. +export const CUT_NAMES = CUT_NAMES_RAW as readonly CutName[]; +export type { CutName }; export function isCutName(v: string): v is CutName { return (CUT_NAMES as readonly string[]).includes(v); @@ -357,19 +364,8 @@ export async function readSong(id: string): Promise<Song | null> { }; } -/** Every song directory under videos/, sorted. One readdir, no stats. */ -export async function songIds(): Promise<string[]> { - let entries; - try { - entries = await readdir(BROWSE_ROOT, { withFileTypes: true }); - } catch { - return []; - } - return entries - .filter((e) => e.isDirectory() && isSegment(e.name)) - .map((e) => e.name) - .sort(); -} +/** Every song, by the basename the song routes are keyed on. See song.mjs. */ +export const songIds = (): Promise<string[]> => songIdsUnder(REPORTS_ROOT, BROWSE_ROOT); /** Every song, newest first. Never probes -- the index must not shell out. */ export async function listSongs(): Promise<SongSummary[]> { diff --git a/umtool/lib/decisions.ts b/umtool/lib/decisions.ts @@ -1,4 +1,4 @@ -import { readSong, resolveRendition, songIds, thumbAliasesFor } from "./browse"; +import { readSong, resolveRendition, thumbAliasesFor } from "./browse"; import { cachedLoudness } from "./loudness"; import { DEFAULT_TARGET, loudnessVerdict } from "./loudness-types"; import { buildStatus, readManifest } from "./manifest"; @@ -27,19 +27,24 @@ import { acceptedFor, readThumbAccepted, readThumbManifest, thumbNamesFor } from // song, a cut or a plan, and that is deliberate. // --------------------------------------------------------------------------- -export type DecisionKind = - | "unjudged-variant" - | "missing-cut" - | "no-recipe" - | "stale-recipe" - | "spec-problem" - | "no-plan" - | "unattributed" - | "thumb-unaccepted" - // Reported from the loudness CACHE only. This reducer must never measure -- - // an inbox that shells out to ffmpeg once per rendition is an inbox that - // takes a minute to open, which is the one thing it cannot afford to be. - | "loudness"; +/** + * A decision kind, as a plain string. + * + * It used to be a closed union of the nine the song reducer emits. It cannot + * stay one: a kind's vocabulary belongs to that kind, and a union here would + * mean every future kind -- report video, supercut, cover set -- editing this + * shared file to say a word only it uses. The real list is assembled from the + * registry as `ALL_DECISION_KINDS` in lib/projects.ts, and an e2e spec asserts + * the ids are unique across kinds. + * + * The song kind's own nine are below, for reference and for the registry entry: + * unjudged-variant, missing-cut, no-recipe, stale-recipe, spec-problem, + * no-plan, unattributed, thumb-unaccepted, and loudness -- which is reported + * from the loudness CACHE only, because this reducer must never measure. An + * inbox that shells out to ffmpeg once per rendition is an inbox that takes a + * minute to open, which is the one thing it cannot afford to be. + */ +export type DecisionKind = string; /** * `blocking` something downstream would LIE if you acted on it -- a spec error @@ -55,7 +60,11 @@ export type Severity = "blocking" | "open" | "info"; export type Decision = { kind: DecisionKind; + /** The project's id -- a POSIX path relative to REPORTS_ROOT. */ project: string; + /** Filled by the dispatcher, so a list can name a project without re-reading it. */ + projectTitle?: string; + projectKind?: string; /** A rel, a cut name, a spec field -- whatever the decision is about. */ target: string; /** One line, already human. */ @@ -276,12 +285,9 @@ export async function decisionsForSong(id: string): Promise<Decision[]> { return sortDecisions(out); } -/** Every open decision, every project, worst first. */ -export async function openDecisions(): Promise<Decision[]> { - const ids = await songIds(); - const per = await Promise.all(ids.map((id) => decisionsForSong(id))); - return sortDecisions(per.flat()); -} +// openDecisions() moved to lib/projects.ts, which is where the enumerator lives +// now: it walks every project of every kind and dispatches to that kind's own +// provider. This file kept the song reducer, which is what it always was. /** The decisions in one song, as the paste already renders everything else. */ export function decisionsMarkdown(items: Decision[]): string[] { diff --git a/umtool/lib/jobs.ts b/umtool/lib/jobs.ts @@ -31,6 +31,12 @@ export type Job = { steps: Step[]; stepIndex: number; log: string[]; + /** NDJSON progress from steps that emit it. See push(). */ + events: Record<string, unknown>[]; + /** Every child pid started, so a stray one can be named after the fact. */ + pids: number[]; + /** Set by cancel(); the running step notices and kills its group. */ + cancelling: boolean; error: string | null; }; @@ -48,14 +54,35 @@ export const recentJobs = (n = 10) => [...jobs.values()].sort((a, b) => b.startedAt - a.startedAt).slice(0, n); /** The accidental-hour-long-job guard. None of the .sh builds fits under any - * cap worth setting, which is the other reason they stay out. */ + * cap worth setting, which is the other reason they stay out. + * + * A step may ask for more (Step.timeoutMs). A 19-clip crossfaded report build + * runs 20 to 40 minutes and would otherwise be SIGKILLed at 15 -- but raising + * this for everything would remove the guard from the jobs that need it. */ const STEP_TIMEOUT_MS = 15 * 60 * 1000; +/** How long a killed process group gets to go quietly before SIGKILL. */ +const KILL_GRACE_MS = 5000; /** Mirrors mix's stderr clamp: enough to diagnose, not enough to blow up RAM. */ const LOG_LIMIT = 400; +/** One build emits a handful of events per clip; this is generous for any of them. */ +const EVENT_LIMIT = 2000; -function push(job: Job, line: string) { +function push(job: Job, line: string, ndjson = false) { for (const l of line.split("\n")) { if (!l.trim()) continue; + // A step declared as NDJSON emits one JSON object per line. They are kept + // separately so the UI can render per-clip state from them, and kept OUT of + // the rolling log so twenty clips of events cannot push the command that + // started the job off the top of it. + if (ndjson && l.startsWith("{")) { + try { + job.events.push(JSON.parse(l)); + if (job.events.length > EVENT_LIMIT) job.events.splice(0, job.events.length - EVENT_LIMIT); + continue; + } catch { + /* not an event after all; fall through and log it */ + } + } job.log.push(l); } if (job.log.length > LOG_LIMIT) job.log.splice(0, job.log.length - LOG_LIMIT); @@ -71,27 +98,79 @@ function runStep(job: Job, step: Step): Promise<void> { cwd: step.cwd, env: { ...process.env, ...step.env }, stdio: ["ignore", "pipe", "pipe"], + // Its own process GROUP, so it can be killed as one. + // + // build-video.mjs shells out to yt-dlp and ffmpeg through execFile, so the + // thing actually burning CPU (or holding a download open) is a GRANDCHILD. + // child.kill() reaps the node process and leaves those running -- the same + // failure the diarize backfill had, where killing the CLI left + // diarize-sherpa.py burning four threads. + detached: true, }); + job.pids.push(child.pid ?? 0); + + const stop = (why: string) => { + push(job, ` ** ${why}`); + killGroup(child.pid); + }; + + const limit = step.timeoutMs ?? STEP_TIMEOUT_MS; + const timer = setTimeout( + () => stop(`killed after ${Math.round(limit / 60000)} minutes`), + limit, + ); + + // A cancel that arrives mid-step is what the abort flag is for; the step + // itself has no other way to hear about it. + const cancelTimer = setInterval(() => { + if (job.cancelling) { + clearInterval(cancelTimer); + stop("cancelled"); + } + }, 250); - const timer = setTimeout(() => { - push(job, ` ** killed after ${STEP_TIMEOUT_MS / 60000} minutes`); - child.kill("SIGKILL"); - }, STEP_TIMEOUT_MS); + const done = () => { + clearTimeout(timer); + clearInterval(cancelTimer); + }; - child.stdout.on("data", (b: Buffer) => push(job, b.toString())); + child.stdout.on("data", (b: Buffer) => push(job, b.toString(), step.ndjson)); child.stderr.on("data", (b: Buffer) => push(job, b.toString())); child.on("error", (e) => { - clearTimeout(timer); + done(); reject(e); }); child.on("close", (code) => { - clearTimeout(timer); - if (code === 0) resolve(); + done(); + if (job.cancelling) reject(new Error("cancelled")); + else if (code === 0) resolve(); else reject(new Error(`${step.argv[0]} exited ${code}`)); }); }); } +/** + * SIGTERM the process group, then SIGKILL what is left. + * + * The negative pid is the whole point: it addresses the GROUP, which is what + * `detached: true` created and what contains the yt-dlp and ffmpeg grandchildren. + */ +function killGroup(pid: number | undefined) { + if (!pid) return; + try { + process.kill(-pid, "SIGTERM"); + } catch { + /* already gone */ + } + setTimeout(() => { + try { + process.kill(-pid, "SIGKILL"); + } catch { + /* already gone */ + } + }, KILL_GRACE_MS); +} + let seq = 0; export function startJob(kind: string, steps: Step[]): Job { @@ -108,6 +187,9 @@ export function startJob(kind: string, steps: Step[]): Job { steps, stepIndex: 0, log: [], + events: [], + pids: [], + cancelling: false, error: null, }; jobs.set(job.id, job); @@ -137,8 +219,24 @@ export function startJob(kind: string, steps: Step[]): Job { return job; } +/** + * Cancel the running job. + * + * Safe to do at any point, and worth saying why: every artefact a report build + * makes is content-addressed -- a fetched window by its window, a segment by its + * clip id -- so re-running skips whatever finished. A cancelled build is a + * paused one. + */ +export function cancelJob(id: string): boolean { + const job = jobs.get(id); + if (!job || job.state !== "running") return false; + job.cancelling = true; + push(job, "** cancel requested"); + return true; +} + /** What the client sees. The steps are included so the command is inspectable. */ -export function jobView(job: Job, since = 0) { +export function jobView(job: Job, since = 0, sinceEvent = 0) { return { id: job.id, kind: job.kind, @@ -150,5 +248,7 @@ export function jobView(job: Job, since = 0) { error: job.error, log: job.log.slice(Math.max(0, since)), next: job.log.length, + events: job.events.slice(Math.max(0, sinceEvent)), + nextEvent: job.events.length, }; } diff --git a/umtool/lib/media.ts b/umtool/lib/media.ts @@ -5,7 +5,7 @@ import { promisify } from "node:util"; import { mkdir, readdir, readFile, rename, stat, writeFile } from "node:fs/promises"; import { existsSync } from "node:fs"; import path from "node:path"; -import { MEDIA_ROOTS, MIX_CACHE, labelFor } from "./paths"; +import { MEDIA_ROOTS, MIX_CACHE, REPORTS_ROOT, SONG_REPORTS, labelFor } from "./paths"; import { brightnessCurve, brightnessSteps } from "../song/flatness.mjs"; const run = promisify(execFile); @@ -69,14 +69,65 @@ export async function probeMedia(abs: string): Promise<MediaInfo> { // `poly-song-<name>.wav` beside it. Unfiltered they outnumbered the actual // renders four to one in the picker, and they are all newest-first, so the // deliverables were pushed off the end of the list. -const SCRATCH_DIR = /^(polytmp-|frames?\d*$|snap\d*$|qrtest$|facedet$|models$|vtest?$|verify$|tism$|fr$|vt$)/; +// The four added names are a report video's INTERMEDIATES. Its out/ directory +// holds one deliverable and 40-60 working files -- the fetched source windows, +// the per-clip segments, the card PNGs, the QR codes. Reaching the deliverable +// means walking one level deeper, and walking one level deeper without these +// would put ~260 intermediates in a picker that is already saturated. +const SCRATCH_DIR = /^(polytmp-|frames?\d*$|snap\d*$|qrtest$|facedet$|models$|vtest?$|verify$|tism$|fr$|vt$|segments$|clips-raw$|cards$|qr$)/; const SCRATCH_FILE = /^(poly-song-|polytmp-|seg_|i_|o_|ms\d?seg|out\.raw)/; /** Below this is a fragment, a probe or a one-note extraction, not a track. */ const MIN_INTERESTING = 256 * 1024; -/** Every media file under the roots, two directories deep, newest first. */ -export async function listMedia(limit = 400): Promise<{ path: string; label: string; size: number; mtimeMs: number }[]> { - const out: { path: string; label: string; size: number; mtimeMs: number }[] = []; +export type MediaRow = { path: string; label: string; size: number; mtimeMs: number }; + +/** + * Every media file under one directory, newest first. + * + * Split out of listMedia() so a PROJECT can be enumerated directly. The picker + * needs that: `listMedia` is newest-first over the whole tree with a cap, and + * the cap is saturated -- measured, four of the six report deliverables fell off + * the end of a 600-entry list. Asking each project for its own files instead + * makes coverage a property of the enumeration rather than of the cap. + */ +export async function listMediaUnder(root: string, maxDepth = 2, limit = 200): Promise<MediaRow[]> { + const out: MediaRow[] = []; + const seen = new Set<string>(); + const walk = async (dir: string, depth: number) => { + let entries; + try { + entries = await readdir(dir, { withFileTypes: true }); + } catch { + return; + } + for (const e of entries) { + if (e.name.startsWith(".")) continue; + const abs = path.join(dir, e.name); + if (e.isDirectory()) { + if (depth > 0 && !SCRATCH_DIR.test(e.name)) await walk(abs, depth - 1); + continue; + } + if (!MEDIA_EXT.has(path.extname(e.name).toLowerCase())) continue; + if (SCRATCH_FILE.test(e.name)) continue; + if (seen.has(abs)) continue; + seen.add(abs); + try { + const st = await stat(abs); + if (st.size < MIN_INTERESTING) continue; + out.push({ path: abs, label: labelFor(abs), size: st.size, mtimeMs: st.mtimeMs }); + } catch { + /* vanished between readdir and stat */ + } + } + }; + await walk(root, maxDepth); + out.sort((a, b) => b.mtimeMs - a.mtimeMs); + return out.slice(0, limit); +} + +/** Every media file under the roots, newest first. */ +export async function listMedia(limit = 400): Promise<MediaRow[]> { + const out: MediaRow[] = []; const seen = new Set<string>(); const walk = async (dir: string, depth: number) => { let entries; @@ -105,7 +156,19 @@ export async function listMedia(limit = 400): Promise<{ path: string; label: str } } }; - for (const r of MEDIA_ROOTS) await walk(r, 1); + + // Depth is PER ROOT, not one number. + // + // The project roots need two levels: a report video's deliverable is at + // <project>/out/<slug>.mp4, and a song's cut is at videos/<song>/<cut>.mp4. + // Both were invisible here before -- six of seven projects had no entry at all. + // + // SONG_DATA and SONG_SCRATCH stay at one. They are the 39 GB corpus and the + // render scratch; a second level there is thousands of stats of clip fragments + // to find nothing anybody would load, which is the opposite of the problem + // this is fixing. + const depthFor = (r: string) => (r === REPORTS_ROOT || r === SONG_REPORTS ? 2 : 1); + for (const r of MEDIA_ROOTS) await walk(r, depthFor(r)); out.sort((a, b) => b.mtimeMs - a.mtimeMs); return out.slice(0, limit); } diff --git a/umtool/lib/mix-preset.ts b/umtool/lib/mix-preset.ts @@ -0,0 +1,91 @@ +import path from "node:path"; +import { labelFor, resolveInRoots } from "./paths"; +import { probeMedia } from "./media"; +import { projectRef } from "./projects"; + +// --------------------------------------------------------------------------- +// A deep link into /mix. +// +// Resolved SERVER-SIDE, in the page, which is why this feature needs no +// useSearchParams and no Suspense boundary -- app/mix/page.tsx is a server +// component that simply did not read searchParams before. +// +// REFUSE, NEVER CLAMP. A body outside the roots produces a bench with no preset +// and a visible reason. Silently opening a DIFFERENT file than the link named is +// the one outcome worse than an error, and lib/mix.ts already takes this line +// for the same reason. +// --------------------------------------------------------------------------- + +export type MixPreset = { + body: string; + label: string; + start: number; + end: number; + duration: number; + hasAudio: boolean; + /** Where the link came from, for the back-crumb. Never used to resolve media. */ + from: string | null; + fromHref: string | null; + clip: string | null; + /** The file's own span, so "there is more material here" is visible. */ + fileFrom: number; + fileTo: number; +}; + +export type MixRefusal = { reason: string }; + +export async function resolvePreset(params: { + body?: string; + start?: string; + end?: string; + from?: string; + clip?: string; +}): Promise<MixPreset | MixRefusal | null> { + if (!params.body) return null; + + const abs = resolveInRoots(params.body); + if (!abs) return { reason: `that file is not inside any known root: ${params.body}` }; + + let info; + try { + info = await probeMedia(abs); + } catch (e) { + return { reason: `could not read it: ${e instanceof Error ? e.message : String(e)}` }; + } + // A body with no audio fails at render time anyway (lib/mix.ts throws), so + // say it now rather than after the numbers have been set. + if (!info.hasAudio) return { reason: `${labelFor(abs)} has no audio track` }; + + const num = (v: string | undefined) => { + const n = Number(v); + return Number.isFinite(n) && n >= 0 ? n : 0; + }; + + // `from` and `clip` are PROVENANCE, for the crumb and the header line. They + // are deliberately not used to resolve the media -- that would be a second, + // divergent resolver for the thing this one already did. + const project = params.from ? await projectRef(params.from) : null; + + // The file's own span, when its name carries one. A raw clip is + // <video>_<from>-<to>.mp4, so the bench can say "this file runs 0–31.4s" + // rather than leaving the material outside the window looking unreachable. + const m = /_(\d+(?:\.\d+)?)-(\d+(?:\.\d+)?)\.mp4$/.exec(path.basename(abs)); + const fileFrom = m ? Number(m[1]) : 0; + + return { + body: abs, + label: labelFor(abs), + start: num(params.start), + end: num(params.end), + duration: info.duration, + hasAudio: info.hasAudio, + from: project?.id ?? null, + fromHref: project ? `/browse/${project.id}` : null, + clip: params.clip ?? null, + fileFrom, + fileTo: fileFrom + info.duration, + }; +} + +export const isRefusal = (v: MixPreset | MixRefusal | null): v is MixRefusal => + !!v && "reason" in v; diff --git a/umtool/lib/mix.ts b/umtool/lib/mix.ts @@ -1,7 +1,7 @@ import { spawn } from "node:child_process"; import { mkdir } from "node:fs/promises"; import path from "node:path"; -import { MEDIA_ROOTS, SONG_REPORTS, labelFor, resolveInRoots } from "./paths"; +import { SONG_REPORTS, WRITE_ROOTS, labelFor, resolveInRoots } from "./paths"; import { readJson, writeJsonAtomic, withStateLock } from "./state"; import { stateFile } from "./paths"; import { probeMedia } from "./media"; @@ -128,8 +128,13 @@ export async function resolveMix(raw: Partial<MixSpec>): Promise<ResolvedMix> { if (!outName) throw new Error("an output name is required"); if (outName.includes("..")) throw new Error("output name may not contain .."); const outAbs = path.isAbsolute(outName) ? outName : path.join(SONG_REPORTS, outName); - const outPath = resolveInRoots(outAbs); - if (!outPath) throw new Error(`output would land outside ${MEDIA_ROOTS.join(", ")}`); + // The WRITE roots, deliberately narrower than the read roots. Report videos + // became readable so their clips could be mixed; their out/ directories hold + // deliverables that cost an hour of network fetches each, and one typo here + // would overwrite one. Refuse rather than clamp: a refused path must never + // silently become a different path. + const outPath = resolveInRoots(outAbs, "write"); + if (!outPath) throw new Error(`output would land outside ${WRITE_ROOTS.join(", ")}`); if (outPath === bodyPath || (bgPath && outPath === bgPath)) { throw new Error("refusing to write the output over one of its own inputs"); } diff --git a/umtool/lib/paths.mjs b/umtool/lib/paths.mjs @@ -0,0 +1,122 @@ +// Where things live, in plain ESM so the CLI and the app share one definition. +// +// This is the same trick song/paths.mjs plays: the values that decide which +// directory is read and which is written must not be able to differ between +// `umtool ls` and the page it is supposed to describe. lib/paths.ts re-exports +// everything here with types; nothing computes a root twice. +import os from "node:os"; +import path from "node:path"; +import { SONG_DATA } from "../song/paths.mjs"; + +export { SONG_DATA }; + +// Derived output (sliced mp3s, waveform peaks, the project index). Lives with +// the data, not in the repo, and is safe to delete at any time. +export const CACHE_DIR = path.join(SONG_DATA, ".cache", "umtool"); + +// Render scratch: the body render and the cut background sit in the job temp +// dir ABOVE SONG_DATA, not inside it, because render-poly.mjs writes them next +// to its logs. +export const SONG_SCRATCH = path.dirname(SONG_DATA); + +/** The um-song deliverables tree. Also BROWSE_ROOT's parent. */ +export const SONG_REPORTS = path.resolve( + process.env.SONG_REPORTS_DIR ?? path.join(os.homedir(), "reports", "quartering-uh-song"), +); + +// --------------------------------------------------------------------------- +// REPORTS_ROOT -- the tree every PROJECT hangs off. +// +// ~/reports holds an um-song project tree AND six report videos AND a report +// with no video yet. Before this, only the um-song subtree was reachable: six +// of seven projects resolved to null in the mix bench, because SONG_REPORTS was +// the widest root there was. +// +// The e2e default is dirname(SONG_REPORTS_DIR) rather than a new env var, so a +// fixture that sets SONG_REPORTS_DIR=<fixture>/reports gets REPORTS_ROOT= +// <fixture> for free and stays confined. The walk skips `data`, which is where +// that fixture symlinks 39 GB of audio. +// --------------------------------------------------------------------------- +export const REPORTS_ROOT = path.resolve( + process.env.REPORTS_DIR ?? + (process.env.SONG_REPORTS_DIR + ? path.dirname(SONG_REPORTS) + : path.join(os.homedir(), "reports")), +); + +const dedupe = (list) => [...new Set(list.map((p) => path.resolve(p)))]; + +// --------------------------------------------------------------------------- +// READ vs WRITE, and why they are two lists. +// +// resolveInRoots() guards both what may be OPENED and what may be RENDERED TO. +// Those were the same list, which meant widening the read root to reach report +// videos would in the same stroke have made every report's out/ a legal render +// target -- a 46 MB deliverable that cost an hour of fetches, one typo from +// being overwritten by /api/mix/render. +// +// So: reports become readable and mixable, and NOTHING new becomes writable. +// A mix of a report clip still lands in SONG_REPORTS, and a hand-typed path +// outside the write set is refused exactly as it was before. +// +// SONG_REPORTS stays FIRST. It is a subdirectory of REPORTS_ROOT, so whichever +// comes first decides every relative label -- and putting REPORTS_ROOT first +// would silently rewrite every existing `videos/<song>/wide.mp4` label into +// `quartering-uh-song/videos/<song>/wide.mp4`. labelFor and resolveInRoots read +// the same ordered list, which is what keeps a label a round-trip. +// --------------------------------------------------------------------------- +export const READ_ROOTS = dedupe( + process.env.MIX_ROOTS + ? process.env.MIX_ROOTS.split(":").filter(Boolean) + : [SONG_REPORTS, REPORTS_ROOT, SONG_DATA, SONG_SCRATCH], +); + +export const WRITE_ROOTS = dedupe( + process.env.MIX_WRITE_ROOTS + ? process.env.MIX_WRITE_ROOTS.split(":").filter(Boolean) + : [SONG_REPORTS, SONG_SCRATCH], +); + +/** Back-compat alias. Every existing caller means "may this be read". */ +export const MEDIA_ROOTS = READ_ROOTS; + +/** Analysis caches for the mix bench (envelopes, brightness curves). */ +export const MIX_CACHE = path.join(CACHE_DIR, "mix"); + +/** Where the project index lives. Under CACHE_DIR, so a fixture gets its own. */ +export const INDEX_DIR = process.env.UMTOOL_INDEX_DIR + ? path.resolve(process.env.UMTOOL_INDEX_DIR) + : path.join(CACHE_DIR, "index"); + +export const inside = (root, abs) => abs === root || abs.startsWith(root + path.sep); + +const rootsFor = (mode) => (mode === "write" ? WRITE_ROOTS : READ_ROOTS); + +/** + * Resolve a client-supplied media path to an absolute one inside a known root, + * or null. Relative paths are tried against each root in order, so the UI can + * pass the short label it displays. + * + * It does not stat, so a relative path binds to the FIRST root it could live + * under whether or not it is there. That is survivable because every path that + * crosses the wire from a picker or a project link is absolute; a relative one + * is a display label being handed back, and those were produced by labelFor + * against this same ordered list. + */ +export function resolveInRoots(p, mode = "read") { + if (!p) return null; + const roots = rootsFor(mode); + const candidates = path.isAbsolute(p) ? [path.resolve(p)] : roots.map((r) => path.resolve(r, p)); + for (const abs of candidates) { + if (roots.some((r) => inside(r, abs))) return abs; + } + return null; +} + +/** The shortest root-relative label for an absolute path, for display. */ +export function labelFor(abs) { + for (const r of READ_ROOTS) { + if (inside(r, abs)) return path.relative(r, abs) || path.basename(abs); + } + return abs; +} diff --git a/umtool/lib/paths.ts b/umtool/lib/paths.ts @@ -1,11 +1,38 @@ -import os from "node:os"; import path from "node:path"; -// Two roots, and keeping them apart is the whole point of the rescue: +// --------------------------------------------------------------------------- +// The roots, and keeping them apart is the whole point of the rescue: // -// SONG_CODE umtool/song -- the tooling and the small JSON state, in the repo -// SONG_DATA wav48/, media/, cand2/, asr/ -- 39 GB, re-derivable, not in the repo +// SONG_CODE umtool/song -- the tooling and the small JSON state, in the repo +// SONG_DATA wav48/, media/, cand2/, asr/ -- 39 GB, re-derivable, not in the repo +// SONG_REPORTS the um-song deliverables -- quartering-*.mp4 and their .plan.json +// SONG_SCRATCH render scratch, ABOVE SONG_DATA +// REPORTS_ROOT the tree every PROJECT hangs off -- songs AND report videos // +// Everything the bench reads or writes must resolve inside one of these. Not +// because this is exposed -- it is a local tool on a loopback port -- but +// because it RENDERS to a path the client names, and a typo that escapes the +// tree would overwrite something that took an hour to make. +// +// The values themselves live in ./paths.mjs, in plain ESM, so `umtool` on the +// command line and this app can never disagree about which directory is which. +// --------------------------------------------------------------------------- +export { + CACHE_DIR, + INDEX_DIR, + MEDIA_ROOTS, + MIX_CACHE, + READ_ROOTS, + REPORTS_ROOT, + SONG_DATA, + SONG_REPORTS, + SONG_SCRATCH, + WRITE_ROOTS, + inside, + labelFor, + resolveInRoots, +} from "./paths.mjs"; + // Resolved from cwd rather than import.meta.url: Turbopack rewrites module URLs // into .next/server/chunks, so a path derived from one points at the build // output instead of the source tree. `next dev` and `next start` both run with @@ -14,74 +41,9 @@ export const SONG_CODE = process.env.SONG_CODE_DIR ? path.resolve(process.env.SONG_CODE_DIR) : path.join(process.cwd(), "song"); -// song/paths.mjs owns the SONG_DATA default so the CLI scripts and this app can -// never disagree about where the audio is. Same env var, same fallback. -export const SONG_DATA = path.resolve( - process.env.SONG_DIR ?? "/home/user/.claude/jobs/efbe67a7/tmp/song", -); - -// Derived output (sliced mp3s, waveform peaks). Lives with the data, not in the -// repo, and is safe to delete at any time. -export const CACHE_DIR = path.join(SONG_DATA, ".cache", "umtool"); - export const stateFile = (name: string) => path.join(SONG_CODE, name); -export const dataFile = (...parts: string[]) => path.join(SONG_DATA, ...parts); - -// --------------------------------------------------------------------------- -// Where finished and in-progress RENDERS live, for the mix bench. -// -// Three roots, and they are genuinely three different things: -// -// SONG_REPORTS the deliverables -- quartering-*.mp4 and their .plan.json -// SONG_DATA the bulk corpus (wav48/, media/, cand2/) -// SONG_SCRATCH render scratch: the body render and the cut background sit -// in the job temp dir ABOVE SONG_DATA, not inside it, because -// render-poly.mjs writes them next to its logs -// -// Everything the bench reads or writes must resolve inside one of these. Not -// because this is exposed -- it is a local tool on a loopback port -- but -// because it RENDERS to a path the client names, and a typo that escapes the -// tree would overwrite something that took an hour to make. -// --------------------------------------------------------------------------- - -export const SONG_SCRATCH = path.dirname(SONG_DATA); - -export const SONG_REPORTS = path.resolve( - process.env.SONG_REPORTS_DIR ?? path.join(os.homedir(), "reports", "quartering-uh-song"), -); - -export const MEDIA_ROOTS: string[] = [ - ...new Set( - (process.env.MIX_ROOTS - ? process.env.MIX_ROOTS.split(":").filter(Boolean) - : [SONG_REPORTS, SONG_DATA, SONG_SCRATCH] - ).map((p) => path.resolve(p)), - ), -]; - -/** Analysis caches for the mix bench (envelopes, brightness curves). */ -export const MIX_CACHE = path.join(CACHE_DIR, "mix"); - -const inside = (root: string, abs: string) => abs === root || abs.startsWith(root + path.sep); - -/** - * Resolve a client-supplied media path to an absolute one inside a known root, - * or null. Relative paths are tried against each root in order, so the UI can - * pass the short label it displays. - */ -export function resolveInRoots(p: string | null | undefined): string | null { - if (!p) return null; - const candidates = path.isAbsolute(p) ? [path.resolve(p)] : MEDIA_ROOTS.map((r) => path.resolve(r, p)); - for (const abs of candidates) { - if (MEDIA_ROOTS.some((r) => inside(r, abs))) return abs; - } - return null; -} -/** The shortest root-relative label for an absolute path, for display. */ -export function labelFor(abs: string): string { - for (const r of MEDIA_ROOTS) { - if (inside(r, abs)) return path.relative(r, abs) || path.basename(abs); - } - return abs; -} +// dataFile needs SONG_DATA at module scope, which the re-export above does not +// bind locally -- so it is imported again rather than duplicated. +import { SONG_DATA as DATA } from "./paths.mjs"; +export const dataFile = (...parts: string[]) => path.join(DATA, ...parts); diff --git a/umtool/lib/project-types.ts b/umtool/lib/project-types.ts @@ -0,0 +1,130 @@ +// Types only. No `node:` import, ever. +// +// This file exists because of a failure this repo has already had: a value +// import that dragged `node:fs` into a client component passed `tsc --noEmit` +// and then 500'd every page. The registry is exactly that hazard -- a client +// component wants the kind labels for its chips, and the registry also reads +// the disk. So the shapes live here, the reading lives in lib/projects/*.mjs, +// and a component that only needs a label never touches the latter. + +/** A registry id. Deliberately `string`: kinds are data, not a closed union. */ +export type ProjectKindId = string; + +/** + * The song kind's four deliverables. + * + * The VALUE lives in lib/projects/song.mjs, because `umtool ls` needs it from a + * terminal; the TYPE lives here, because a .mjs export infers as string[] and + * every existing caller wants the literal union. One list, two shapes -- which + * is the only place in the registry that duplication is unavoidable. + */ +export type CutName = "wide" | "wide-short" | "vertical" | "vertical-short"; + +export type ProjectStage = { + id: string; + label: string; +}; + +/** + * The shared state vocabulary, across every kind. + * + * One vocabulary rather than per-kind words, because the whole point of the + * filter is to ask "what is half-done" without first asking "half-done at + * what". A kind maps its own situation onto these; it does not invent a sixth. + */ +export type ProjectState = "draft" | "windows" | "fetched" | "built" | "shipped" | "stale"; + +export const PROJECT_STATES: ProjectState[] = [ + "draft", + "windows", + "fetched", + "built", + "shipped", + "stale", +]; + +export const STATE_LABEL: Record<ProjectState, string> = { + draft: "draft", + windows: "windows", + fetched: "fetched", + built: "built", + shipped: "shipped", + stale: "stale", +}; + +/** How a project is reachable, and whether it is reachable at all. */ +export type Routing = "ok" | "shadowed" | "unroutable"; + +export type ProjectRef = { + /** + * The POSIX path relative to REPORTS_ROOT -- `ferret-rescue`, or + * `quartering-uh-song/videos/yoshi`. NOT the basename: bare names collide + * across folders (a second `pokemon` is a matter of time) and the path is + * what makes a link stable. + */ + id: string; + /** Absolute. */ + dir: string; + /** The last segment, for display. */ + name: string; + /** The POSIX path of the containing folder, or "" at the root. */ + folder: string; + kind: ProjectKindId; + template: string; + routing: Routing; + /** Set when two kinds matched. Never resolved by picking one. */ + ambiguousWith?: ProjectKindId[]; +}; + +export type ProjectSummary = ProjectRef & { + /** The kind's short badge, copied in so a card needs no registry lookup. */ + badge: string; + title: string; + /** One line under the title. Kind-specific. */ + subtitle: string | null; + state: ProjectState; + /** Newest mtime of anything the summary looked at. Drives `sort=recent`. */ + newestMtimeMs: number; + /** Small facts for the card, already rendered as text. */ + facts: string[]; + /** Loud, short, and only when something is wrong. */ + flags: string[]; + /** Project-relative path of a poster candidate, or null. */ + posterRel: string | null; + /** Everything a `?q=` substring match should see. */ + haystack: string; + /** + * `data-*` attributes the kind wants on its card. + * + * The grid renders strings and knows no kinds, so this is how a kind keeps an + * assertion surface of its own -- a song's missing cut list, a report's clip + * count -- without the grid growing a branch per kind. A key whose value is + * absent is OMITTED, so "nothing is missing" is the attribute not being there + * rather than an empty string. + */ + attrs?: Record<string, string>; +}; + +export type FolderNode = { + /** POSIX path relative to REPORTS_ROOT. "" is the root itself. */ + path: string; + /** + * The display label. A pass-through chain (`quartering-uh-song/videos`) is + * collapsed to one label; the URL is never collapsed. + */ + label: string; + /** The segments folded into `label`, oldest first. Display only. */ + collapsedFrom: string[]; + projects: string[]; + children: string[]; +}; + +export type ProjectKindMeta = { + id: ProjectKindId; + template: string; + label: string; + /** Two to six characters. Goes on the card. */ + badge: string; + stages: ProjectStage[]; + decisionKinds: string[]; +}; diff --git a/umtool/lib/projects.ts b/umtool/lib/projects.ts @@ -0,0 +1,436 @@ +import path from "node:path"; +import { REPORTS_ROOT } from "./paths"; +import { KIND_META, PROJECT_KINDS, kindById } from "./projects/kinds.mjs"; +import { collapseFolders, foldersFor, walkProjects } from "./projects/walk.mjs"; +import { SONG_KIND } from "./projects/song.mjs"; +import { openIndex, signRecord } from "./projects/index-db.mjs"; +import { BROWSE_ROOT } from "./browse"; +import { decisionsForSong } from "./decisions"; +import { listMedia, listMediaUnder, type MediaRow } from "./media"; +import type { Decision } from "./decisions"; +import type { + FolderNode, + ProjectKindMeta, + ProjectRef, + ProjectState, + ProjectSummary, +} from "./project-types"; + +export type { FolderNode, ProjectKindMeta, ProjectRef, ProjectState, ProjectSummary }; +export { PROJECT_STATES, STATE_LABEL } from "./project-types"; + +// --------------------------------------------------------------------------- +// The app's view of the registry. +// +// SERVER ONLY. lib/projects/kinds.mjs reaches the disk through its per-kind +// modules, so importing it from a client component drags `node:fs` into the +// browser bundle -- which this repo has already been bitten by once: it passed +// `tsc --noEmit` and then 500'd every page. A client component that wants kind +// labels or state names imports lib/project-types.ts and is handed the rest as +// props. `pnpm build`, not typecheck, is what catches a regression here. +// --------------------------------------------------------------------------- + +export const KINDS: ProjectKindMeta[] = KIND_META(); + +/** + * Which kinds emit which decisions, and the union of all of them. + * + * Assembled rather than hand-written, so a kind's vocabulary lives with the + * kind. A closed union in lib/decisions.ts would put every kind's words in one + * shared file -- exactly the coupling the registry exists to remove. + */ +/** Emitted by the walk rather than by any kind: how a project is reachable. */ +export const ROUTING_DECISION_KINDS = ["shadowed-name", "unroutable-name", "ambiguous-project"]; + +export const ALL_DECISION_KINDS: string[] = [ + ...new Set([...KINDS.flatMap((k) => k.decisionKinds), ...ROUTING_DECISION_KINDS]), +].sort(); + +// --------------------------------------------------------------------------- +// Caching. +// +// The WALK is memoised for a second: it is 25 ms on the real tree, and a page +// render asks for it two or three times. +// +// A SUMMARY is memoised against its own kind's signature -- a tuple of mtimes +// and sizes, never the bytes -- so it survives for as long as the project has +// not changed and is thrown away the moment it has. That is the same rule the +// export build's incremental signatures follow, and it is the reason no index +// is needed yet at twelve projects. +// --------------------------------------------------------------------------- +let walkCache: { at: number; refs: ProjectRef[] } | null = null; +const summaryCache = new Map<string, { sig: string; value: ProjectSummary }>(); +const decisionCache = new Map<string, { sig: string; value: Decision[] }>(); + +// --------------------------------------------------------------------------- +// The persistent index, opened once and held. +// +// FS FIRST, INDEX AFTER, BEST EFFORT -- in a swallowed try/catch, always. A +// crash between the two leaves a signature that no longer matches, which the +// next read repairs. Index-FIRST could claim something the filesystem does not +// say, and that is the one failure this refuses. +// +// It is also entirely optional: openIndex() hands back a no-op when the native +// module or the store is missing, and every read verifies a signature anyway. +// Deleting the .mdb changes nothing but latency. +// --------------------------------------------------------------------------- +type IndexHandle = Awaited<ReturnType<typeof openIndex>>; +let indexPromise: Promise<IndexHandle> | null = null; +const indexHandle = (): Promise<IndexHandle> => (indexPromise ??= openIndex()); + +/** fresh / total over the last listProjects(), for the footer note and x-index. */ +let lastIndexHits = { fresh: 0, total: 0, ok: false }; +export const indexHealth = () => ({ ...lastIndexHits }); + +/** Drop every cache. The CLI's `scan` and the e2e suite want this. */ +export function invalidateProjects(): void { + walkCache = null; + summaryCache.clear(); + decisionCache.clear(); +} + +export async function projectRefs(): Promise<ProjectRef[]> { + if (walkCache && Date.now() - walkCache.at < 1000) return walkCache.refs; + const refs = (await walkProjects(REPORTS_ROOT)) as ProjectRef[]; + walkCache = { at: Date.now(), refs }; + return refs; +} + +/** Drop index records for projects the walk no longer finds. */ +export async function pruneIndex(): Promise<number> { + const ix = await indexHandle(); + if (!ix.ok) return 0; + const live = new Set((await projectRefs()).map((p) => p.id)); + let dropped = 0; + for (const rec of ix.recent(10_000) as { id: string }[]) { + if (live.has(rec.id)) continue; + ix.del(rec.id); + dropped += 1; + } + return dropped; +} + +export async function projectRef(id: string): Promise<ProjectRef | null> { + return (await projectRefs()).find((p) => p.id === id) ?? null; +} + +type KindSummary = { + title?: string; + subtitle?: string | null; + state?: string; + newestMtimeMs?: number; + facts?: string[]; + flags?: string[]; + posterRel?: string | null; + haystack?: string; + attrs?: Record<string, string>; +}; + +type Ctx = ProjectRef & { root: string }; +const ctxFor = (p: ProjectRef): Ctx => ({ ...p, root: REPORTS_ROOT }); + +async function signatureOf(p: ProjectRef): Promise<string> { + const k = kindById(p.kind); + if (!k?.signature) return "0"; + try { + return String(await k.signature(p.dir)); + } catch { + return "0"; + } +} + +/** The card for one project. Never probes; never shells out. */ +export async function summariseProject(p: ProjectRef): Promise<ProjectSummary> { + const kindSig = await signatureOf(p); + const sig = signRecord({ kindSig, dirMs: 0, markerMs: 0, markerSize: 0, outMs: 0 }); + + const hit = summaryCache.get(p.id); + if (hit && hit.sig === sig) return hit.value; + + // The persistent one. Verified, never trusted: a record whose signature no + // longer matches the disk is discarded, not migrated. + const ix = await indexHandle(); + lastIndexHits.ok = ix.ok; + lastIndexHits.total += 1; + const cached = ix.get(p.id) as (ProjectSummary & { sig?: string }) | null; + if (cached && cached.sig === sig && cached.kind === p.kind && cached.routing === p.routing) { + lastIndexHits.fresh += 1; + summaryCache.set(p.id, { sig, value: cached }); + return cached; + } + + const k = kindById(p.kind); + // The boundary between a .mjs summariser and a typed summary. The modules + // return more than the card needs (a song's verdict tally, a report's parsed + // manifest) and their `state` infers as string, so it is narrowed once here + // rather than asserted at every read. + let body: KindSummary = {}; + if (k?.summarise) { + try { + body = await k.summarise(ctxFor(p)); + } catch { + // A project that cannot be read is still a project. Saying so beats + // dropping it from a list whose whole job is to be complete. + body = { flags: ["could not be read"] }; + } + } + + const flags = [...(body.flags ?? [])]; + // Routing failures are kind-independent, so they are added here rather than + // in any kind's summariser. + if (p.routing === "shadowed") flags.push(`/browse/${p.id.split("/")[0]} is a tool page`); + if (p.routing === "unroutable") flags.push("name will not route"); + if (p.ambiguousWith) flags.push(`two kinds match: ${p.ambiguousWith.join(", ")}`); + + const value: ProjectSummary = { + ...p, + badge: k?.badge ?? p.kind, + title: body.title ?? p.name, + subtitle: body.subtitle ?? null, + state: (body.state as ProjectState) ?? "draft", + newestMtimeMs: body.newestMtimeMs ?? 0, + facts: body.facts ?? [], + flags, + posterRel: body.posterRel ?? null, + haystack: `${body.haystack ?? ""} ${p.id} ${p.kind} ${p.template}`.toLowerCase(), + attrs: body.attrs ?? {}, + }; + summaryCache.set(p.id, { sig, value }); + // After the read, never before, and its failure is a cache miss next time. + try { + ix.put({ ...value, sig }); + } catch { + /* the filesystem is the model */ + } + return value; +} + +/** Every project, newest first. */ +export async function listProjects(): Promise<ProjectSummary[]> { + lastIndexHits = { fresh: 0, total: 0, ok: false }; + const refs = await projectRefs(); + const out = await Promise.all(refs.map(summariseProject)); + return out.sort((a, b) => b.newestMtimeMs - a.newestMtimeMs || a.id.localeCompare(b.id)); +} + +/** The folder tree, collapsed for display. URLs are never collapsed. */ +export async function listFolders(): Promise<Map<string, FolderNode>> { + return collapseFolders(foldersFor(await projectRefs())) as Map<string, FolderNode>; +} + +// --------------------------------------------------------------------------- +// Decisions. +// +// The dispatch is here rather than in lib/decisions.ts because the song reducer +// IS lib/decisions.ts -- it reads verdicts, plans, recipes, loudness and the +// accepted cover set, all of which live in TypeScript beside readSong(). Naming +// it from this side keeps the import graph acyclic and keeps the one kind-id +// literal the app needs inside the registry's own module. +// --------------------------------------------------------------------------- +export async function decisionsForProject(p: ProjectRef): Promise<Decision[]> { + const sig = await signatureOf(p); + const hit = decisionCache.get(p.id); + if (hit && hit.sig === sig) return hit.value; + + let out: Decision[] = []; + try { + if (p.kind === SONG_KIND) { + // Keyed by BASENAME while the song routes still are -- and the test for + // "is this song reachable by basename" has to be the same one songIds() + // uses, which is BROWSE_ROOT. Writing the production path out by hand here + // meant the fixture (whose songs live elsewhere) silently produced no song + // decisions at all, and the inbox looked clean because it was empty. + out = p.dir === path.join(BROWSE_ROOT, p.name) ? await decisionsForSong(p.name) : []; + } else { + const k = kindById(p.kind); + if (k?.decisions) { + const summary = k.summarise ? await k.summarise(ctxFor(p)) : null; + out = (await k.decisions(ctxFor(p), summary)) as Decision[]; + } + } + } catch { + out = []; + } + + // Routing failures are kind-independent, so they are added here rather than + // asked of every kind. Both of these used to be SILENT: a shadowed project was + // listed with a link that rendered a tool page, and an unroutable one simply + // did not appear. A thing that cannot be opened has to say so. + if (p.routing === "shadowed") { + out.push({ + kind: "shadowed-name", + project: p.id, + target: p.id.split("/")[0], + why: `/browse/${p.id.split("/")[0]} is a tool page and always wins the route — rename the directory, or open it from here`, + href: `/browse/at?path=${encodeURIComponent(p.id)}`, + severity: "blocking", + at: Date.now(), + }); + } + if (p.routing === "unroutable") { + out.push({ + kind: "unroutable-name", + project: p.id, + target: p.name, + why: "this name cannot be a URL segment, so the project has no address of its own", + href: `/browse/at?path=${encodeURIComponent(p.id)}`, + severity: "info", + at: Date.now(), + }); + } + if (p.ambiguousWith) { + out.push({ + kind: "ambiguous-project", + project: p.id, + target: p.ambiguousWith.join(" + "), + why: "two kinds match this directory — it is read as the first, which is a bug rather than a choice", + href: `/browse/${p.id}`, + severity: "blocking", + at: Date.now(), + }); + } + + const title = (await summariseProject(p)).title; + const value = out.map((d) => ({ ...d, project: p.id, projectTitle: title, projectKind: p.kind })); + decisionCache.set(p.id, { sig, value }); + return value; +} + +const RANK: Record<string, number> = { blocking: 0, open: 1, info: 2 }; + +/** Every open decision, every project, worst first. */ +export async function openDecisions(): Promise<Decision[]> { + const refs = await projectRefs(); + const per = await Promise.all(refs.map(decisionsForProject)); + return per + .flat() + .sort((a, b) => RANK[a.severity] - RANK[b.severity] || b.at - a.at); +} + +/** blocking + open counts per project id, for the index's chips. */ +export async function decisionCounts(): Promise<Map<string, { blocking: number; open: number }>> { + const refs = await projectRefs(); + const out = new Map<string, { blocking: number; open: number }>(); + await Promise.all( + refs.map(async (p) => { + const ds = await decisionsForProject(p); + out.set(p.id, { + blocking: ds.filter((d) => d.severity === "blocking").length, + open: ds.filter((d) => d.severity === "open").length, + }); + }), + ); + return out; +} + +// --------------------------------------------------------------------------- +// Routing. +// --------------------------------------------------------------------------- + +export type Resolved = + | { project: ProjectRef; rest: string[]; folder: null; via?: "alias" } + | { project: null; rest: []; folder: FolderNode } + | { project: null; rest: []; folder: null; ambiguous: ProjectRef[] } + | null; + +/** + * Resolve a `/browse/...` path to the LONGEST prefix that is a project, and + * hand the remaining segments to that kind's view. + * + * Longest-prefix rather than "first project found" because a project id is a + * path, and `a/b` being a project must not stop `a/b/c` from being one too. + */ +export async function resolveProjectPath(segments: string[]): Promise<Resolved> { + const refs = await projectRefs(); + const id = segments.join("/"); + + let best: ProjectRef | null = null; + for (const p of refs) { + if (p.routing !== "ok") continue; + if (id === p.id || id.startsWith(`${p.id}/`)) { + if (!best || p.id.length > best.id.length) best = p; + } + } + if (best) { + const rest = id === best.id ? [] : id.slice(best.id.length + 1).split("/"); + return { project: best, rest, folder: null }; + } + + const folders = await listFolders(); + const folder = folders.get(id); + if (folder) return { project: null, rest: [], folder }; + + // ---- the basename alias ------------------------------------------------- + // + // A project id is a PATH, so a song's canonical URL is now + // /browse/quartering-uh-song/videos/yoshi. But /browse/yoshi and + // /browse/yoshi/wide are the URLs that exist -- in every decision href, in + // the e2e suite, and in whatever anybody has open in a tab. A URL has to mean + // the same thing in six weeks, so the bare name keeps working as an alias. + // + // ONLY when it is unique. Two projects sharing a basename is precisely why an + // id is a path in the first place, and picking one of them would be the guess + // this whole design refuses to make -- so it reports the collision instead. + const [first, ...restSegs] = segments; + if (first) { + const named = refs.filter((p) => p.name === first && p.routing === "ok"); + if (named.length === 1) { + return { project: named[0], rest: restSegs, folder: null, via: "alias" }; + } + if (named.length > 1) { + return { project: null, rest: [], folder: null, ambiguous: named }; + } + } + return null; +} + +/** Kind metadata by id, for a page that has a summary and wants its label. */ +export const kindMetaOf = (id: string): ProjectKindMeta | null => + KINDS.find((k) => k.id === id) ?? null; + +export { PROJECT_KINDS }; + + +// --------------------------------------------------------------------------- +// Media, grouped by the project that owns it. +// +// The mix picker was a single newest-first list with a cap, and the cap was +// saturated: measured against the real tree, four of the six report deliverables +// fell off the end of a 600-entry list, and per-song cuts never appeared at all. +// Raising the cap only moves the cliff. +// +// So coverage becomes a property of the enumeration. Each project is asked for +// its own files -- with its own small cap, so one busy project cannot push every +// other project off the end -- and whatever is left over (the corpus, the render +// scratch) becomes one final group. `<optgroup>` renders it with zero client JS +// and the control stays a native select. +// --------------------------------------------------------------------------- + +export type MediaGroup = { + project: string; + label: string; + kind: string; + files: MediaRow[]; +}; + +const PER_PROJECT = 40; + +export async function mediaGroups(): Promise<{ groups: MediaGroup[]; other: MediaRow[] }> { + const refs = await projectRefs(); + const groups: MediaGroup[] = []; + for (const p of refs) { + const files = await listMediaUnder(p.dir, 2, PER_PROJECT); + if (!files.length) continue; + const summary = await summariseProject(p); + groups.push({ project: p.id, label: `${p.id} (${summary.badge})`, kind: p.kind, files }); + } + groups.sort((a, b) => a.project.localeCompare(b.project)); + + // Longest-prefix ownership, so a file inside a project never also appears in + // `other` -- and so nothing needs a second walk to work out who owns what. + const dirs = refs.map((p) => p.dir + path.sep); + const owned = (abs: string) => dirs.some((d) => abs.startsWith(d)); + const other = (await listMedia(600)).filter((f) => !owned(f.path)); + + return { groups, other }; +} diff --git a/umtool/lib/projects/core.mjs b/umtool/lib/projects/core.mjs @@ -0,0 +1,137 @@ +// The registry, assembled — for callers with no server. +// +// lib/projects.ts is the app's façade: it memoises, it dispatches song +// decisions into the TypeScript reducer, and it is full of `import type`. This +// is the same thing for `umtool` on the command line, in plain ESM, sharing the +// walk, the kinds and every per-kind module with the app so the two cannot +// disagree about what a project is or what is wrong with it. +// +// WHAT IT DOES NOT DO: the song reducer. Nine of the decision kinds +// (unjudged-variant, stale-recipe, loudness, …) are computed by lib/decisions.ts +// against readSong(), readSpec(), the loudness cache and the accepted cover set, +// all of which are TypeScript beside the app. Porting them would be a second +// implementation of the thing this file exists to avoid having two of. So the +// CLI reports every report-video decision and every routing one — which is what +// `umtool check` is for, since the two defects that shipped in real videos were +// both manifest problems — and says plainly that song decisions live in the UI. +import path from "node:path"; +import { REPORTS_ROOT } from "../paths.mjs"; +import { PROJECT_KINDS, kindById } from "./kinds.mjs"; +import { collapseFolders, foldersFor, walkProjects } from "./walk.mjs"; + +export { PROJECT_KINDS, REPORTS_ROOT }; + +export const ROUTING_DECISION_KINDS = ["shadowed-name", "unroutable-name", "ambiguous-project"]; + +export async function projectRefs(root = REPORTS_ROOT) { + return walkProjects(root); +} + +export async function summarise(p, root = REPORTS_ROOT) { + const k = kindById(p.kind); + const ctx = { ...p, root }; + let body = {}; + if (k?.summarise) { + try { + body = await k.summarise(ctx); + } catch (err) { + body = { flags: [`could not be read: ${err?.message ?? err}`] }; + } + } + const flags = [...(body.flags ?? [])]; + if (p.routing === "shadowed") flags.push(`/${p.id.split("/")[0]} is a tool page`); + if (p.routing === "unroutable") flags.push("name will not route"); + if (p.ambiguousWith) flags.push(`two kinds match: ${p.ambiguousWith.join(", ")}`); + return { + ...p, + badge: k?.badge ?? p.kind, + title: body.title ?? p.name, + subtitle: body.subtitle ?? null, + state: body.state ?? "draft", + newestMtimeMs: body.newestMtimeMs ?? 0, + facts: body.facts ?? [], + flags, + attrs: body.attrs ?? {}, + haystack: `${body.haystack ?? ""} ${p.id} ${p.kind} ${p.template}`.toLowerCase(), + }; +} + +/** Every decision this side can compute for one project. */ +export async function decisionsFor(p, root = REPORTS_ROOT) { + const out = []; + const k = kindById(p.kind); + const ctx = { ...p, root }; + if (k?.decisions) { + try { + const s = k.summarise ? await k.summarise(ctx) : null; + out.push(...(await k.decisions(ctx, s))); + } catch (err) { + out.push({ + kind: "unreadable", + project: p.id, + target: p.id, + why: `could not be read: ${err?.message ?? err}`, + severity: "blocking", + at: 0, + }); + } + } + if (p.routing === "shadowed") { + out.push({ + kind: "shadowed-name", + project: p.id, + target: p.id.split("/")[0], + why: `/browse/${p.id.split("/")[0]} is a tool page and always wins the route`, + severity: "blocking", + at: 0, + }); + } + if (p.routing === "unroutable") { + out.push({ + kind: "unroutable-name", + project: p.id, + target: p.name, + why: "this name cannot be a URL segment, so the project has no address of its own", + severity: "info", + at: 0, + }); + } + if (p.ambiguousWith) { + out.push({ + kind: "ambiguous-project", + project: p.id, + target: p.ambiguousWith.join(" + "), + why: "two kinds match this directory — it is read as the first, which is a bug", + severity: "blocking", + at: 0, + }); + } + return out.map((d) => ({ ...d, project: p.id })); +} + +/** Whether this kind's decisions are fully computable without the app. */ +export const decisionsAreComplete = (kind) => !!kindById(kind)?.decisions; + +/** + * Resolve a user-typed project argument. + * + * An exact id wins; otherwise a UNIQUE basename does. Two projects answering to + * one name is the reason ids are paths, so it is reported rather than resolved. + */ +export async function resolveProject(arg, root = REPORTS_ROOT) { + const refs = await projectRefs(root); + const exact = refs.find((p) => p.id === arg); + if (exact) return { project: exact }; + // A path the user typed relative to the cwd, or absolute. + const abs = path.resolve(arg); + const byDir = refs.find((p) => p.dir === abs); + if (byDir) return { project: byDir }; + const named = refs.filter((p) => p.name === arg); + if (named.length === 1) return { project: named[0] }; + if (named.length > 1) return { ambiguous: named }; + return {}; +} + +export async function folders(root = REPORTS_ROOT) { + return collapseFolders(foldersFor(await projectRefs(root))); +} diff --git a/umtool/lib/projects/index-db.mjs b/umtool/lib/projects/index-db.mjs @@ -0,0 +1,223 @@ +// A cache of the walk, in LMDB. +// +// HONEST SIZING FIRST. At twelve projects this saves 50 to 150 ms per load. It +// is NOT a speed fix today and is not presented as one. What it buys is: +// +// --since an agent asking what changed is a range read, not a diff of +// two full scans +// pagination recency as a range read, for when the tree is 500 projects +// counts decision counts without running the reducer, which is the +// cost that grows fastest -- it is the only part of a project +// read that touches megabytes of cue files +// +// THE STANDING RULE, and it is in the code because it is the one that keeps +// lib/browse.ts's "No database. The filesystem is the model." true: +// +// IF A VALUE EXISTS ONLY IN THE INDEX, THAT IS A BUG. +// +// Every read verifies a signature against the filesystem and falls back to a +// full read when it differs. A stale index self-heals on the next load and the +// user sees nothing but latency. An index-first read could claim something the +// filesystem does not say; that is the one failure this refuses. +import { createHash } from "node:crypto"; +import { mkdirSync } from "node:fs"; +import path from "node:path"; +import { INDEX_DIR } from "../paths.mjs"; + +/** Bump to invalidate every cached record at once. Folded into every signature. */ +export const INDEX_SCHEMA = 1; + +const NOOP = { + ok: false, + get: () => null, + put: () => {}, + del: () => {}, + recent: () => [], + byKind: () => [], + since: () => [], + stats: () => ({ ok: false, records: 0, schema: INDEX_SCHEMA, path: null }), + close: async () => {}, +}; + +/** + * Open the index, or hand back a no-op that answers null to everything. + * + * Copied in posture from common/lib/channelSignature.ts: a missing or + * unopenable index is a normal state (a fresh checkout, a deleted cache, a + * different machine), so it degrades rather than throwing. Callers treat null + * as "read it from disk". + */ +export async function openIndex({ readOnly = false, dir = INDEX_DIR } = {}) { + let open; + try { + // Imported lazily and inside the try, so a missing or unbuildable native + // module is the "no index yet" path rather than a crash at import time. + ({ open } = await import("lmdb")); + } catch { + return NOOP; + } + const file = path.join(dir, "projects.mdb"); + let root; + try { + if (!readOnly) mkdirSync(dir, { recursive: true }); + root = open({ path: file, maxDbs: 8, compression: false, readOnly }); + } catch { + return NOOP; + } + + let projects; + let meta; + let recentDb; + try { + meta = root.openDB({ name: "meta", encoding: "msgpack" }); + projects = root.openDB({ name: "projects", encoding: "msgpack" }); + // Key is [MAX - mtimeMs, id], so an ASCENDING range read is newest-first. + // The alternative -- reading everything and sorting -- is the thing an index + // is supposed to remove. + recentDb = root.openDB({ name: "recent", encoding: "msgpack" }); + } catch { + return NOOP; + } + + const schema = Number(meta.get("schema") ?? 0); + if (!readOnly && schema !== INDEX_SCHEMA) { + // A schema bump rewrites everything rather than migrating: the whole store + // is derived, and CACHE_DIR is documented as safe to delete at any time. + try { + projects.clearSync(); + recentDb.clearSync(); + meta.putSync("schema", INDEX_SCHEMA); + meta.putSync("builtAt", 0); + } catch { + return NOOP; + } + } else if (schema !== INDEX_SCHEMA) { + return NOOP; + } + + const MAX = 9_999_999_999_999; + const recentKey = (rec) => [MAX - Math.round(rec.newestMtimeMs ?? 0), rec.id]; + + return { + ok: true, + file, + + /** The cached record for an id, or null. Callers still verify the sig. */ + get(id) { + try { + return projects.get(id) ?? null; + } catch { + return null; + } + }, + + /** Best effort, always. A failed write is a cache miss next time, no more. */ + put(rec) { + try { + const old = projects.get(rec.id); + if (old) recentDb.removeSync(recentKey(old)); + projects.putSync(rec.id, rec); + recentDb.putSync(recentKey(rec), rec.id); + meta.putSync("builtAt", Date.now()); + } catch { + /* the filesystem is the model; this is only a cache */ + } + }, + + del(id) { + try { + const old = projects.get(id); + if (old) recentDb.removeSync(recentKey(old)); + projects.removeSync(id); + } catch { + /* ignore */ + } + }, + + /** Newest first, as a range read rather than a sort. */ + recent(limit = 50, offset = 0) { + try { + const out = []; + let i = 0; + for (const { value } of recentDb.getRange({})) { + if (i++ < offset) continue; + const rec = projects.get(value); + if (rec) out.push(rec); + if (out.length >= limit) break; + } + return out; + } catch { + return []; + } + }, + + byKind(kind) { + try { + return [...projects.getRange({})].map((e) => e.value).filter((r) => r?.kind === kind); + } catch { + return []; + } + }, + + /** What changed since a timestamp -- the reason this exists at all. */ + since(ms) { + try { + const out = []; + for (const { value } of recentDb.getRange({})) { + const rec = projects.get(value); + if (!rec) continue; + // The range is newest-first, so the first record older than the cutoff + // ends it. + if ((rec.newestMtimeMs ?? 0) <= ms) break; + out.push(rec); + } + return out; + } catch { + return []; + } + }, + + stats() { + try { + let records = 0; + for (const _ of projects.getRange({})) records += 1; + return { + ok: true, + records, + schema: INDEX_SCHEMA, + builtAt: Number(meta.get("builtAt") ?? 0), + path: file, + }; + } catch { + return { ok: false, records: 0, schema: INDEX_SCHEMA, path: file }; + } + }, + + async close() { + try { + await root.close(); + } catch { + /* ignore */ + } + }, + }; +} + +/** + * The freshness signature. INPUTS, never bytes. + * + * Signing the produced record instead would be circular, and signing bytes is + * what the export build learned not to do: an artefact with a timestamp in it is + * never byte-reproducible. The schema is folded in so a bump invalidates + * everything -- deliberately NOT a generation counter, which would invalidate + * every project whenever any one of them changed. + */ +export function signRecord({ kindSig, dirMs, markerMs, markerSize, outMs }) { + const h = createHash("sha1"); + h.update(`schema:${INDEX_SCHEMA}\n`); + h.update(`kind:${kindSig ?? ""}\n`); + h.update(`dir:${Math.round(dirMs ?? 0)}\n`); + h.update(`marker:${Math.round(markerMs ?? 0)}:${markerSize ?? 0}\n`); + h.update(`out:${Math.round(outMs ?? 0)}\n`); + return h.digest("hex").slice(0, 16); +} diff --git a/umtool/lib/projects/kinds.mjs b/umtool/lib/projects/kinds.mjs @@ -0,0 +1,193 @@ +// The project registry. +// +// A KIND is what a thing is; a TEMPLATE is which configuration of that kind it +// is. `um-song` is not a kind -- it is the one template of the `song` kind that +// exists so far, and the distinction is the whole point: the next thing this +// tool has to hold (a supercut, a cover set, a vertical short) will be a new +// template of an existing kind at least as often as a new kind. +// +// Plain ESM and NO `node:fs`, so `umtool` on the command line, a server +// component and a client component that only wants the chip labels all read one +// definition. The reading each kind does lives in its own module. +// +// TO ADD A KIND: add an entry here and a view. Nothing else. There is an e2e +// spec that fails if a kind id appears as a string literal anywhere outside +// lib/projects/ and components/projects/ -- that is the mechanical form of this +// rule, and it is what stops `if (kind === "report-video")` appearing in a page. +import { MANIFEST_NAME, REPORT_DECISION_KINDS, reportDecisions, reportSignature, summariseReport } from "./report.mjs"; +import { CUT_NAMES, songSignature, summariseSong } from "./song.mjs"; +import { sweepSignature, summariseSweep } from "./sweep.mjs"; + +/** + * Static children of app/browse/. A project whose FIRST path segment is one of + * these can never be opened at its own URL -- a static segment beats a dynamic + * one, so `/browse/find` renders the phrase console no matter what is on disk. + * + * Before this it failed SILENTLY, which is the worst available outcome: the + * project is listed, the link is there, and it goes somewhere else entirely. + * e2e/projects.spec.ts asserts this list equals the real directory listing, so + * a tenth tool page cannot quietly invalidate it. + */ +export const RESERVED_BROWSE = ["at", "decisions", "faces", "find", "sources", "trim"]; + +/** Directory names the walk never descends into. */ +export const SKIP_DIRS = new Set([ + "node_modules", + // A project is a LEAF, so these are only ever reached inside one -- but a + // half-built tree can have them at folder level too, and none of them can + // contain a project. + "out", + "variants", + "plan", + "thumbs", + // The e2e fixture symlinks 39 GB of audio here. + "data", +]); + +const has = (names, n) => names.has(n); +const someMatch = (names, re) => [...names].some((n) => re.test(n)); + +const SWEEP_RE = /(^|[-_])sweep([-_]report)?\.md$|^sweep-report\.md$/i; + +export const PROJECT_KINDS = [ + { + id: "report-video", + template: "cited-timeline", + label: "report video", + badge: "video", + // The manifest is both the marker and the edit decision list. Nothing else + // needs to exist for a directory to be a report video. + detect: (names) => has(names, MANIFEST_NAME), + stages: [ + { id: "windows", label: "windows" }, + { id: "fetched", label: "clips fetched" }, + { id: "built", label: "built" }, + ], + decisionKinds: REPORT_DECISION_KINDS, + summarise: summariseReport, + signature: reportSignature, + decisions: reportDecisions, + // Views reachable under the project's own URL: + // /browse/<project>/clip/<id> the window bench + // /browse/<project>/claim/<id> the adjudication bench + views: ["clip", "claim"], + }, + { + id: "song", + template: "um-song", + label: "song", + badge: "song", + detect: (names) => + has(names, "spec.json") || + has(names, "verdicts.json") || + CUT_NAMES.some((c) => has(names, `${c}.mp4`)), + stages: [ + { id: "draft", label: "no cuts" }, + { id: "built", label: "some cuts" }, + { id: "shipped", label: "all cuts" }, + ], + // Assembled from lib/decisions.ts, which still owns the song reducer -- it + // reads verdicts, plans, recipes, loudness and the accepted cover set, all + // of which live in TypeScript beside readSong(). Listed here so the union + // of every kind's vocabulary is computable without importing any of it. + decisionKinds: [ + "unjudged-variant", "missing-cut", "no-recipe", "stale-recipe", + "spec-problem", "no-plan", "unattributed", "thumb-unaccepted", "loudness", + ], + summarise: summariseSong, + signature: songSignature, + // Wired in lib/projects.ts, because the reducer is TypeScript. + decisions: null, + // `/browse/<song>/wide` -- today's cut page, unchanged. + views: CUT_NAMES, + }, + { + id: "sweep-report", + template: "sweep", + label: "sweep report", + badge: "report", + // A report that is not yet a video. It earns a kind on day one because one + // exists (~/reports/hasan-bike), and because a kind with no decisions, no + // build and no rich read is the cheapest possible proof that the registry + // is extensible. + detect: (names) => !has(names, MANIFEST_NAME) && someMatch(names, SWEEP_RE), + stages: [{ id: "draft", label: "no manifest" }], + decisionKinds: [], + summarise: summariseSweep, + signature: sweepSignature, + decisions: null, + views: [], + }, +]; + +// An escape hatch for the extensibility spec: it injects a fourth kind and +// asserts the index, the chips, the CLI and the project page all pick it up +// with no code edit anywhere. If adding a kind needs an edit outside this file, +// that spec fails loudly. +if (process.env.UMTOOL_EXTRA_KINDS) { + try { + for (const k of JSON.parse(process.env.UMTOOL_EXTRA_KINDS)) { + PROJECT_KINDS.push({ + stages: [{ id: "draft", label: "draft" }], + decisionKinds: [], + summarise: null, + signature: null, + decisions: null, + views: [], + ...k, + detect: (names) => has(names, k.marker), + }); + } + } catch { + /* a malformed override must not take the app down */ + } +} + +export const kindById = (id) => PROJECT_KINDS.find((k) => k.id === id) ?? null; + +/** What a client component needs, with none of what it must not have. */ +export const kindMeta = (k) => ({ + id: k.id, + template: k.template, + label: k.label, + badge: k.badge, + stages: k.stages, + decisionKinds: k.decisionKinds, +}); + +export const KIND_META = () => PROJECT_KINDS.map(kindMeta); + +/** Every decision kind any registered kind can emit. */ +export const ALL_DECISION_KINDS = () => [ + ...new Set(PROJECT_KINDS.flatMap((k) => k.decisionKinds)), +]; + +/** + * Which kind a directory is, from its entry names alone. + * + * `project.json` wins outright, so a directory can always declare itself and + * nothing on disk has to move to adopt the registry. Otherwise every kind's + * detect() runs and TWO matches is an error, never a guess -- a directory that + * is two kinds is a bug, and picking one would hide it. + */ +export function detectKind(names, declared = null) { + if (declared?.kind) { + const k = kindById(declared.kind); + if (k) return { kind: k.id, template: declared.template ?? k.template, declared: true }; + // A declared kind nobody registers is still a statement of intent; keep it + // so the index can say so rather than silently falling back to a guess. + return { kind: declared.kind, template: declared.template ?? "unknown", unknownKind: true }; + } + const hits = PROJECT_KINDS.filter((k) => { + try { + return k.detect(names); + } catch { + return false; + } + }); + if (!hits.length) return null; + if (hits.length > 1) { + return { kind: hits[0].id, template: hits[0].template, ambiguousWith: hits.map((k) => k.id) }; + } + return { kind: hits[0].id, template: hits[0].template }; +} diff --git a/umtool/lib/projects/report.mjs b/umtool/lib/projects/report.mjs @@ -0,0 +1,800 @@ +// The report-video kind's own reading: the manifest, what is wrong with it, and +// what state the build is in. +// +// Plain ESM, no TypeScript and no app imports, because `umtool check` has to run +// this from a terminal with no server. It is also the only place that knows the +// manifest's shape -- the walk knows a marker file, the page knows a summary, +// and neither parses JSON. +import { readdir, readFile, stat } from "node:fs/promises"; +import path from "node:path"; +import { DEFAULT_VARIANT, cachedWindowsFor, findContainingWindow } from "report-to-video/build-video"; + +/** + * Where a build's per-entry segments live. + * + * They moved under `out/<variant>/` when the pipeline learned to cut two + * versions of the same manifest. `out/segments` is still checked, because every + * other report on disk was built before that and its files are still there — + * and a poster tile silently going blank is exactly the kind of regression + * nothing would have reported. + */ +const segmentDirs = (outDir) => [ + path.join(outDir, DEFAULT_VARIANT, "segments"), + path.join(outDir, "segments"), +]; +import { adjudicationGaps, ledgerTotals, unadjudicatedOf } from "report-to-video/ledger-totals"; +import { widen } from "report-to-video/resolve-windows"; + +// widen() and the cache's window naming are IMPORTED, never reimplemented. The +// bench's "extend to sentence end" has to be the same function the CLI runs, or +// the UI and `resolve-windows --write` will disagree about where a clip ends -- +// and the CLI is the one that wins, silently, on the next build. +export { widen }; + +export const MANIFEST_NAME = "video.manifest.json"; + +const GLOBAL_CHANNELS_DIR = () => + process.env.CHANNELS_DIR ?? + "/home/user/Projects/yt-dlp-transcript-browser/transcripts/channels"; + +/** The conventional name make-shadow-channels.sh builds. */ +export const SHADOW_CHANNELS = ".shadow-channels"; + +// --------------------------------------------------------------------------- +// Which channels directory THIS project reads. +// +// Not always the global one, and assuming it was produced three confident, +// wrong "the build dies here" findings on the first run against real data. +// quartering-flagging-takedowns cites three deleted YouTube uploads and cuts +// them from live Rumble mirrors the corpus has no entry for; rather than write +// into transcripts/channels/ (which would perturb the export build's change +// detection) it builds a SHADOW tree that symlinks every real channel and adds +// just those three. Its documented build command sets CHANNELS_DIR to it. +// +// So the project gets to say. `provenance.channelsDir` is the explicit form and +// is resolved relative to the project; `.shadow-channels/` is the convention, +// detected because it already exists and nothing had to change on disk for it +// to work. Neither is a guess: both are things the project itself wrote down. +// --------------------------------------------------------------------------- +export function channelsDirFor(dir, manifest, { shadowExists = false } = {}) { + const declared = manifest?.provenance?.channelsDir; + if (declared) return path.resolve(dir, declared); + if (shadowExists) return path.join(dir, SHADOW_CHANNELS); + return GLOBAL_CHANNELS_DIR(); +} + +export const hasShadowChannels = (dir) => + stat(path.join(dir, SHADOW_CHANNELS)).then((s) => s.isDirectory(), () => false); + +/** The same regex resolve-windows.mjs uses. Imported there, duplicated nowhere. */ +export const ENDS_SENTENCE = /[.!?]["'”’)\]]*\s*$/; +export const IS_FILLER = /^\s*(\[[^\]]*\]|>>|♪|—|-)*\s*$/; + +// A QR built from any of these resolves to nothing on somebody else's phone. +// Both of the real defects this catches had already shipped: one manifest has no +// siteOrigin at all (19 codes reading `undefined/?v=…`) and one has localhost. +const DEAD_ORIGIN = + /^https?:\/\/(localhost|127\.|0\.0\.0\.0|\[::1\]|10\.|192\.168\.|172\.(1[6-9]|2\d|3[01])\.)/i; + +export const isDeadOrigin = (o) => !o || typeof o !== "string" || DEAD_ORIGIN.test(o); + +const stat0 = (p) => stat(p).then((s) => s, () => null); + +export const manifestPath = (dir) => path.join(dir, MANIFEST_NAME); + +export async function readManifest(dir) { + try { + return JSON.parse(await readFile(manifestPath(dir), "utf8")); + } catch { + return null; + } +} + +/** Clips only, in timeline order. Array order IS the cut; nothing sorts. */ +export const clipsOf = (m) => (m?.timeline ?? []).filter((e) => e?.type === "clip"); +export const cardsOf = (m) => (m?.timeline ?? []).filter((e) => e?.type === "card"); + +export const channelFor = (m, e) => e.channel ?? m?.provenance?.channelSlug ?? null; + +export const cuePathFor = (m, e, channelsDir) => { + const chan = channelFor(m, e); + if (!chan) return null; + return path.join(channelsDir ?? GLOBAL_CHANNELS_DIR(), chan, "data", e.video, "transcript.cues.json"); +}; + +// --------------------------------------------------------------------------- +// Cue files are the expensive input: a nine-hour stream's cues run to megabytes, +// and six reports citing thirteen videos each would be a hundred-odd megabytes of +// JSON on every index load. So the DERIVED answers are memoised against the +// file's own mtime and size -- a cue file is an archive artefact and does not +// change under us, so a hit is permanent in practice. +// --------------------------------------------------------------------------- +const cueMemo = new Map(); + +export async function readCues(file) { + const st = await stat0(file); + if (!st) return null; + const key = `${file}|${Math.round(st.mtimeMs)}|${st.size}`; + const hit = cueMemo.get(key); + if (hit) return hit; + let doc; + try { + doc = JSON.parse(await readFile(file, "utf8")); + } catch { + return null; + } + const cues = doc.cues ?? []; + const value = { + cues, + title: doc.title, + uploadDate: doc.uploadDate, + webpageUrl: doc.webpageUrl, + duration: doc.duration, + // What fraction of cues close a sentence. Below ~10% the upload's ASR + // carries no punctuation worth the name, and sentence-widening cannot help + // -- which is a thing to SAY, not a thing to let somebody rediscover. + punctuationRate: + cues.length === 0 + ? 0 + : cues.filter((c) => ENDS_SENTENCE.test(c.text ?? "")).length / cues.length, + }; + cueMemo.set(key, value); + return value; +} + +/** The cue whose span contains t, else the nearest on the right. */ +export function cueAt(cues, t, which = "start") { + const EPS = 0.02; + if (!cues.length) return null; + if (which === "end") { + const j = cues.findIndex((c) => c.end >= t - EPS); + return cues[j < 0 ? cues.length - 1 : j]; + } + const i = cues.findIndex((c) => c.end > t); + return cues[i < 0 ? cues.length - 1 : i]; +} + +// --------------------------------------------------------------------------- +// The build's own state, read from the output directory. +// +// Deliberately NOT a readdir of out/ -- it holds 43 to 63 files per project and +// the index would pay for all of them. One stat for the deliverable, one readdir +// of clips-raw (a few dozen names) to tell "windows written" from "clips +// fetched", and nothing else. +// --------------------------------------------------------------------------- +export async function buildStateOf(dir, manifest) { + const outDir = path.join(dir, "out"); + const slug = manifest?.slug ?? path.basename(dir); + const finalPath = path.join(outDir, `${slug}.mp4`); + const [fin, man] = await Promise.all([stat0(finalPath), stat0(manifestPath(dir))]); + + let rawCount = 0; + let segCount = 0; + let firstSeg = null; + let firstRaw = null; + let segRel = path.posix.join("out", "segments"); + if (!fin) { + const dirs = segmentDirs(outDir); + const [raws, ...segLists] = await Promise.all([ + readdir(path.join(outDir, "clips-raw")).catch(() => []), + ...dirs.map((d) => readdir(d).catch(() => [])), + ]); + const which = segLists.findIndex((l) => l.length); + const segs = which < 0 ? [] : segLists[which]; + segRel = + which === 0 + ? path.posix.join("out", DEFAULT_VARIANT, "segments") + : path.posix.join("out", "segments"); + const rawMp4 = raws.filter((n) => n.endsWith(".mp4")).sort(); + const segMp4 = segs.filter((n) => n.endsWith(".mp4")).sort(); + rawCount = rawMp4.length; + segCount = segMp4.length; + firstSeg = segMp4[0] ?? null; + firstRaw = rawMp4[0] ?? null; + } + + const stale = !!(fin && man && fin.mtimeMs < man.mtimeMs); + return { + outDir, + slug, + finalPath, + built: !!fin, + stale, + finalSize: fin?.size ?? 0, + finalMtimeMs: fin?.mtimeMs ?? 0, + manifestMtimeMs: man?.mtimeMs ?? 0, + rawCount, + segCount, + firstSeg, + firstRaw, + segRel, + }; +} + +/** The signature the summary and the decisions are cached against. */ +export async function reportSignature(dir) { + const [man, out] = await Promise.all([ + stat0(manifestPath(dir)), + stat0(path.join(dir, "out")), + ]); + return [ + Math.round(man?.mtimeMs ?? 0), + man?.size ?? 0, + Math.round(out?.mtimeMs ?? 0), + ].join(":"); +} + +const fmtDur = (s) => { + const t = Math.round(s); + const m = Math.floor(t / 60); + return m >= 60 + ? `${Math.floor(m / 60)}h${String(m % 60).padStart(2, "0")}m` + : `${m}m${String(t % 60).padStart(2, "0")}s`; +}; + +/** + * The card. Cheap by construction: the manifest, one stat, one readdir. + * + * Runtime is SUMMED FROM THE WINDOWS, not probed. It is the right number anyway + * -- it is what the cut will be if it is built -- and a probe per project would + * put a second in front of the index. + */ +export async function summariseReport(ctx) { + const { dir } = ctx; + const m = await readManifest(dir); + const clips = clipsOf(m); + const cards = cardsOf(m); + // Anything that is neither. The vocabulary is open, so a card that says + // "19 clips · 10 cards" about a 31-entry timeline is lying by omission. + const others = (m?.timeline ?? []).filter((e) => e?.type !== "clip" && e?.type !== "card"); + const build = await buildStateOf(dir, m); + + const runtime = clips.reduce((n, e) => n + Math.max(0, (e.end ?? 0) - (e.start ?? 0)), 0); + const sources = new Set(clips.map((e) => `${channelFor(m, e)}/${e.video}`)).size; + const locked = clips.filter((e) => e.lock).length; + + const state = !m + ? "draft" + : build.stale + ? "stale" + : build.built + ? "built" + : build.segCount > 0 || build.rawCount > 0 + ? "fetched" + : clips.length + ? "windows" + : "draft"; + + const facts = []; + if (clips.length) facts.push(`${clips.length} clip${clips.length === 1 ? "" : "s"}`); + if (cards.length) facts.push(`${cards.length} card${cards.length === 1 ? "" : "s"}`); + if (others.length) { + const kinds = [...new Set(others.map((e) => e.type ?? "entry"))].sort(); + facts.push(`${others.length} ${kinds.join("/")}`); + } + if (runtime > 0) facts.push(fmtDur(runtime)); + if (sources) facts.push(`${sources} source${sources === 1 ? "" : "s"}`); + if (locked) facts.push(`${locked} locked`); + if (build.built) facts.push(`${(build.finalSize / 1e6).toFixed(0)} MB`); + + const flags = []; + if (isDeadOrigin(m?.provenance?.siteOrigin)) flags.push("dead QR origin"); + if (!m?.provenance?.channelSlug) flags.push("no channelSlug"); + if (build.stale) flags.push("older than its manifest"); + + return { + title: m?.title ?? ctx.name, + subtitle: m?.subtitle ?? m?.generatedOn ?? null, + state, + newestMtimeMs: Math.max(build.manifestMtimeMs, build.finalMtimeMs), + facts, + flags, + // A card should look like the video as soon as anything of it exists. A + // built SEGMENT already carries the chrome, so it is the better mid-build + // poster than a raw clip; a raw clip is better than a blank tile. + posterRel: build.built + ? path.posix.join("out", `${build.slug}.mp4`) + : build.firstSeg + ? path.posix.join(build.segRel, build.firstSeg) + : build.firstRaw + ? path.posix.join("out", "clips-raw", build.firstRaw) + : null, + haystack: [ + ctx.id, m?.title, m?.subtitle, m?.slug, + m?.provenance?.channel, m?.provenance?.channelSlug, + ...clips.map((e) => e.video), + ] + .filter(Boolean) + .join(" ") + .toLowerCase(), + attrs: { + clips: String(clips.length), + cards: String(cards.length), + other: String(others.length), + sources: String(sources), + locked: String(locked), + ...(build.built ? { built: "1" } : {}), + ...(isDeadOrigin(m?.provenance?.siteOrigin) ? { "dead-origin": "1" } : {}), + }, + // Kept for the project page and the decisions pass, so neither re-reads. + manifest: m, + build, + }; +} + +// --------------------------------------------------------------------------- +// Decisions. +// +// Severity is EARNED. `blocking` means something downstream would LIE or die if +// you acted on it: a QR that resolves nowhere, a clip whose cue file is absent +// (the build dies there), a source that is gone. Everything else is `open` at +// most, and a manifest with no build yet is `info` -- that is a normal state to +// be in, not a decision anybody is waiting on. +// --------------------------------------------------------------------------- +export const REPORT_DECISION_KINDS = [ + "manifest-invalid", + "clip-no-cues", + "clip-unfetchable", + "clip-mid-sentence", + "window-overlap", + "no-punctuation", + // A ledger entry nobody has ruled on. BLOCKING, which is earned here: both + // the stated and the implied total lie if you act on an unadjudicated ledger, + // and they lie quietly, in a chart, with his name on it. + "claim-unadjudicated", + // A fired coherence predicate. `open`, not blocking, because the decision is + // EDITORIAL rather than mechanical: is the flag right, and does it belong on + // screen? Neither answer stops a build. + "claim-incoherent", + "stale-build", + "unbuilt", +]; + +export async function reportDecisions(ctx, summary) { + const { id, dir } = ctx; + const s = summary ?? (await summariseReport(ctx)); + const m = s.manifest; + const href = `/browse/${id}`; + const clipHref = (cid) => `/browse/${id}/clip/${cid}`; + const at = s.newestMtimeMs; + const out = []; + if (!m) return out; + + const add = (kind, target, why, severity, extra = {}) => + out.push({ kind, project: id, target, why, href, severity, at, ...extra }); + + const shadowExists = await hasShadowChannels(dir); + const channelsDir = channelsDirFor(dir, m, { shadowExists }); + // A project that ships a builder for its shadow tree but has not run it is a + // DIFFERENT problem from one whose sources are gone, and the fix is one + // command rather than an editorial decision. + const shadowBuilder = + !shadowExists && + (await stat(path.join(dir, "make-shadow-channels.sh")).then(() => true, () => false)); + + // --- the manifest itself ------------------------------------------------- + const origin = m.provenance?.siteOrigin; + if (isDeadOrigin(origin)) { + add( + "manifest-invalid", + "provenance.siteOrigin", + origin + ? `\`${origin}\` — every QR in this cut resolves to nothing on anyone else's phone` + : "missing — every QR in this cut encodes `undefined/?v=…`", + "blocking", + ); + } + if (!m.provenance?.channelSlug) { + add( + "manifest-invalid", + "provenance.channelSlug", + "missing — a clip with no `channel` of its own has no cue file to find", + "blocking", + ); + } + + const clips = clipsOf(m); + const seen = new Map(); + for (const e of m.timeline ?? []) { + if (!e?.id) continue; + seen.set(e.id, (seen.get(e.id) ?? 0) + 1); + } + for (const [eid, n] of seen) { + if (n > 1) { + add( + "manifest-invalid", + eid, + `${n} entries share the id \`${eid}\` — segments overwrite each other`, + "blocking", + ); + } + } + + const nodeCount = (m.timelineNodes ?? []).length; + if (nodeCount > 0) { + for (const e of clips) { + if (typeof e.section === "number" && (e.section < 0 || e.section >= nodeCount)) { + add( + "manifest-invalid", + e.id, + `section ${e.section} but only ${nodeCount} timeline node(s) — the footer marker would run off the track`, + "blocking", + { href: clipHref(e.id) }, + ); + } + } + } + + // --- availability, from the recorded check (never a live one) ------------ + // The reducer must not shell out: an inbox that runs yt-dlp once per source is + // an inbox that takes a minute to open. So it reads what the preflight wrote, + // and says when nobody has run one. + const avail = await readFile(path.join(dir, "out", "availability.json"), "utf8").then( + (t) => JSON.parse(t), + () => null, + ); + if (avail) { + for (const src of avail.sources ?? []) { + if (src.ok) continue; + // SEVERITY FOLLOWS THE CLIPS, not the source. + // + // The probe covers ledger sources as well as clip sources now, and most + // dead ledger sources are cited by no clip at all — the cut quotes them on + // a card precisely BECAUSE they are gone. Calling those blocking made the + // inbox say "0 clip(s) cite it; the build dies here", which refutes itself + // in its own sentence, and put six permanent red rows in front of a + // manifest that builds cleanly. + const cited = src.clips?.length ?? 0; + const claimed = src.claims?.length ?? 0; + add( + src.state === "no-cues" ? "clip-no-cues" : "clip-unfetchable", + src.key, + cited + ? `${src.state} — ${cited} clip(s) cite it; the build dies here` + : `${src.state} — no clip cites it${claimed ? `, ${claimed} ledger claim(s) do` : ""}. ` + + "A cut that quotes it on a card is unaffected; one that adds a clip from it will not build", + cited ? "blocking" : "info", + ); + } + } + + // --- per clip, against the cues ----------------------------------------- + const byVideo = new Map(); + for (const e of clips) { + const key = `${channelFor(m, e)}/${e.video}`; + if (!byVideo.has(key)) byVideo.set(key, []); + byVideo.get(key).push(e); + } + + const unpunctuated = []; + for (const [key, list] of byVideo) { + const file = cuePathFor(m, list[0], channelsDir); + const doc = file ? await readCues(file) : null; + + if (!doc) { + // Reported once per source, not once per clip. Almost always the Rumble + // two-ids trap: the manifest names the site/MCP id while the cue file + // lives under the URL slug. + if (!avail?.sources?.some((s) => s.key === key && !s.ok)) { + add( + "clip-no-cues", + key, + shadowBuilder + ? "no transcript.cues.json — but this project ships make-shadow-channels.sh, " + + `which is what puts it there. Run it before building ${list.map((e) => e.id).join(", ")}` + : `no transcript.cues.json — the build dies at ${list.map((e) => e.id).join(", ")}. ` + + "On Rumble, check `video` is the URL slug and not the site id", + "blocking", + ); + } + continue; + } + + if (doc.punctuationRate < 0.1) unpunctuated.push(key); + + for (const e of list) { + // The standing rule, mechanised. One 14-clip cut shipped with 8 clips + // ending mid-thought. `lockEnd` is the ACKNOWLEDGEMENT -- setting it is + // the author saying "I meant to cut here" -- so it clears this. + if (!e.lockEnd && !e.lock && doc.punctuationRate >= 0.1) { + const c = cueAt(doc.cues, e.end, "end"); + if (c && !ENDS_SENTENCE.test(c.text ?? "") && !IS_FILLER.test(c.text ?? "")) { + add( + "clip-mid-sentence", + e.id, + `ends mid-sentence: “…${String(c.text ?? "").trim().slice(-48)}”`, + "open", + { href: clipHref(e.id) }, + ); + } + } + } + + // Two clips from one source that overlap play as the same footage twice. + // resolve-windows de-overlaps them -- unless the earlier one has lockEnd, + // where it warns and refuses. That refusal is the decision. + const sorted = [...list].sort((a, b) => a.start - b.start); + for (let i = 0; i < sorted.length - 1; i += 1) { + const a = sorted[i]; + const b = sorted[i + 1]; + if (a.end <= b.start) continue; + add( + "window-overlap", + a.id, + `overlaps ${b.id} by ${(a.end - b.start).toFixed(1)}s` + + (a.lockEnd ? " and has lockEnd, so nothing will trim it" : ""), + a.lockEnd ? "open" : "info", + { href: clipHref(a.id) }, + ); + } + } + + // One row, not one per source. Six real projects produce forty-odd of these + // between them, and an inbox that says the same true thing forty times is an + // inbox whose blocking rows have scrolled off the top. + if (unpunctuated.length) { + add( + "no-punctuation", + unpunctuated.length === 1 ? unpunctuated[0] : `${unpunctuated.length} sources`, + `unpunctuated ASR in ${unpunctuated.slice(0, 3).join(", ")}` + + `${unpunctuated.length > 3 ? ` and ${unpunctuated.length - 3} more` : ""}` + + " — widening cannot help there; set those edges by ear and lock them", + "info", + ); + } + + // --- the ledger ---------------------------------------------------------- + // The same rule the availability read follows: this reducer NEVER measures. + // ledgerTotals() is pure arithmetic over JSON already parsed into memory, so + // the inbox opens in the time it did before -- audio is fetched only when a + // claim page is opened, one claim at a time. + const ledger = Array.isArray(m.ledger) ? m.ledger : []; + if (ledger.length) { + const claimHref = (cid) => `/browse/${id}/claim/${cid}`; + + // One row per entry, not one collapsed row. Forty unpunctuated sources are + // the same true thing said forty times; forty unadjudicated claims are + // forty DIFFERENT decisions, and the inbox being empty is the sign-off. + for (const e of unadjudicatedOf(ledger)) { + const gaps = adjudicationGaps(e); + add( + "claim-unadjudicated", + e.id, + `${e.date ?? "undated"} · ${e.display ?? e.value ?? "—"} — no ruling on ` + + `${gaps.join(", ")}. Settle it against ±90s of context, not the quote`, + "blocking", + { href: claimHref(e.id) }, + ); + } + + // Coherence over whatever HAS been ruled on, so these appear as the work + // proceeds rather than all at once at the end. + let totals = null; + try { + totals = ledgerTotals(ledger, { strict: false }); + } catch { + /* a malformed ledger is already reported as unadjudicated rows */ + } + for (const step of totals?.steps ?? []) { + for (const f of step.flags) { + add( + "claim-incoherent", + step.id, + `${f.rule.replace(/_/g, " ")} — ${f.text}`, + "open", + { href: claimHref(step.id) }, + ); + } + } + } + + // --- the build ----------------------------------------------------------- + if (s.build.stale) { + add( + "stale-build", + `out/${s.build.slug}.mp4`, + "older than the manifest that describes it — the file is a lead, not a fact", + "open", + ); + } else if (!s.build.built) { + add("unbuilt", `out/${s.build.slug}.mp4`, "never built", "info"); + } + + return out; +} + + +// --------------------------------------------------------------------------- +// Per-clip detail: what the project page lists and the clip bench edits. +// +// This is the expensive read -- one cue file per distinct source -- so it is +// NOT what the index calls. summariseReport() is. +// --------------------------------------------------------------------------- +export async function readClipDetail(dir, { manifest = null } = {}) { + const m = manifest ?? (await readManifest(dir)); + if (!m) return null; + + const shadowExists = await hasShadowChannels(dir); + const channelsDir = channelsDirFor(dir, m, { shadowExists }); + const build = await buildStateOf(dir, m); + const rawDir = path.join(dir, "out", "clips-raw"); + const segDirs = segmentDirs(path.join(dir, "out")); + const segLists = await Promise.all(segDirs.map((d) => readdir(d).catch(() => []))); + const segWhich = segLists.findIndex((l) => l.length); + const segNames = new Set(segWhich < 0 ? [] : segLists[segWhich]); + const segRel = + segWhich === 0 + ? path.posix.join("out", DEFAULT_VARIANT, "segments") + : path.posix.join("out", "segments"); + const cueCache = new Map(); + + const entries = []; + for (const e of m.timeline ?? []) { + // A CLIP is `type === "clip"`. Everything else is a non-clip entry with no + // window and no source. + // + // Not `!== "card"`: the timeline's vocabulary is OPEN. quartering-employee- + // count carries `scroll` and `chart` entries beside its cards, and treating + // anything-that-is-not-a-card as a clip sent `undefined` into path.join() + // and 500'd the whole project page. Every other real manifest is clips only, + // which is exactly why this survived testing. + if (e.type !== "clip") { + entries.push({ ...e, kind: e.type ?? "entry" }); + continue; + } + const chan = channelFor(m, e); + const key = `${chan}/${e.video}`; + if (!cueCache.has(key)) { + const file = cuePathFor(m, e, channelsDir); + cueCache.set(key, file ? await readCues(file) : null); + } + const doc = cueCache.get(key); + + // The pad the build would use, so "is this clip cached" answers the same + // question the build will ask. + const pad = m.render?.fetchPad ?? 3.0; + const from = Math.max(0, e.start - pad); + const to = e.end + pad; + const cached = await findContainingWindow(rawDir, e.video, from, to); + const allWindows = await cachedWindowsFor(rawDir, e.video); + // The bench wants the WIDEST containing file (room to drag); the build wants + // the tightest (least to decode). They are different questions. + const widest = allWindows + .filter((w) => w.from <= e.start && w.to >= e.end) + .sort((a, b) => b.to - b.from - (a.to - a.from))[0] ?? null; + + const endCue = doc ? cueAt(doc.cues, e.end, "end") : null; + // NULL means "cannot be known", and that is a third answer worth having. + // + // In an upload whose ASR emitted no terminators, every clip "ends + // mid-sentence" and the fact says nothing about the cut. Reporting it as + // FALSE put four ferret-rescue clips under a warning the decisions inbox + // (which has always had this gate) correctly stayed silent about -- the page + // and the inbox disagreeing about the same clip. The honest answer is that + // the source cannot support the question. + const noPunctuation = !!doc && doc.punctuationRate < 0.1; + const endsSentence = + !endCue || noPunctuation ? null : ENDS_SENTENCE.test(endCue.text ?? ""); + // What resolve-windows WOULD do, computed in-process because widen() is pure + // once the cues are read. It is the difference between "run the widener and + // see" and knowing before you touch anything. + let proposed = null; + if (doc?.cues?.length && !e.lock) { + const w = widen(doc.cues, e.start, e.end); + if (e.lockStart) w.start = e.start; + if (e.lockEnd) w.end = e.end; + const moved = Math.abs(w.start - e.start) > 0.05 || Math.abs(w.end - e.end) > 0.05; + proposed = moved ? { start: Number(w.start.toFixed(2)), end: Number(w.end.toFixed(2)) } : null; + } + + entries.push({ + ...e, + kind: "clip", + channel: chan, + cueFile: cuePathFor(m, e, channelsDir), + hasCues: !!doc, + duration: doc?.duration ?? null, + sourceTitle: doc?.title ?? null, + punctuationRate: doc?.punctuationRate ?? null, + endCueText: endCue?.text ?? null, + endsSentence, + noPunctuation, + proposed, + cached: cached ? { name: cached.name, from: cached.from, to: cached.to } : null, + widest: widest ? { name: widest.name, from: widest.from, to: widest.to } : null, + segment: segNames.has(`${e.id}.mp4`) ? path.posix.join(segRel, `${e.id}.mp4`) : null, + wantFrom: from, + wantTo: to, + }); + } + + return { manifest: m, build, channelsDir, shadowExists, entries }; +} + +/** The cues a clip bench draws, trimmed to a window. Absolute source seconds. */ +export async function cuesInWindow(dir, clipId, from, to) { + const m = await readManifest(dir); + const e = clipsOf(m).find((x) => x.id === clipId); + if (!e) return null; + const shadowExists = await hasShadowChannels(dir); + const file = cuePathFor(m, e, channelsDirFor(dir, m, { shadowExists })); + const doc = file ? await readCues(file) : null; + if (!doc) return null; + return { + duration: doc.duration ?? null, + punctuationRate: doc.punctuationRate, + cues: doc.cues + .filter((c) => c.end >= from && c.start <= to) + .map((c) => ({ + start: c.start, + end: c.end, + text: c.text, + endsSentence: ENDS_SENTENCE.test(c.text ?? "") && !IS_FILLER.test(c.text ?? ""), + })), + }; +} + +// --------------------------------------------------------------------------- +// One CLAIM, ready to adjudicate. +// +// The claim bench asks a different question from the clip bench. A clip asks +// "where exactly does this cut?", so it wants sample-accurate edges. A claim +// asks "what did he mean by that?", so it wants ROOM -- the standing rule is +// that a first-person quote is routinely the host reading someone else's words +// or being sarcastic, and neither is visible inside the quote itself. Hence a +// default of +/-90s, and hence context is returned as cues rather than as a +// waveform: the words on either side are what settle a scope. +// --------------------------------------------------------------------------- + +export const CLAIM_CONTEXT_PAD = 90; + +export async function readClaimDetail(dir, claimId, { manifest = null, pad = CLAIM_CONTEXT_PAD } = {}) { + const m = manifest ?? (await readManifest(dir)); + if (!m) return null; + const claim = (m.ledger ?? []).find((e) => e.id === claimId); + if (!claim) return null; + + const shadowExists = await hasShadowChannels(dir); + const channelsDir = channelsDirFor(dir, m, { shadowExists }); + const file = claim.video ? cuePathFor(m, claim, channelsDir) : null; + const doc = file ? await readCues(file) : null; + + const at = Number.isFinite(Number(claim.cite)) ? Number(claim.cite) : null; + const from = at == null ? 0 : Math.max(0, at - pad); + const to = at == null ? 0 : at + pad; + + // Which cached raw windows already cover this moment. A claim that lands + // inside one a clip fetched earlier is playable with no download at all, + // which is most of them once a build has run. + const rawDir = path.join(dir, "out", "clips-raw"); + const windows = claim.video + ? (await cachedWindowsFor(rawDir, claim.video)).filter( + (w) => at != null && w.from <= at && w.to >= at, + ) + : []; + windows.sort((a, b) => b.to - b.from - (a.to - a.from)); + + return { + claim, + at, + view: { from, to }, + pad, + // Every cue in the window, with the cited one marked. The mark is what + // makes "he is quoting a tweet here" visible: the quote sits in a paragraph + // rather than alone. + cues: (doc?.cues ?? []) + .filter((c) => c.end >= from && c.start <= to) + .map((c) => ({ + start: c.start, + end: c.end, + text: c.text, + cited: at != null && c.start <= at && c.end >= at, + })), + source: doc + ? { title: doc.title ?? null, uploadDate: doc.uploadDate ?? null, duration: doc.duration ?? null, webpageUrl: doc.webpageUrl ?? null } + : null, + noCues: !doc, + windows: windows.map((w) => ({ name: w.name, from: w.from, to: w.to })), + gaps: adjudicationGaps(claim), + }; +} diff --git a/umtool/lib/projects/song-ids.mjs b/umtool/lib/projects/song-ids.mjs @@ -0,0 +1,35 @@ +// songIds(), in its own module for one specific reason. +// +// It needs the WALK, and the walk needs the registry, and the registry needs +// this kind's cut list -- so putting it in song.mjs closes the cycle +// song -> walk -> kinds -> song. Under plain node that survives (the back edge +// is only used at call time), but Turbopack evaluates the bundle in an order +// where kinds.mjs reads CUT_NAMES while song.mjs is still initialising, and +// every song page 500s with "Cannot access 'CUT_NAMES' before initialization". +// +// `pnpm build` does not catch it, because nothing prerenders. A page render +// does. Hence a third module, which only the app's edge imports. +import path from "node:path"; +import { walkProjects } from "./walk.mjs"; +import { SONG_KIND } from "./song.mjs"; + +/** + * Every song, by the BASENAME the song routes are keyed on. + * + * A caller of the project walk rather than its own readdir, so there is one + * enumerator and a song cannot be a project in one listing and absent from the + * other. + * + * Filtered to songs that live directly under `browseRoot`, because every song + * route resolves through songDir() -- path.join(BROWSE_ROOT, id). A song project + * found anywhere else is still listed and summarised on /browse by the registry; + * returning it here would hand those routes an id that resolves to a directory + * that is not it, which is worse than omitting it. + */ +export async function songIdsUnder(reportsRoot, browseRoot) { + const projects = await walkProjects(reportsRoot); + return projects + .filter((p) => p.kind === SONG_KIND && p.dir === path.join(browseRoot, p.name)) + .map((p) => p.name) + .sort(); +} diff --git a/umtool/lib/projects/song.mjs b/umtool/lib/projects/song.mjs @@ -0,0 +1,101 @@ +// The song kind, read cheaply. +// +// The RICH read is lib/browse.ts's readSong(): verdict tallies against thumb +// aliases, plan provenance, spec validation, the accepted-cover set. That stays +// where it is and the project page still uses it. This is the index's version -- +// two readdirs and one small JSON -- because `umtool ls` has to work from a +// terminal with no server, and because the index pays this per project. +import { readdir, readFile, stat } from "node:fs/promises"; +import path from "node:path"; +/** This kind's registry id. The one place the string lives. */ +export const SONG_KIND = "song"; + +// The cut list is FIXED, not derived from the directory: mortal-kombat and +// mario-rpg have no vertical.mp4, and a scan-derived list would render those +// songs as complete. A hole in the deliverable set is information. +// +// Defined here rather than in lib/browse.ts so the CLI and the app share one +// list; lib/browse.ts re-exports it. +export const CUT_NAMES = ["wide", "wide-short", "vertical", "vertical-short"]; + +const MEDIA_RE = /\.(mp4|mkv|webm|mov|m4v)$/i; + +const stat0 = (p) => stat(p).then((s) => s, () => null); + +const listMediaIn = (dir) => + readdir(dir, { withFileTypes: true }).then( + (es) => es.filter((e) => e.isFile() && MEDIA_RE.test(e.name) && !e.name.startsWith(".")).map((e) => e.name), + () => [], + ); + +export async function songSignature(dir) { + const [d, v] = await Promise.all([stat0(dir), stat0(path.join(dir, "verdicts.json"))]); + return [Math.round(d?.mtimeMs ?? 0), Math.round(v?.mtimeMs ?? 0), v?.size ?? 0].join(":"); +} + +export async function summariseSong(ctx) { + const { dir, name } = ctx; + const [top, variants, readme, verdicts] = await Promise.all([ + listMediaIn(dir), + listMediaIn(path.join(dir, "variants")), + readFile(path.join(dir, "README.md"), "utf8").catch(() => null), + readFile(path.join(dir, "verdicts.json"), "utf8").then( + (t) => JSON.parse(t), + () => ({}), + ), + ]); + + const bases = new Set(top.map((n) => n.replace(MEDIA_RE, ""))); + const present = CUT_NAMES.filter((c) => bases.has(c)); + const missing = CUT_NAMES.filter((c) => !bases.has(c)); + + const counts = { keep: 0, reject: 0, undecided: 0 }; + const all = [...top, ...variants.map((n) => path.posix.join("variants", n))]; + for (const rel of all) { + const v = verdicts[rel]?.verdict; + counts[v === "keep" || v === "reject" ? v : "undecided"] += 1; + } + + let newest = 0; + for (const rel of all) { + const st = await stat0(path.join(dir, rel)); + if (st) newest = Math.max(newest, st.mtimeMs); + } + + const title = readme?.match(/^#\s+(.+)$/m)?.[1]?.trim() ?? name; + const state = present.length === 0 ? "draft" : present.length === CUT_NAMES.length ? "shipped" : "built"; + + const facts = [ + `${present.length}/${CUT_NAMES.length} cuts`, + `${variants.length} variant${variants.length === 1 ? "" : "s"}`, + ]; + if (counts.undecided) facts.push(`${counts.undecided} undecided`); + + return { + title, + subtitle: missing.length ? `no ${missing.join(", ")}` : null, + state, + newestMtimeMs: newest, + facts, + flags: [], + posterRel: present.length ? `${present[0]}.mp4` : null, + haystack: [ctx.id, title, ...bases].join(" ").toLowerCase(), + attrs: { + song: name, + cuts: String(present.length), + variants: String(variants.length), + keep: String(counts.keep), + reject: String(counts.reject), + undecided: String(counts.undecided), + // Omitted entirely when the set is complete: a hole in the deliverables + // is information, and its ABSENCE has to be readable as "nothing missing" + // rather than as an empty list. + ...(missing.length ? { missing: missing.join(",") } : {}), + }, + counts, + present, + missing, + variantCount: variants.length, + }; +} + diff --git a/umtool/lib/projects/sweep.mjs b/umtool/lib/projects/sweep.mjs @@ -0,0 +1,49 @@ +// The sweep-report kind: a cited report that is not yet a video. +// +// It has no decisions, no build and no rich read, which is exactly why it is +// here on day one: it is the cheapest possible proof that adding a kind costs a +// registry entry and nothing else. +import { readdir, readFile, stat } from "node:fs/promises"; +import path from "node:path"; + +const stat0 = (p) => stat(p).then((s) => s, () => null); + +const SWEEP_RE = /(^|[-_])sweep([-_]report)?\.md$|^sweep-report\.md$/i; + +export async function sweepSignature(dir) { + const st = await stat0(dir); + return String(Math.round(st?.mtimeMs ?? 0)); +} + +export async function summariseSweep(ctx) { + const { dir, name } = ctx; + const names = await readdir(dir).catch(() => []); + const reportName = names.find((n) => SWEEP_RE.test(n)); + const file = reportName ? path.join(dir, reportName) : null; + const [text, st] = await Promise.all([ + file ? readFile(file, "utf8").catch(() => null) : null, + file ? stat0(file) : null, + ]); + + // A citation in these reports is a markdown link carrying a `?v=` moment. + // Counting them is the one number worth putting on the card: it is how much + // work a manifest would be. + const citations = text ? (text.match(/\]\([^)]*[?&]v=/g) ?? []).length : 0; + const title = text?.match(/^#\s+(.+)$/m)?.[1]?.trim() ?? name; + + const facts = []; + if (citations) facts.push(`${citations} citation${citations === 1 ? "" : "s"}`); + facts.push("no manifest"); + + return { + title, + subtitle: "a report, not yet a video", + state: "draft", + newestMtimeMs: st?.mtimeMs ?? 0, + facts, + flags: [], + posterRel: null, + haystack: [ctx.id, title, reportName].filter(Boolean).join(" ").toLowerCase(), + reportName, + }; +} diff --git a/umtool/lib/projects/walk.mjs b/umtool/lib/projects/walk.mjs @@ -0,0 +1,195 @@ +// The folder walk. +// +// Two rules do almost all the work here: +// +// A PROJECT IS A LEAF. Detection stops the descent, which is what keeps out/ +// (1,210 files, 3.1 GB across ~/reports) out of the walk entirely. Nothing +// here ever sees a clip, a segment, a card PNG or a variant. +// +// A FOLDER WITH NO PROJECT BENEATH IT DOES NOT EXIST. That is what silently +// drops ~40 loose test directories under quartering-uh-song -- alarm-tests, +// chop-tests, run-visual-tests, sfx, pipeline -- with no denylist to maintain +// and nothing to update when the 41st appears. +// +// Measured cost on the real tree: ~25 readdirs, no stats of anything inside a +// project. The 3.1 GB is never touched. +import { readdir, readFile, realpath, stat } from "node:fs/promises"; +import path from "node:path"; +import { RESERVED_BROWSE, SKIP_DIRS, detectKind } from "./kinds.mjs"; + +/** A single safe path segment: no separators, no traversal, no dotfiles. */ +export const isSegment = (v) => /^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(v) && !v.includes(".."); + +/** + * Four levels from REPORTS_ROOT. `quartering-uh-song/videos/yoshi` is two, so + * this is two levels of headroom and a cheap guard against an accident -- a + * symlink into a home directory, say -- turning the index into a filesystem + * crawl. + */ +export const MAX_DEPTH = 4; + +/** + * How a project is reachable. + * + * Both failures were SILENT before. A directory whose name isSegment() dislikes + * (a space is enough) simply vanished from the listing; a directory called + * `find` was listed with a link that rendered the phrase console instead. Each + * is now a state the project carries, an entry in the index and a decision -- + * never a disappearance. + */ +export function routingFor(id) { + const segs = id.split("/"); + if (!segs.every(isSegment)) return "unroutable"; + if (RESERVED_BROWSE.includes(segs[0])) return "shadowed"; + return "ok"; +} + +async function readDeclared(dir, names) { + if (!names.has("project.json")) return null; + try { + return JSON.parse(await readFile(path.join(dir, "project.json"), "utf8")); + } catch { + return null; + } +} + +/** + * Every project under `root`, and the folders that contain them. + * + * Symlinked directories ARE followed -- a project symlinked into the tree is a + * reasonable thing to do -- but every real path is visited once, so a link that + * points at an ancestor terminates instead of spinning. + */ +export async function walkProjects(root, { maxDepth = MAX_DEPTH } = {}) { + const projects = []; + const visited = new Set(); + + const visit = async (abs, rel, depth) => { + let real; + try { + real = await realpath(abs); + } catch { + return; + } + if (visited.has(real)) return; + visited.add(real); + + let entries; + try { + entries = await readdir(abs, { withFileTypes: true }); + } catch { + return; + } + + const names = new Set(entries.map((e) => e.name)); + const hit = detectKind(names, await readDeclared(abs, names)); + if (hit) { + const id = rel; + projects.push({ + id, + dir: abs, + name: path.basename(abs), + folder: path.posix.dirname(id) === "." ? "" : path.posix.dirname(id), + kind: hit.kind, + template: hit.template, + routing: routingFor(id), + ...(hit.ambiguousWith ? { ambiguousWith: hit.ambiguousWith } : {}), + ...(hit.unknownKind ? { unknownKind: true } : {}), + }); + return; // a project is a leaf + } + + if (depth >= maxDepth) return; + + for (const e of entries) { + if (e.name.startsWith(".")) continue; + if (SKIP_DIRS.has(e.name)) continue; + let isDir = e.isDirectory(); + if (!isDir && e.isSymbolicLink()) { + isDir = await stat(path.join(abs, e.name)).then((s) => s.isDirectory(), () => false); + } + if (!isDir) continue; + await visit(path.join(abs, e.name), rel ? `${rel}/${e.name}` : e.name, depth + 1); + } + }; + + await visit(root, "", 0); + projects.sort((a, b) => a.id.localeCompare(b.id)); + return projects; +} + +// --------------------------------------------------------------------------- +// Folders, derived from the project paths rather than recorded during the walk. +// +// Deriving them is what makes "a folder with no project beneath it is invisible" +// true by construction rather than by a filter somebody has to remember. +// --------------------------------------------------------------------------- +export function foldersFor(projects) { + const nodes = new Map(); + const node = (p) => { + let n = nodes.get(p); + if (!n) { + n = { + path: p, + label: p === "" ? "" : p.split("/").pop(), + collapsedFrom: [], + projects: [], + children: [], + }; + nodes.set(p, n); + } + return n; + }; + node(""); + + for (const pr of projects) { + const parts = pr.id.split("/").slice(0, -1); + let acc = ""; + node("").children; + for (const seg of parts) { + const parent = acc; + acc = acc ? `${acc}/${seg}` : seg; + node(acc); + const pn = node(parent); + if (!pn.children.includes(acc)) pn.children.push(acc); + } + node(acc).projects.push(pr.id); + } + return nodes; +} + +/** + * Collapse pass-through folders FOR DISPLAY. + * + * `quartering-uh-song` holds no projects and exactly one child that matters + * (`videos`), so the index shows one heading, `quartering-uh-song / videos`. + * + * The URL is never collapsed -- /browse/quartering-uh-song/videos/yoshi stays + * the one true address. A URL has to mean the same thing in six weeks, and a + * display convenience is not allowed to decide what a link is. + */ +export function collapseFolders(nodes) { + const out = new Map(nodes); + let changed = true; + while (changed) { + changed = false; + for (const [p, n] of [...out]) { + if (p === "") continue; + if (n.projects.length !== 0 || n.children.length !== 1) continue; + const childPath = n.children[0]; + const child = out.get(childPath); + if (!child) continue; + child.label = `${n.label} / ${child.label}`; + child.collapsedFrom = [...n.collapsedFrom, n.path]; + // Re-parent: whoever pointed at n now points at the child. + for (const other of out.values()) { + const i = other.children.indexOf(p); + if (i >= 0) other.children[i] = childPath; + } + out.delete(p); + changed = true; + break; + } + } + return out; +} diff --git a/umtool/lib/report/driver.mjs b/umtool/lib/report/driver.mjs @@ -0,0 +1,145 @@ +// Turning a report project into a chain of steps lib/jobs.ts can run. +// +// SPAWN, not import. A 40-minute chain of yt-dlp and ffmpeg inside a request +// handler has no cancellation story, its execFile buffers live in the server's +// heap, and a runaway grandchild outlives the request that started it. The +// scripts are given a progress protocol instead (`--progress ndjson`), so the +// UI reads events rather than scraping prose. +// +// The client sends a PROJECT and a PRESET NAME. It never sends a path, an argv +// or an env map -- the same contract /api/browse/build already keeps. +import path from "node:path"; + +/** Where the pipeline lives. One place, so a move is one edit. */ +export const PIPELINE_DIR = path.resolve(process.cwd(), "..", "scripts", "report-to-video"); + +const script = (name) => path.join(PIPELINE_DIR, name); + +/** + * Presets, in the order somebody actually works. + * + * `preview` is for looking at ONE clip after moving its edges; `fast` is hard + * cuts over the whole timeline, which is minutes rather than tens of minutes and + * is what you watch to check the argument; `final` is the deliverable. + */ +export const PRESETS = { + preview: { label: "preview one clip", xfade: false, chapters: false, only: true }, + fast: { label: "fast pass (hard cuts)", xfade: false, chapters: true, only: false }, + final: { label: "final", xfade: true, chapters: true, only: false }, +}; + +/** A build's timeout, scaled to the work rather than to a global guess. */ +export const buildTimeoutMs = (clipCount, xfade) => + Math.max(15 * 60_000, clipCount * (xfade ? 120_000 : 60_000)); + +/** + * The chain. + * + * Step 1 is the availability preflight and it is a STEP, not a preamble: it is + * the one fact about a manifest that goes stale in both directions, it costs + * seconds, and without it a dead source is discovered twenty minutes and a + * dozen paid-for fetches into the build. + * + * Step 2 runs resolve-windows DRY. If it reports changes, the chain stops and + * shows them -- a widener silently rewriting windows somebody just set in the + * bench is exactly the surprise `lock` exists to prevent. Applying is a second, + * explicit action. + */ +/** + * @param {{ id: string, dir: string }} project + * @param {{ preset?: string, only?: string | null, skipFetch?: boolean, env?: Record<string,string>, clipCount?: number }} [opts] + * @returns {import("../trim").Step[]} + */ +export function buildSteps(project, { preset = "fast", only = null, skipFetch = false, env = {}, clipCount = 20 } = {}) { + const p = PRESETS[preset] ?? PRESETS.fast; + const manifest = path.join(project.dir, "video.manifest.json"); + const outDir = path.join(project.dir, "out"); + const base = { cwd: PIPELINE_DIR, env }; + + const steps = [ + { + ...base, + label: "check every source is still fetchable", + argv: ["node", script("check-availability.mjs"), manifest, "--out", outDir], + timeoutMs: 10 * 60_000, + }, + { + ...base, + label: "resolve windows (dry — nothing is written)", + argv: ["node", script("resolve-windows.mjs"), manifest], + timeoutMs: 5 * 60_000, + }, + ]; + + const buildArgv = [ + "node", + script("build-video.mjs"), + manifest, + "--out", + outDir, + "--progress", + "ndjson", + "--continue-on-error", + ]; + if (!p.xfade) buildArgv.push("--no-xfade"); + if (!p.chapters) buildArgv.push("--no-chapters"); + if (skipFetch) buildArgv.push("--skip-fetch"); + if (p.only && only) buildArgv.push("--only", only); + + steps.push({ + ...base, + label: p.label, + argv: buildArgv, + ndjson: true, + timeoutMs: buildTimeoutMs(clipCount, p.xfade), + }); + + // A build can exit 0 and still be wrong: a concat that produced nothing, a + // chapter pass that dropped markers, a timeline that lost a clip because + // --continue-on-error let it. Each looks like success at the terminal. + // + // Skipped for a one-clip preview, which deliberately does not produce a + // deliverable to measure. + if (!p.only || !only) { + steps.push({ + ...base, + label: "verify the file that came out", + argv: ["node", script("verify-build.mjs"), manifest, "--out", outDir], + timeoutMs: 5 * 60_000, + }); + } + + return steps; +} + +/** + * Fetch ONE clip's window, wide. What the bench's "fetch more" runs. + * @param {{ dir: string }} project + * @param {string} clipId + * @param {number} pad + * @returns {import("../trim").Step[]} + */ +export function fetchSteps(project, clipId, pad) { + return [ + { + cwd: PIPELINE_DIR, + env: {}, + label: `fetch ${clipId} with ${pad}s of pad`, + argv: [ + "node", + script("build-video.mjs"), + path.join(project.dir, "video.manifest.json"), + "--out", + path.join(project.dir, "out"), + "--fetch-only", + clipId, + "--pad", + String(pad), + "--progress", + "ndjson", + ], + ndjson: true, + timeoutMs: 10 * 60_000, + }, + ]; +} diff --git a/umtool/lib/report/manifest.mjs b/umtool/lib/report/manifest.mjs @@ -0,0 +1,270 @@ +// Writing a window back into video.manifest.json. +// +// This app is a SECOND writer of a file resolve-windows.mjs also writes, and an +// agent running `umtool` is a third. Four rules follow from that, and each of +// them is here because getting it wrong is silent: +// +// 1. ROUND TO 2 dp. resolve-windows.mjs is a fixed point, and both its EPS +// lookup tolerance and its 0.05 s deadband assume 2 dp storage. Writing 4 dp +// makes the widener grow the same clip a little on every subsequent run -- +// the exact bug its own comments document. +// +// 2. PRESERVE THE CLI'S FORMATTING. It writes `JSON.stringify(m, null, 2)` plus +// a trailing newline. lib/state.ts's writeJsonAtomic uses indent 1, which +// would turn a two-number edit into a 600-line diff and make the next real +// change unreviewable. +// +// 3. TMP + RENAME, under the process-wide state lock. A reader must never see +// half a manifest, and two requests must not interleave a read-modify-write. +// +// 4. GUARD ON THE FILE'S OWN MTIME. A PUT carrying a stale token is a 409, not +// a silent overwrite -- somebody may have run `resolve-windows --write` in +// between, and losing that is losing human judgement. +import { copyFile, readFile, rename, stat, writeFile } from "node:fs/promises"; +import path from "node:path"; +import { + POPULATIONS, + SCOPES, + SCOPE_CONFIDENCE, + VALUE_KINDS, + rolesGaps, +} from "report-to-video/ledger-totals"; + +// Its own write queue, not lib/state.ts's. +// +// Two reasons, and the second is the real one. lib/state.ts is TypeScript, so +// importing it would stop `umtool window` running under plain node -- and the +// whole point of one writer is that the CLI and the app go through it. And the +// scope is genuinely different: lib/state's queue serialises the SONG state +// files, which have nothing to do with a manifest. +// +// Within a process this serialises read-modify-write. ACROSS processes -- an +// agent running the CLI while the app has a page open -- the guard is the +// tmp+rename plus the mtime token, which is what actually stops a lost update. +/** @type {Promise<unknown>} */ +let queue = Promise.resolve(); +/** + * @template T + * @param {() => Promise<T>} fn + * @returns {Promise<T>} + */ +function withManifestLock(fn) { + const run = queue.then(fn, fn); + queue = run.then( + () => undefined, + () => undefined, + ); + return run; +} + +export const MANIFEST_NAME = "video.manifest.json"; +const manifestFile = (dir) => path.join(dir, MANIFEST_NAME); + +/** 2 dp, and never NaN. The one number format this file will write. */ +const round2 = (n) => Number(Number(n).toFixed(2)); + +/** + * The mtime a client must hand back to be allowed to write. + * @param {string} dir + * @returns {Promise<string | null>} + */ +export async function manifestToken(dir) { + const st = await stat(manifestFile(dir)).catch(() => null); + return st ? String(Math.round(st.mtimeMs)) : null; +} + +const serialise = (m) => JSON.stringify(m, null, 2) + "\n"; + +/** How long to leave between .bak copies of the same manifest. */ +const BAK_INTERVAL_MS = 10 * 60 * 1000; + +async function backupOnce(file) { + // A rolling stack of backups is worth less than one copy of the last + // hand-authored state, which is the precedent ferret-rescue already set by + // having a single video.manifest.json.bak beside it. + const bak = `${file}.bak`; + const [src, dst] = await Promise.all([ + stat(file).catch(() => null), + stat(bak).catch(() => null), + ]); + if (!src) return; + if (dst && src.mtimeMs - dst.mtimeMs < BAK_INTERVAL_MS) return; + await copyFile(file, bak).catch(() => {}); +} + +async function writeManifestAtomic(dir, manifest) { + const file = manifestFile(dir); + await backupOnce(file); + const tmp = `${file}.tmp-${process.pid}-${Math.random().toString(36).slice(2, 8)}`; + await writeFile(tmp, serialise(manifest), "utf8"); + await rename(tmp, file); + return manifestToken(dir); +} + +export class StaleToken extends Error { + constructor(expected, got) { + super(`the manifest changed since you read it (${got} vs ${expected})`); + this.name = "StaleToken"; + this.expected = expected; + this.got = got; + } +} + +const WINDOW_FIELDS = ["start", "end"]; +const FLAG_FIELDS = ["lock", "lockStart", "lockEnd"]; + +/** + * Patch ONE clip. Windows and locks only. + * + * Deliberately not a general editor: re-ordering is a different operation with + * different consequences (it has to recompute `sectionEnter`), and letting a + * window save quietly move an entry is how a cut changes without anybody + * deciding to change it. + */ +/** + * @param {string} dir + * @param {string} clipId + * @param {Record<string, unknown>} patch + * @param {{ token?: string | null }} [opts] + * @returns {Promise<{ entry: Record<string, unknown>, before: {start:number,end:number}, token: string | null }>} + */ +export async function updateClip(dir, clipId, patch, { token = null } = {}) { + return withManifestLock(async () => { + const current = await manifestToken(dir); + if (token !== null && current !== token) throw new StaleToken(token, current); + + // Re-read INSIDE the lock, every time. Nothing is held across requests. + const raw = await readFile(manifestFile(dir), "utf8"); + const manifest = JSON.parse(raw); + const entry = (manifest.timeline ?? []).find((e) => e.id === clipId); + if (!entry) throw new Error(`no timeline entry with id ${clipId}`); + // `!== "clip"`, not `=== "card"`. The timeline's vocabulary is open -- one + // real manifest carries `scroll` and `chart` entries -- and the card-only + // check would have let a window be written onto one of those. + if (entry.type !== "clip") throw new Error(`${clipId} is a ${entry.type ?? "non-clip"} entry, not a clip`); + + const before = { start: entry.start, end: entry.end }; + for (const k of WINDOW_FIELDS) { + if (patch[k] === undefined) continue; + const v = Number(patch[k]); + if (!Number.isFinite(v) || v < 0) throw new Error(`${k} must be a number ≥ 0`); + entry[k] = round2(v); + } + if (entry.end - entry.start < 0.5) { + throw new Error(`a clip must be at least half a second (${entry.start}–${entry.end})`); + } + for (const k of FLAG_FIELDS) { + if (patch[k] === undefined) continue; + // `false` REMOVES the key rather than writing it. The manifests are read + // by humans, and `"lockEnd": false` is noise that reads like a decision. + if (patch[k]) entry[k] = true; + else delete entry[k]; + } + if (patch.note !== undefined) { + if (patch.note) entry.note = String(patch.note); + else delete entry.note; + } + + const nextToken = await writeManifestAtomic(dir, manifest); + return { entry, before, token: nextToken }; + }); +} + + +// --------------------------------------------------------------------------- +// Writing an ADJUDICATION back into a ledger entry. +// +// Separate from updateClip() on purpose. A clip edit moves a window; a claim +// edit records a RULING on what a sentence meant, and the two have nothing in +// common but the file they land in. Sharing a function would mean one of them +// could quietly write the other's fields. +// +// Every value is checked against the vocabulary ledger-totals.mjs publishes, +// imported rather than restated -- a page offering a seventh population that +// the arithmetic has never heard of is exactly the silent divergence this whole +// module exists to prevent. +// --------------------------------------------------------------------------- + +const CLAIM_ENUMS = { + scope: SCOPES, + scopeConfidence: SCOPE_CONFIDENCE, + population: POPULATIONS, + valueKind: VALUE_KINDS, +}; + +/** + * Patch ONE ledger claim's six adjudication fields. + * + * @param {string} dir + * @param {string} claimId + * @param {Record<string, unknown>} patch + * @param {{ token?: string | null }} [opts] + */ +export async function updateClaim(dir, claimId, patch, { token = null } = {}) { + return withManifestLock(async () => { + const current = await manifestToken(dir); + if (token !== null && current !== token) throw new StaleToken(token, current); + + const raw = await readFile(manifestFile(dir), "utf8"); + const manifest = JSON.parse(raw); + const entry = (manifest.ledger ?? []).find((e) => e.id === claimId); + if (!entry) throw new Error(`no ledger claim with id ${claimId}`); + + for (const [field, allowed] of Object.entries(CLAIM_ENUMS)) { + if (patch[field] === undefined) continue; + const v = String(patch[field]); + if (!allowed.includes(v)) { + throw new Error(`${field} must be one of ${allowed.join(", ")} (got \`${v}\`)`); + } + entry[field] = v; + } + + if (patch.scopeBasis !== undefined) { + // The phrase from the quote that settles it. Refused when blank: an + // adjudication with no basis is an opinion, and a reviewer cannot check + // an opinion against the audio. + const v = String(patch.scopeBasis ?? "").trim(); + if (!v) throw new Error("scopeBasis must quote the phrase that settles the scope"); + entry.scopeBasis = v; + } + + if (patch.flags !== undefined) { + if (!Array.isArray(patch.flags)) throw new Error("flags must be an array"); + const list = patch.flags.map((f) => String(f).trim()).filter(Boolean); + // Always WRITTEN, even empty. `flags` is one of the six fields the gate + // checks, so an absent array reads as "nobody has looked" -- which is + // exactly the state an adjudication is supposed to leave behind. + entry.flags = list; + } + + // ---- the roster ---- + // Optional, and NOT one of the six: most claims are a number and nothing + // else, and gating the inbox on a field only a handful of entries can carry + // would leave it permanently red. But when it IS written it is a ruling + // like any other -- who he named, how many of them, and his words for it -- + // so it is checked here rather than trusted. + if (patch.roles !== undefined) { + if (patch.roles === null || (Array.isArray(patch.roles) && !patch.roles.length)) { + delete entry.roles; + } else { + if (!Array.isArray(patch.roles)) throw new Error("roles must be an array"); + const list = patch.roles.map((r) => ({ + role: String(r?.role ?? "").trim(), + count: Number(r?.count), + verbatim: String(r?.verbatim ?? "").trim(), + })); + const bad = rolesGaps(list); + if (bad.length) throw new Error(`roles: ${bad.join(", ")}`); + entry.roles = list; + } + } + + if (patch.note !== undefined) { + if (patch.note) entry.note = String(patch.note); + else delete entry.note; + } + + const nextToken = await writeManifestAtomic(dir, manifest); + return { entry, token: nextToken }; + }); +} diff --git a/umtool/lib/report/serve.mjs b/umtool/lib/report/serve.mjs @@ -0,0 +1,68 @@ +// Resolving what a clip-bench request is allowed to open. +// +// Nothing here takes a path. A request names a PROJECT and a CLIP, both of which +// must be members of the current scan, and a `file` -- which must be one of the +// cached windows the server itself finds for that clip's video. So a plausible +// name that is simply not there fails, and a traversal fails twice: once on the +// membership check and once on resolveInRoots. +import path from "node:path"; +import { REPORTS_ROOT, resolveInRoots } from "../paths.mjs"; +import { walkProjects } from "../projects/walk.mjs"; +import { cachedWindowsFor } from "report-to-video/build-video"; +import { clipsOf, readManifest } from "../projects/report.mjs"; + +export async function resolveClip(projectId, clipId) { + const projects = await walkProjects(REPORTS_ROOT); + const project = projects.find((p) => p.id === projectId); + if (!project) return { error: "no such project", status: 404 }; + const manifest = await readManifest(project.dir); + if (!manifest) return { error: "no manifest", status: 404 }; + const clip = clipsOf(manifest).find((e) => e.id === clipId); + if (!clip) return { error: "no such clip", status: 404 }; + return { project, manifest, clip }; +} + +/** + * The cached source windows for a clip, widest first. + * + * The bench wants the WIDEST containing file, because that is how much room + * there is to drag before anything has to be fetched. The BUILD wants the + * tightest, because it decodes the whole file to find a silence. They are + * different questions and both are asked here. + */ +export async function windowsFor(project, clip) { + const rawDir = path.join(project.dir, "out", "clips-raw"); + const all = await cachedWindowsFor(rawDir, clip.video); + return all.sort((a, b) => b.to - b.from - (a.to - a.from)); +} + +export function pickWindow(windows, wantedName) { + if (wantedName) { + const hit = windows.find((w) => w.name === wantedName); + return hit ?? null; + } + return windows[0] ?? null; +} + +/** Absolute, inside a read root, and a member of that clip's own scan. */ +export function absOf(win) { + return win ? resolveInRoots(win.path) : null; +} + +/** + * The same membership rule, for a LEDGER claim. + * + * A claim id is checked against the ledger of the project named in the request, + * exactly as a clip id is checked against its timeline. Nothing here takes a + * path either. + */ +export async function resolveClaim(projectId, claimId) { + const projects = await walkProjects(REPORTS_ROOT); + const project = projects.find((p) => p.id === projectId); + if (!project) return { error: "no such project", status: 404 }; + const manifest = await readManifest(project.dir); + if (!manifest) return { error: "no manifest", status: 404 }; + const claim = (manifest.ledger ?? []).find((e) => e.id === claimId); + if (!claim) return { error: "no such claim", status: 404 }; + return { project, manifest, claim }; +} diff --git a/umtool/lib/trim.ts b/umtool/lib/trim.ts @@ -180,7 +180,23 @@ export async function writeTrims(set: TrimSet, entries: TrimEntry[]): Promise<Tr // half-finished run cannot corrupt the directory the shipped overlays name. // --------------------------------------------------------------------------- -export type Step = { label: string; argv: string[]; cwd: string; env: Record<string, string> }; +export type Step = { + label: string; + argv: string[]; + cwd: string; + env: Record<string, string>; + /** + * How long this step may run, overriding the default. + * + * Per-step because the default is 15 minutes and exists to catch the + * accidental hour-long job -- while a 19-clip crossfaded build legitimately + * runs 20 to 40. Raising the default to fit the build would remove the guard + * for everything else, so the build asks for what it needs instead. + */ + timeoutMs?: number; + /** Lines matching this are progress events, not log noise. */ + ndjson?: boolean; +}; /** Env keys a request may never set. The recipe owns both, for the reason above. */ export const ENV_DENY = ["HOOKDIR", "TAKE_DIR"]; diff --git a/umtool/lib/utils.ts b/umtool/lib/utils.ts @@ -0,0 +1,14 @@ +import { clsx, type ClassValue } from "clsx"; +import { twMerge } from "tailwind-merge"; + +/** + * The class merger shadcn's variants are written against. + * + * twMerge is the part that earns its place: it resolves CONFLICTS by Tailwind's + * own semantics, so a variant's `px-2` and a caller's `px-4` do not both end up + * in the class list with the winner decided by stylesheet order. Without it, + * every override has to be a longer selector or a `!`. + */ +export function cn(...inputs: ClassValue[]): string { + return twMerge(clsx(inputs)); +} diff --git a/umtool/next.config.ts b/umtool/next.config.ts @@ -11,6 +11,13 @@ const nextConfig: NextConfig = { // the suite its own is enough to let both run. Judging clips and running the // tests at the same time is the normal case here, not an edge one. distDir: process.env.NEXT_DIST_DIR ?? ".next", + // lmdb is a NATIVE module and must be required at runtime, not bundled. + // + // Bundling it makes Turbopack try to resolve `moduleRequire('cbor-x')` -- an + // OPTIONAL dependency lmdb only reaches when an encoding asks for it -- and + // fail the whole module graph. Every page importing lib/projects then 500s + // with "Can't resolve 'cbor-x'", which names a package nothing here uses. + serverExternalPackages: ["lmdb"], turbopack: { // Same reasoning as editor/next.config.ts: Turbopack infers the workspace // root by walking up for the outermost lockfile, and a stray pnpm-lock.yaml diff --git a/umtool/package.json b/umtool/package.json @@ -11,9 +11,14 @@ "e2e": "node ../scripts/queue-lock.mjs --ports UMTOOL_E2E_PORT:3051 -- playwright test" }, "dependencies": { + "class-variance-authority": "^0.7.1", + "clsx": "^2.1.1", + "lmdb": "^3.5.4", "next": "16.2.3", "react": "19.2.4", "react-dom": "19.2.4", + "report-to-video": "workspace:*", + "tailwind-merge": "^3.6.0", "yt-dlp-transcript-common": "workspace:*" }, "devDependencies": { @@ -24,5 +29,8 @@ "@types/react-dom": "^19.2.3", "tailwindcss": "^4.2.2", "typescript": "^5.9.3" + }, + "bin": { + "umtool": "bin/umtool.mjs" } } diff --git a/umtool/playwright.config.ts b/umtool/playwright.config.ts @@ -33,6 +33,14 @@ export default defineConfig({ command: `node e2e/fixtures/make-fixture.mjs ${FIXTURE} && ` + `SONG_CODE_DIR=${FIXTURE}/code SONG_DIR=${FIXTURE}/data SONG_REPORTS_DIR=${FIXTURE}/reports ` + + // The project walk reads REPORTS_ROOT, which defaults to + // dirname(SONG_REPORTS_DIR) -- so the fixture is confined with no new env + // var. CHANNELS_DIR has to be said explicitly: it is where a report + // video's cue files live, and its default is the real 3 GB corpus. + `CHANNELS_DIR=${FIXTURE}/channels ` + + // Stub binaries, so a build spec is offline and deterministic. The + // pipeline already reads both as overrides; the fixture writes them. + `YTDLP_BIN=${FIXTURE}/bin/yt-dlp QRENCODE_BIN=${FIXTURE}/bin/qrencode ` + `NEXT_DIST_DIR=.next-e2e pnpm exec next dev --port ${PORT}`, port: PORT, reuseExistingServer: false,