Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 142106ad889f50dd05f1738b421fc42194d4e1fb
parent 5f3c62f4957b9b2af03cfcdc40bac7af39738a5a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 21 Sep 2026 02:31:00 -0400

Merge branch 'main' into storage/debts-1

Diffstat:
Aumtool/app/api/report/deliver/route.ts | 180+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/bin/cut-from-cache.mjs | 61+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/bin/share-batch.mjs | 45+++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/projects/DeliverActions.tsx | 257+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/projects/DeliverSection.tsx | 265+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/components/projects/ReportProject.tsx | 7+++++++
Mumtool/docs/clip-bench.md | 73+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/e2e/deliver.spec.ts | 294+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/e2e/fixtures/make-fixture.mjs | 207+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/report/cut.mjs | 152+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/report/deliver.mjs | 421+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/lib/report/driver.mjs | 129+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/report/encode.mjs | 83+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/playwright.config.ts | 4++++
Mumtool/report-to-video/build-video.mjs | 31+++++++++++++++++++++++++++++--
15 files changed, 2207 insertions(+), 2 deletions(-)

diff --git a/umtool/app/api/report/deliver/route.ts b/umtool/app/api/report/deliver/route.ts @@ -0,0 +1,180 @@ +import { cancelJob, getJob, jobView, runningJob, startJob } from "@/lib/jobs"; +import { + applyRulingsSteps, + cutSteps, + rebuildReportSteps, + shareBatchSteps, +} from "@/lib/report/driver.mjs"; +import { deliverStateOf } from "@/lib/report/deliver.mjs"; +import { readClipDetail } from "@/lib/projects/report.mjs"; +import { projectRef } from "@/lib/projects"; +import type { Step } from "@/lib/trim"; + +export const dynamic = "force-dynamic"; + +// DELIVERING a walked report: the four things that happen after the last clip +// is judged. +// +// cut the confirmed clips that have no mp4 yet, out of the cache +// share a batch of those, in three encodes, for whoever is writing +// apply the project's own apply-manifest.py, then `umtool corrections` +// rebuild build.py, once per content variant present +// +// Each is a JOB, and jobs.ts allows exactly one at a time. That is not a +// limitation to work around here: every one of these reads or writes the same +// clips/ directory that a bench fetch writes into, and two of them at once +// would race over the same file. A second request gets 409 with the name of +// what is running, in the same words /api/report/build uses. +// +// The client sends a project id, an action, and at most a batch name. Never a +// path and never an argv: the step list is built server-side from the +// directory's own contents, which is what stops "rebuild" from being able to +// run an arbitrary python file. + +const ACTIONS = ["cut", "share", "apply", "rebuild"] as const; +type Action = (typeof ACTIONS)[number]; + +export async function GET(request: Request) { + const url = new URL(request.url); + const headers = { "cache-control": "no-store" }; + const id = url.searchParams.get("job"); + if (id) { + const job = getJob(id); + if (!job) return Response.json({ error: "no such job" }, { status: 404, headers }); + return Response.json(jobView(job, Number(url.searchParams.get("since") ?? 0)), { headers }); + } + + // "Is anything running" is the ONLY question the panel asks on mount, and it + // is not worth a second full read of the project: the state below re-reads + // every cue file, every share folder and every content module, and the + // server component beside it has just done exactly that. The job registry is + // process-wide, so this needs no project at all -- the caller compares the + // job's own `project` field. + if (url.searchParams.get("running") === "1") { + const r = runningJob(); + return Response.json({ running: r ? jobView(r) : null }, { headers }); + } + + const projectId = url.searchParams.get("project") ?? ""; + const project = await projectRef(projectId); + if (!project) return Response.json({ error: "no such project" }, { status: 404, headers }); + const detail = await readClipDetail(project.dir); + if (!detail) return Response.json({ error: "no manifest" }, { status: 400, headers }); + const state = await deliverStateOf(project, { + manifest: detail.manifest, + entries: detail.entries, + }); + const running = runningJob(); + return Response.json({ ...state, running: running ? jobView(running) : null }, { headers }); +} + +export async function POST(request: Request) { + const url = new URL(request.url); + const cancel = url.searchParams.get("cancel"); + // Stop ABANDONS THE REST, it does not undo what finished. Every artefact + // here is content-addressed by clip id, so a cancelled cut is a paused one: + // pressing the button again skips the clips that already have a file. + if (cancel) return Response.json({ cancelled: cancelJob(cancel) }); + + const body = (await request.json().catch(() => ({}))) as Record<string, unknown>; + const action = String(body.action ?? "") as Action; + if (!ACTIONS.includes(action)) { + return Response.json({ error: `action must be one of ${ACTIONS.join(", ")}` }, { status: 400 }); + } + + const project = await projectRef(String(body.project ?? "")); + if (!project) return Response.json({ error: "no such project" }, { status: 404 }); + const detail = await readClipDetail(project.dir); + if (!detail) return Response.json({ error: "no manifest" }, { status: 400 }); + const state = await deliverStateOf(project, { + manifest: detail.manifest, + entries: detail.entries, + }); + if (!state) return Response.json({ error: "no manifest" }, { status: 400 }); + + let steps: Step[] = []; + let kind = ""; + + if (action === "cut") { + // The list is the SERVER's: confirmed, no mp4, and a window on this disk + // that holds it. A client-supplied list could name a clip nobody judged. + const ids = state.needCut.map((c: { id: string }) => c.id); + if (!ids.length) { + return Response.json( + { error: "every confirmed clip already has a file — nothing to cut", ok: false }, + { status: 400 }, + ); + } + steps = cutSteps(project, ids); + kind = `cut ${project.id} (${ids.length} clips)`; + } else if (action === "share") { + const name = String(body.name ?? state.nextName); + if (!/^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(name)) { + return Response.json( + { error: "a batch name is letters, digits, dot, dash and underscore" }, + { status: 400 }, + ); + } + if (!state.candidates.length) { + return Response.json( + { error: "nothing to ship: every confirmed clip is already shared, or not cut yet" }, + { status: 400 }, + ); + } + steps = shareBatchSteps(project, name); + kind = `share-${name} ${project.id} (${state.candidates.length} clips)`; + } else if (action === "apply") { + if (!state.hasApplyScript) { + return Response.json( + { error: "this project has no apply-manifest.py — there is nothing to fold back into" }, + { status: 400 }, + ); + } + // REFUSED WHILE THE WALK IS UNFINISHED, and this is the one guard here + // that is about judgement rather than about files. apply-manifest.py syncs + // clips.json from the manifest and deletes the mp4 of every clip whose + // window moved; running it over a half-walked cut bakes "nobody has looked + // at this yet" into the deliverable as though it were a verdict. `partial` + // is how somebody says they meant it. + if (state.review.unreviewed > 0 && !body.partial) { + return Response.json( + { + error: + `${state.review.unreviewed} of ${state.review.total} clips have not been judged — ` + + "finish the walk, or tick “apply partial” to fold back what there is", + unreviewed: state.review.unreviewed, + needsPartial: true, + }, + { status: 409 }, + ); + } + steps = applyRulingsSteps(project); + kind = `apply rulings ${project.id}`; + } else { + if (!state.hasBuildScript || !state.variants.length) { + return Response.json( + { error: "this project has no build.py and content*.py to render" }, + { status: 400 }, + ); + } + steps = rebuildReportSteps(project, state.variants); + kind = `rebuild ${project.id} (${state.variants.length} variants)`; + } + + const running = runningJob(); + if (running) { + return Response.json( + { error: `a job is already running (${running.kind})`, running: jobView(running) }, + { status: 409 }, + ); + } + try { + const job = startJob(kind, steps, { project: project.id }); + return Response.json( + { ok: true, action, job: jobView(job) }, + { headers: { "cache-control": "no-store" } }, + ); + } catch (e) { + return Response.json({ error: e instanceof Error ? e.message : String(e) }, { status: 409 }); + } +} diff --git a/umtool/bin/cut-from-cache.mjs b/umtool/bin/cut-from-cache.mjs @@ -0,0 +1,61 @@ +#!/usr/bin/env node +// Cut ONE clip out of the window already on disk. +// +// node bin/cut-from-cache.mjs --project <id> --clip <id> [--reencode|--no-reencode] +// +// A CLI so the Deliver panel's "cut the confirmed clips" can be ONE job with +// one step per clip: lib/jobs.ts then owns the sequencing, the `k of n` the UI +// reads off stepIndex, the timeout, and a Stop that kills the process group +// rather than a promise nobody can interrupt. A loop inside a request handler +// would have had to reinvent all four. +// +// Never fetches. A clip with no containing window exits 3 and says so, which +// is a fact about the cache and not a failure of the cut. +import process from "node:process"; +import { resolveProject } from "../lib/projects/core.mjs"; +import { cutClipFromCache } from "../lib/report/cut.mjs"; + +const argv = process.argv.slice(2); +const val = (flag) => { + const i = argv.indexOf(flag); + return i < 0 ? null : argv[i + 1]; +}; + +const projectArg = val("--project"); +const clipId = val("--clip"); +if (!projectArg || !clipId) { + console.error("usage: cut-from-cache.mjs --project <id> --clip <id> [--reencode|--no-reencode]"); + process.exit(2); +} + +const r = await resolveProject(projectArg); +if (!r.project) { + console.error(`no project matches "${projectArg}"`); + process.exit(2); +} + +const reencode = argv.includes("--reencode") + ? "always" + : argv.includes("--no-reencode") + ? "never" + : "auto"; + +// TEST-ONLY PACING, and the only reason it exists: a cut of a cached window is +// about a second of ffmpeg, which is too fast for a spec to press Stop in the +// middle of -- and "Stop abandons the REST, and the next run resumes rather +// than re-cutting" is the whole argument for one step per clip. The e2e +// fixture's server sets this; nothing else ever should. +const delay = Number(process.env.UMTOOL_CUT_DELAY_MS ?? 0); +if (delay > 0) await new Promise((r) => setTimeout(r, delay)); + +const res = await cutClipFromCache(r.project, clipId, { reencode }); +if (!res.ok) { + console.error(`${clipId}: ${res.error}`); + // 3 is "nothing on disk holds this clip", which the panel already lists + // under its own heading with a fetch link. Every other failure is a 1. + process.exit(res.reason === "not-fetched" ? 3 : 1); +} +console.log( + `CUT-OK ${res.id} ${res.seconds?.toFixed(2)}s (wanted ${res.want.toFixed(2)}s) ` + + `${res.mode} from ${res.window.name}`, +); diff --git a/umtool/bin/share-batch.mjs b/umtool/bin/share-batch.mjs @@ -0,0 +1,45 @@ +#!/usr/bin/env node +// Package the confirmed clips nobody has been sent yet. +// +// node bin/share-batch.mjs --project <id> --name <batch name> +// +// Spawned by the Deliver panel through lib/jobs.ts, for the reason every other +// long step is spawned: a 53-clip batch is two encodes per clip, its ffmpeg +// children have to be killable as a group, and its progress has to be a log +// somebody can read while it runs. +// +// The work is lib/report/deliver.mjs's, so `umtool` and the button cannot +// produce different batches. +import process from "node:process"; +import { resolveProject } from "../lib/projects/core.mjs"; +import { buildShareBatch } from "../lib/report/deliver.mjs"; + +const argv = process.argv.slice(2); +const val = (flag) => { + const i = argv.indexOf(flag); + return i < 0 ? null : argv[i + 1]; +}; + +const projectArg = val("--project"); +const name = val("--name"); +if (!projectArg || !name) { + console.error("usage: share-batch.mjs --project <id> --name <batch name>"); + process.exit(2); +} +if (!/^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(name)) { + console.error(`"${name}" is not a batch name: letters, digits, dot, dash and underscore`); + process.exit(2); +} + +const r = await resolveProject(projectArg); +if (!r.project) { + console.error(`no project matches "${projectArg}"`); + process.exit(2); +} + +try { + await buildShareBatch(r.project, name, { log: (l) => console.log(l) }); +} catch (e) { + console.error(e instanceof Error ? e.message : String(e)); + process.exit(1); +} diff --git a/umtool/components/projects/DeliverActions.tsx b/umtool/components/projects/DeliverActions.tsx @@ -0,0 +1,257 @@ +"use client"; + +import { useCallback, useEffect, useRef, useState } from "react"; +import { useRouter } from "next/navigation"; +import { buttonVariants } from "@/components/ui/button"; + +// --------------------------------------------------------------------------- +// The four things that happen after the last clip is judged. +// +// ONE JOB AT A TIME, and every button here says so rather than queueing: the +// server refuses a second one with the name of what is running (409), and the +// same rule is why a bench fetch and a cut cannot overlap. They write the same +// clips/ directory. +// +// `k of n` IS THE JOB'S OWN COUNT. The cut is one step per clip, so stepIndex +// and steps.length are the progress -- no second counter kept in the browser +// that a reload would lose, and a Stop that kills the running step's process +// group rather than a promise nobody can interrupt. +// +// STOP ABANDONS THE REST, it does not undo. Every file here is named for its +// clip, so pressing the button again resumes: the clips that have a file are +// not in the server's list any more. +// --------------------------------------------------------------------------- + +type StepView = { label: string; argv: string[] }; +type JobView = { + id: string; + kind: string; + project: string | null; + state: "running" | "done" | "failed"; + stepIndex: number; + steps: StepView[]; + error: string | null; + log: string[]; + next: number; +}; + +type Variant = { module: string; out: string }; + +export default function DeliverActions({ + project, + needCut, + notFetched, + candidates, + nextName, + unreviewed, + hasApply, + hasBuild, + variants, +}: { + project: string; + /** Confirmed, no mp4, and a window on this disk that holds it. */ + needCut: number; + /** Confirmed, no mp4, and nothing cached — a download, not a cut. */ + notFetched: number; + /** What a batch would ship: confirmed, cut, and not already shared. */ + candidates: number; + nextName: string; + unreviewed: number; + hasApply: boolean; + hasBuild: boolean; + variants: Variant[]; +}) { + const [job, setJob] = useState<JobView | null>(null); + const [error, setError] = useState<string | null>(null); + const [busy, setBusy] = useState(false); + const [partial, setPartial] = useState(false); + const [name, setName] = useState(nextName); + const since = useRef(0); + const router = useRouter(); + + // Adopt a job already running, so a reload does not lose one. Also how a + // second tab sees the first tab's cut. + useEffect(() => { + void fetch("/api/report/deliver?running=1", { cache: "no-store" }) + .then((r) => r.json()) + .then((j) => { + // The registry is process-wide and holds one job, which may be another + // project's build. Adopting that here would show its log under this + // project's panel. + const run = j.running as JobView | null; + if (run && run.project === project) { + setJob(run); + since.current = run.log.length; + } + }) + .catch(() => {}); + }, [project]); + + useEffect(() => { + if (!job || job.state !== "running") return; + const t = setInterval(async () => { + const r = await fetch(`/api/report/deliver?job=${job.id}&since=${since.current}`, { + cache: "no-store", + }); + if (!r.ok) return; + const j = (await r.json()) as JobView; + since.current = j.next; + setJob((prev) => (prev ? { ...j, log: [...prev.log, ...j.log] } : j)); + // The counts above this panel are SERVER-RENDERED off the directory, and + // the job just changed the directory: a cut that finished leaves "cut 2 + // confirmed clips" on screen until something re-reads it. Refreshed + // here, once, when the job ends -- not on a timer, which would fight the + // log the operator is reading. + if (j.state !== "running") router.refresh(); + }, 700); + return () => clearInterval(t); + }, [job, router]); + + const post = useCallback( + async (action: string, extra: Record<string, unknown> = {}) => { + setBusy(true); + setError(null); + const r = await fetch("/api/report/deliver", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ project, action, ...extra }), + }); + const j = (await r.json().catch(() => ({}))) as Record<string, unknown>; + setBusy(false); + if (!r.ok) { + setError(String(j.error ?? `HTTP ${r.status}`)); + return; + } + since.current = 0; + setJob(j.job as JobView); + }, + [project], + ); + + const running = job?.state === "running"; + const n = job?.steps.length ?? 0; + const k = Math.min((job?.stepIndex ?? 0) + 1, n); + + return ( + <div data-deliver-actions="" className="space-y-2"> + <div className="flex flex-wrap items-center gap-2"> + <button + type="button" + data-action="deliver-cut" + disabled={busy || running || !needCut} + className={buttonVariants({ size: "sm" })} + onClick={() => void post("cut")} + > + cut {needCut} confirmed clip{needCut === 1 ? "" : "s"} from cache + </button> + + <span className="flex items-center gap-1"> + <input + data-batch-name="" + aria-label="batch name" + value={name} + onChange={(e) => setName(e.target.value)} + className="w-44 rounded border border-[var(--color-line)] bg-[var(--color-bg)] px-1.5 py-0.5 font-mono text-[11px] text-[var(--color-text)]" + /> + <button + type="button" + data-action="deliver-share" + disabled={busy || running || !candidates} + className={buttonVariants({ variant: "ghost", size: "sm" })} + onClick={() => void post("share", { name })} + > + build share batch ({candidates}) + </button> + </span> + + {hasApply && ( + <span className="flex items-center gap-1"> + <button + type="button" + data-action="deliver-apply" + disabled={busy || running} + className={buttonVariants({ variant: "ghost", size: "sm" })} + onClick={() => void post("apply", partial ? { partial: true } : {})} + > + apply rulings + </button> + {unreviewed > 0 && ( + <label className="flex items-center gap-1 text-[11px] text-[var(--color-dim)]"> + <input + type="checkbox" + data-action="apply-partial" + checked={partial} + onChange={(e) => setPartial(e.target.checked)} + /> + apply partial ({unreviewed} unjudged) + </label> + )} + </span> + )} + + {hasBuild && variants.length > 0 && ( + <button + type="button" + data-action="deliver-rebuild" + disabled={busy || running} + className={buttonVariants({ variant: "ghost", size: "sm" })} + onClick={() => void post("rebuild")} + > + rebuild {variants.length} report{variants.length === 1 ? "" : "s"} + </button> + )} + + {running && ( + <button + type="button" + data-action="deliver-stop" + className={buttonVariants({ variant: "ghost", size: "sm" })} + onClick={() => + void fetch(`/api/report/deliver?cancel=${job!.id}`, { method: "POST" }) + } + > + stop + </button> + )} + </div> + + {notFetched > 0 && ( + <p className="text-[11px] text-[var(--color-dim)]"> + {notFetched} confirmed clip{notFetched === 1 ? " has" : "s have"} nothing cached to cut + from — those are a download, listed below. + </p> + )} + + {error && ( + <p data-deliver-error="" className="text-[11px] text-[var(--color-bad)]"> + {error} + </p> + )} + + {job && ( + <div data-deliver-job={job.id} data-deliver-state={job.state} className="space-y-1"> + <div className="flex flex-wrap items-baseline gap-2 text-[11px]"> + <span className="font-mono text-[var(--color-text)]">{job.kind}</span> + <span data-deliver-progress="" className="num text-[var(--color-dim)]"> + {job.state === "running" ? `${k} of ${n}` : `${job.state} · ${n} step${n === 1 ? "" : "s"}`} + </span> + {job.steps[job.stepIndex] && job.state === "running" && ( + <span className="text-[var(--color-dim)]">{job.steps[job.stepIndex].label}</span> + )} + {job.error && <span className="text-[var(--color-bad)]">{job.error}</span>} + </div> + {/* The log VERBATIM. apply-manifest.py prints the prose lines that + cite each clip ruled incorrect, and that listing is the whole + point of running it — paraphrasing it here would be this page + deciding what the operator has to read. */} + <pre + data-deliver-log="" + className="max-h-64 overflow-auto rounded border border-[var(--color-line)] bg-[var(--color-bg)] px-2 py-1 font-mono text-[10px] leading-tight text-[var(--color-dim)]" + > + {job.log.join("\n")} + </pre> + </div> + )} + </div> + ); +} diff --git a/umtool/components/projects/DeliverSection.tsx b/umtool/components/projects/DeliverSection.tsx @@ -0,0 +1,265 @@ +import Link from "next/link"; +import { deliverStateOf } from "@/lib/report/deliver.mjs"; +import { badgeVariants } from "@/components/ui/badge"; +import type { ProjectRef } from "@/lib/project-types"; +import DeliverActions from "./DeliverActions"; + +// --------------------------------------------------------------------------- +// DELIVER: the half of the work that starts when the walk ends. +// +// The bench answers one question per clip. What the report owes its readers is +// a different list -- which sections were actually confirmed, which confirmed +// clips have no file yet, which prose still cites a clip the walk threw out -- +// and until now that list lived in one project's shell scripts and in whoever +// remembered to run them in order. +// +// COUNTED BY SECTION, because that is how the report is read and how it falls +// apart: "69 of 163 confirmed" says nothing about a section where eleven of +// nineteen clips were ruled wrong, and that section is the one whose argument +// has to change. +// --------------------------------------------------------------------------- + +type Section = { + letter: string; + heading: string | null; + folder: string; + total: number; + confirmed: number; + incorrect: number; + unreviewed: number; + cut: number; +}; + +type State = { + sections: Section[]; + review: { total: number; confirmed: number; incorrect: number; unreviewed: number }; + cut: number; + needCut: { id: string; seconds: number }[]; + notFetched: string[]; + batches: { name: string; label: string; count: number; hasList: boolean }[]; + excluded: { shared: string[]; incorrect: string[] }; + candidates: { id: string; section: string; file: string }[]; + nextName: string; + variants: { file: string; module: string; out: string }[]; + incorrect: { id: string; correction: string; hits: { file: string; line: number; text: string }[] }[]; + hasApplyScript: boolean; + hasBuildScript: boolean; +}; + +export default async function DeliverSection({ + project, + manifest, + entries, +}: { + project: ProjectRef; + manifest: unknown; + /** readClipDetail's entries, so `fetched` is the bench's own answer. */ + entries: unknown[]; +}) { + const state = (await deliverStateOf(project, { manifest, entries })) as State | null; + // A timeline with no clips has nothing to deliver, and a panel of zeroes on + // every card-only cut is noise on a page that is already long. + if (!state || !state.review.total) return null; + const { review } = state; + + return ( + <section + data-deliver="" + data-deliver-confirmed={review.confirmed} + data-deliver-incorrect={review.incorrect} + data-deliver-unreviewed={review.unreviewed} + data-deliver-need-cut={state.needCut.length} + className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] px-3 py-2" + > + <div className="mb-1.5 flex flex-wrap items-baseline gap-3"> + <h2 className="micro">deliver</h2> + <span className="num text-[11px] text-[var(--color-dim)]"> + {review.confirmed} confirmed · {review.incorrect} incorrect · {review.unreviewed} not yet + judged · {state.cut} of {review.confirmed} cut + </span> + </div> + + {/* --- by section ------------------------------------------------- */} + <div className="overflow-x-auto"> + <table className="w-full border-collapse text-[11px]"> + <thead> + <tr className="micro text-left"> + <th className="px-1.5 py-0.5">section</th> + <th className="px-1.5 py-0.5">clips</th> + <th className="px-1.5 py-0.5">confirmed</th> + <th className="px-1.5 py-0.5">incorrect</th> + <th className="px-1.5 py-0.5">unjudged</th> + <th className="px-1.5 py-0.5">cut</th> + </tr> + </thead> + <tbody> + {state.sections.map((s) => ( + <tr + key={s.letter} + data-section={s.letter} + data-section-confirmed={s.confirmed} + data-section-incorrect={s.incorrect} + data-section-unreviewed={s.unreviewed} + className="border-t border-[var(--color-line)] align-top" + > + <td className="px-1.5 py-1"> + <span className="font-mono text-[var(--color-sel)]">{s.letter}</span>{" "} + <span className="text-[var(--color-dim)]">{s.heading ?? "—"}</span> + </td> + <td className="num px-1.5 py-1">{s.total}</td> + <td className="num px-1.5 py-1 text-[var(--color-good)]">{s.confirmed}</td> + <td className="num px-1.5 py-1"> + {s.incorrect ? ( + <span className="text-[var(--color-bad)]">{s.incorrect}</span> + ) : ( + <span className="text-[var(--color-dim)]">—</span> + )} + </td> + <td className="num px-1.5 py-1 text-[var(--color-dim)]">{s.unreviewed || "—"}</td> + <td className="num px-1.5 py-1 text-[var(--color-dim)]"> + {s.cut} / {s.confirmed} + </td> + </tr> + ))} + </tbody> + </table> + </div> + + {/* --- the actions ------------------------------------------------ */} + <div className="mt-2"> + <DeliverActions + project={project.id} + needCut={state.needCut.length} + notFetched={state.notFetched.length} + candidates={state.candidates.length} + nextName={state.nextName} + unreviewed={review.unreviewed} + hasApply={state.hasApplyScript} + hasBuild={state.hasBuildScript} + variants={state.variants.map((v) => ({ module: v.module, out: v.out }))} + /> + </div> + + {/* --- confirmed, but no file yet ---------------------------------- */} + {state.needCut.length > 0 && ( + <p data-need-cut="" className="mt-2 text-[11px] text-[var(--color-dim)]"> + <span className="micro">no file yet — </span> + {state.needCut.map((c, i) => ( + <span key={c.id}> + {i > 0 && " "} + <Link + href={`/browse/${project.id}/clip/${c.id}`} + data-need-cut-id={c.id} + className="font-mono text-[var(--color-sel)] hover:underline" + > + {c.id} + </Link> + </span> + ))} + </p> + )} + + {/* --- confirmed, and nothing cached to cut from -------------------- */} + {/* + A DIFFERENT PROBLEM, and it needs a different button. Nothing here + fetches: the clip's window is a download, which the bench already has a + managed path for (the editor's fetch, with its cookie policy and its + per-platform sleeps). So this lists them and links to the clip page + where that button lives. + */} + {state.notFetched.length > 0 && ( + <p data-not-fetched="" className="mt-1 text-[11px]"> + <span className="micro">not fetched — </span> + {state.notFetched.map((id, i) => ( + <span key={id}> + {i > 0 && " "} + <Link + href={`/browse/${project.id}/clip/${id}`} + data-not-fetched-id={id} + className="font-mono text-[var(--color-dirty)] hover:underline" + > + {id} + </Link> + </span> + ))} + <span className="ml-1 text-[var(--color-dim)]"> + — open one and fetch its window through the editor, then cut. + </span> + </p> + )} + + {/* --- prose that still cites a clip the walk threw out -------------- */} + {state.incorrect.length > 0 && ( + <div data-incorrect-citations="" className="mt-2 space-y-1.5"> + <h3 className="micro"> + prose citing an incorrect clip — {state.incorrect.length} + </h3> + {state.incorrect.map((c) => ( + <div key={c.id} data-incorrect={c.id} className="text-[11px] leading-snug"> + <Link + href={`/browse/${project.id}/clip/${c.id}`} + className="font-mono text-[var(--color-sel)] hover:underline" + > + {c.id} + </Link>{" "} + <span className="text-[var(--color-dim)]">{c.correction}</span> + {c.hits.length === 0 ? ( + <div className="text-[var(--color-dim)]">(no reference found in the prose)</div> + ) : ( + <ul className="mt-0.5 space-y-0.5"> + {c.hits.map((h) => ( + <li key={`${h.file}:${h.line}`} className="font-mono text-[10px] text-[var(--color-dim)]"> + <span className="text-[var(--color-text)]"> + {h.file}:{h.line} + </span>{" "} + {h.text} + </li> + ))} + </ul> + )} + </div> + ))} + <p className="text-[11px] text-[var(--color-dim)]"> + Nothing here rewrites prose: what a wrong clip does to an argument is a judgement about + the argument. Fix the lines, then rebuild. + </p> + </div> + )} + + {/* --- batches already sent ---------------------------------------- */} + <p className="mt-2 text-[11px] text-[var(--color-dim)]"> + {state.batches.length ? ( + <> + <span className="micro">batches — </span> + {state.batches.map((b, i) => ( + <span key={b.name} data-batch={b.name}> + {i > 0 && " · "} + <code className="font-mono">{b.name}</code> ({b.count}) + </span> + ))} + {". "} + </> + ) : ( + "no batch has been packaged yet. " + )} + The next one ships {state.candidates.length} clip + {state.candidates.length === 1 ? "" : "s"}: every confirmed clip with a file, minus the{" "} + {state.excluded.shared.length} already shared and the {state.excluded.incorrect.length}{" "} + ruled incorrect. + </p> + + {state.variants.length > 0 && ( + <p className="mt-1 text-[11px] text-[var(--color-dim)]"> + <span className="micro">reports — </span> + {state.variants.map((v, i) => ( + <span key={v.module}> + {i > 0 && " · "} + <code className="font-mono">{v.file}</code> →{" "} + <span className={badgeVariants({ variant: "info", size: "sm" })}>{v.out}</span> + </span> + ))} + </p> + )} + </section> + ); +} diff --git a/umtool/components/projects/ReportProject.tsx b/umtool/components/projects/ReportProject.tsx @@ -17,6 +17,7 @@ import { import { EXPORT_FORMATS, exportableVariants } from "@/lib/report/export.mjs"; import { diffManifests, formatChange } from "@/lib/report/manifest-diff.mjs"; import { listSnapshots, readSnapshot } from "@/lib/report/snapshots.mjs"; +import DeliverSection from "./DeliverSection"; import FetchUnfetchedButton from "./FetchUnfetchedButton"; import ReportBuildChain from "./ReportBuildChain"; import SnapshotButton from "./SnapshotButton"; @@ -401,6 +402,12 @@ export default async function ReportProject({ entries={entries.map((e) => ({ id: e.id, kind: e.kind }))} /> + {/* --- delivering it --------------------------------------------- */} + {/* After the build chain, because that is the order the work happens + in: the video is one deliverable and the written report with its + own clip files is the other, and both wait on the same walk. */} + <DeliverSection project={project} manifest={m} entries={entries} /> + {/* --- the timeline --------------------------------------------- */} <section> <h2 className="micro mb-1.5"> diff --git a/umtool/docs/clip-bench.md b/umtool/docs/clip-bench.md @@ -351,6 +351,79 @@ the CLI's formatting, tmp+rename under a lock, an mtime token). A stale token is **409 with both values**, never a silent overwrite: the other writer is usually somebody's judgement. +## Deliver — the half that starts when the walk ends + +The bench answers one question per clip. What a report owes its readers after +that is a different list, and it lives on the **project page** as `Deliver` +(`components/projects/DeliverSection.tsx`, `/api/report/deliver`): + +| | | +|---|---| +| **counts by section** | folded on the id's letter prefix, because `69 of 163 confirmed` says nothing about the section where eleven of nineteen clips were thrown out — and that is the section whose argument has to change. Section HEADINGS are read out of the report's own `content.py`, nth heading to nth letter. | +| **cut N confirmed clips from cache** | one job, **one step per clip**, so `k of n` and Stop are the job runner's own (`stepIndex`, and a cancel that kills the step's process group). Stop abandons the REST; the next press is the server's list again — "confirmed, no file" — so it resumes without re-cutting. | +| **not fetched** | a confirmed clip with nothing cached that holds it. Listed, never downloaded here — the managed fetch is the editor's, on the clip page. | +| **build share batch** | `share-<name>/{orig,std,small}/<Section>/<id>_<date>_<title>.mp4` + `LIST.md`. | +| **apply rulings** | runs the project's own `apply-manifest.py`, then `umtool corrections`. | +| **rebuild reports** | runs the project's own `build.py`, once per `content*.py`. | + +**A confirmed clip's frames are already on this disk.** Nobody can judge a clip +until its window is cached, so `clips/<id>.mp4` — what the written report's +players read — is a cut, not a download. `lib/report/cut.mjs` asks the project +cache for the tightest window containing the clip and cuts at +`clip.start − window.from`, through build-video's own `cutArgs()`: one spelling +of the arithmetic, shared with the render's segment pass. + +**The cut is measured, not predicted.** It stream-copies first and then probes +the result; more than 50 ms off the length asked for and it re-encodes. A copy +can only begin on a keyframe, and even a keyframe-aligned copy comes out long +when the aac frames do not end there (0.14 s, measured on the fixture) — +predicting that from the window's keyframes would have been wrong in exactly +the cases that matter. + +**The two share profiles are `lib/report/encode.mjs`, and only there.** They +came off `~/reports/elfpire-eva/share-report-clips/reencode.py`, which is where +the first batch was actually encoded; ported rather than spawned, because a +python file beside one project's deliverables is not a profile the next project +can reach. + +**Exclusions are the folders, not a list.** The next batch skips every id an +existing `share-*/LIST.md` already shipped and every clip ruled `incorrect`. A +`shared.json` would have been a second record to drift from them. Three ways an +id counts as shipped, in order of how much they can be trusted: the file NAMES +a list gives, the `<!-- shared-ids: … -->` marker this writes, and — narrowly — +the phrase `already shared (…)`, taking only the tokens shaped like a clip id. +(the parenthesis has to follow the phrase within ~120 characters, so one line +wrap is fine and a paragraph away is not). The last one is not decoration: six +ElfpireEva clips went out in an earlier set +under DIFFERENT ids (`em01`, `ie01` — a separate cut of the same moments), and +the sentence the next batch wrote down is the only thing on disk tying the two. +Without it, the next batch re-ships all six. A batch this writes phrases it the +same way, so its own list round-trips. + +**Nothing rewrites prose.** `apply-manifest.py` prints the lines of +`content*.py` that cite each clip the walk threw out, and the panel shows the +same grep itself. What a wrong clip does to an argument is a judgement about +the argument. + +**Apply is refused while the walk is unfinished** unless `apply partial` is +ticked: the script syncs `clips.json` from the manifest and deletes the mp4 of +every clip whose window moved, so running it over a half-walked cut bakes +"nobody has looked at this yet" into the deliverable as though it were a +verdict. + +**One job at a time, machine-wide in this process** (`lib/jobs.ts`): a bench +fetch and a Deliver cut cannot run together, and the second one gets a 409 +naming the first. That is not a limitation to route around — both write the +same `clips/` directory. + +**What the fixture cannot cover.** `apply-manifest.py` and `build.py` are the +REPORT's own, hand-written, living in an unversioned `~/reports/<project>/` +directory and differing per report. `e2e/deliver.spec.ts` runs two-line +stand-ins: what is under test is that the panel finds them, runs them in order, +refuses the first over a half-walked cut and shows their output verbatim — +never what they do. A project whose scripts are absent gets the buttons hidden +or a 400, not a guess. + ## Discovered by getting it wrong once **The page and the inbox must answer the same question the same way.** `umtool diff --git a/umtool/e2e/deliver.spec.ts b/umtool/e2e/deliver.spec.ts @@ -0,0 +1,294 @@ +import { test, expect } from "@playwright/test"; +import { execFileSync } from "node:child_process"; +import { existsSync, readdirSync, readFileSync, statSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +// --------------------------------------------------------------------------- +// DELIVER: what happens after the last clip is judged. +// +// The bench's own specs cover judging. This covers the other half — cutting the +// confirmed clips out of the cache, packaging a batch, and running the report's +// own fold-back scripts — and its strongest assertions are on DISK, because +// every one of these produces a file somebody else will open. +// +// The clips are MEASURED, not merely present. A cut that lands on the wrong +// keyframe still writes an mp4; what makes it wrong is that it is not the +// seconds that were asked for, and only ffprobe can say so. +// +// Nothing here reaches the network: the fixture's window was generated by +// ffmpeg, the cut reads it, and the panel has no fetch of its own by design. +// --------------------------------------------------------------------------- + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const FIXTURE = path.join(HERE, "..", ".e2e-song"); +const PROJECT = "reports/deliver-fixture"; +const DIR = path.join(FIXTURE, "reports", "deliver-fixture"); +const STOP_DIR = path.join(FIXTURE, "reports", "deliver-stop-fixture"); + +const seconds = (file: string): number => + Number( + execFileSync("ffprobe", [ + "-v", "error", "-show_entries", "format=duration", + "-of", "default=nw=1:nk=1", file, + ]).toString().trim(), + ); + +/** The job the panel started, polled to a terminal state. */ +async function waitForJob(page: import("@playwright/test").Page, timeout = 120_000) { + await expect(page.locator("[data-deliver-job]")).toHaveAttribute( + "data-deliver-state", + /done|failed/, + { timeout }, + ); + const state = await page.locator("[data-deliver-job]").getAttribute("data-deliver-state"); + const log = (await page.locator("[data-deliver-log]").textContent()) ?? ""; + return { state, log }; +} + +test("the deliver panel counts verdicts and folds them by id prefix", async ({ page }) => { + await page.goto(`/browse/${PROJECT}`); + const panel = page.locator("[data-deliver]"); + await expect(panel).toBeVisible(); + + // 4 confirmed (a01 a02 a03 b01), 1 incorrect (b02), 1 unjudged (b03). + await expect(panel).toHaveAttribute("data-deliver-confirmed", "4"); + await expect(panel).toHaveAttribute("data-deliver-incorrect", "1"); + await expect(panel).toHaveAttribute("data-deliver-unreviewed", "1"); + + // TWO SECTIONS, folded on the letter. The whole reason this is not one + // number: section B is where a clip was thrown out. + const a = panel.locator('[data-section="A"]'); + const b = panel.locator('[data-section="B"]'); + await expect(a).toHaveAttribute("data-section-confirmed", "3"); + await expect(a).toHaveAttribute("data-section-incorrect", "0"); + await expect(b).toHaveAttribute("data-section-confirmed", "1"); + await expect(b).toHaveAttribute("data-section-incorrect", "1"); + await expect(b).toHaveAttribute("data-section-unreviewed", "1"); + // And the heading came out of the report's own prose, not out of the id. + await expect(a).toContainText("The first fixture section"); + + // b01 is confirmed AND already has a file, so it is not waiting on a cut; + // a03 has nothing cached, so it is not a cut either. Two left. + await expect(panel).toHaveAttribute("data-deliver-need-cut", "2"); +}); + +test("a confirmed clip with no cached window is listed as not fetched, not cut", async ({ + page, +}) => { + await page.goto(`/browse/${PROJECT}`); + const notFetched = page.locator("[data-not-fetched]"); + await expect(notFetched).toBeVisible(); + // a03's window is 30–33 s and the cache holds 0–24. Nothing here downloads + // it: the link goes to the bench, where the editor fetch lives. + await expect(notFetched.locator('[data-not-fetched-id="a03"]')).toBeVisible(); + await expect(notFetched.locator('[data-not-fetched-id="a03"]')).toHaveAttribute( + "href", + `/browse/${PROJECT}/clip/a03`, + ); + // And it is NOT in the cut list, which is the distinction the two headings + // exist to make. + await expect(page.locator('[data-need-cut] [data-need-cut-id="a03"]')).toHaveCount(0); + expect(existsSync(path.join(DIR, "clips", "a03.mp4"))).toBe(false); +}); + +test("cutting from cache writes clips/<id>.mp4 of exactly the clip's length", async ({ + page, +}) => { + test.setTimeout(180_000); + await page.goto(`/browse/${PROJECT}`); + await page.locator('[data-action="deliver-cut"]').click(); + + // `k of n` is the JOB's count -- one step per clip -- so the readout and the + // work cannot disagree. + // Either "1 of 2" while it runs or "done · 2 steps" if it beat the first + // poll: what is asserted is that the readout is the job's step count. + await expect(page.locator("[data-deliver-progress]")).toContainText(/of 2|2 steps/); + const { state, log } = await waitForJob(page); + expect(state, log).toBe("done"); + expect(log).toContain("CUT-OK a01"); + expect(log).toContain("CUT-OK a02"); + + // THE MEASUREMENT. a01 starts on one of the window's keyframes and a02 does + // not; both must come out to the length the manifest asked for. + expect(seconds(path.join(DIR, "clips", "a01.mp4"))).toBeCloseTo(3.0, 1); + expect(seconds(path.join(DIR, "clips", "a02.mp4"))).toBeCloseTo(3.0, 1); + + // And the panel has nothing left to cut. + await page.reload(); + await expect(page.locator("[data-deliver]")).toHaveAttribute("data-deliver-need-cut", "0"); +}); + +test("the share batch skips every id an existing LIST.md already shipped", async ({ page }) => { + test.setTimeout(180_000); + await page.goto(`/browse/${PROJECT}`); + await page.locator("[data-batch-name]").fill("second"); + await page.locator('[data-action="deliver-share"]').click(); + const { state, log } = await waitForJob(page); + expect(state, log).toBe("done"); + + const root = path.join(DIR, "share-second"); + const folder = "A-The-first-fixture-section"; + for (const variant of ["orig", "std", "small"]) { + expect( + existsSync(path.join(root, variant, folder, "a01_2025-01-01_A-Fixture-Stream.mp4")), + `${variant}/a01`, + ).toBe(true); + } + + // THE EXCLUSIONS, and there are three kinds. + // + // b01 is in share-first by FILE NAME. a02 is in share-first only as a + // SENTENCE -- it went out in an earlier set under another id, which is the + // real project's case for six clips (em01/ie01 are a separate cut of + // b02/c08/f01/f05/h01/h02) and the only record that ties the two. b02 was + // ruled incorrect. The folders and their lists are the whole record: no + // second file to drift from them. + const list = readFileSync(path.join(root, "LIST.md"), "utf8"); + expect(list).toContain("a01_2025-01-01"); + expect(list).not.toContain("a02_"); + expect(list).not.toContain("b01_"); + expect(list).not.toContain("b02_"); + expect(list).toContain("2 already shared (a02 b01)"); + expect(list).toContain("1 ruled incorrect (b02)"); + expect( + existsSync(path.join(root, "orig", folder, "a02_2025-01-02_A-Fixture-Stream.mp4")), + ).toBe(false); + expect(existsSync(path.join(root, "orig", "B-The-second-fixture-section"))).toBe(false); + + // …and the list this wrote says it the same way, so the NEXT batch reading + // it back excludes all three without anybody restating them. + + // The std encode is the one people download, and it is EXACTLY 1280x720. + const size = execFileSync("ffprobe", [ + "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=width,height", + "-of", "csv=p=0", path.join(root, "std", folder, "a01_2025-01-01_A-Fixture-Stream.mp4"), + ]).toString().trim(); + expect(size).toBe("1280,720"); +}); + +test("apply rulings is refused while the walk is unfinished, and runs the report's own scripts when it is meant", async ({ + page, +}) => { + test.setTimeout(120_000); + await page.goto(`/browse/${PROJECT}`); + + // b03 has never been judged. apply-manifest.py deletes the mp4 of every clip + // whose window moved, so running it over a half-walked cut bakes "nobody has + // looked at this" into the deliverable as though it were a verdict. + await page.locator('[data-action="deliver-apply"]').click(); + await expect(page.locator("[data-deliver-error]")).toContainText("have not been judged"); + expect(existsSync(path.join(DIR, "applied.marker"))).toBe(false); + + // Ticked, it proceeds -- and the second step is `umtool corrections`, which + // is the same list the page shows, in the form the next sweep's prompt wants. + await page.locator('[data-action="apply-partial"]').check(); + await page.locator('[data-action="deliver-apply"]').click(); + const { state, log } = await waitForJob(page); + expect(state, log).toBe("done"); + expect(existsSync(path.join(DIR, "applied.marker"))).toBe(true); + // VERBATIM. The lines of prose that cite the clip the walk threw out are the + // whole point of running it, and the panel must not paraphrase them. + expect(log).toContain("content.py:6:"); + expect(log).toContain("the speaker is the guest, not the host"); + // And step 2's output is a FILE, because that is what the loop consumes: + // finish-sweep.sh opens with the same redirect and the overnight review + // reads corrections.md, not a log somebody copies out of a browser. + const corrections = readFileSync(path.join(DIR, "corrections.md"), "utf8"); + expect(corrections).toContain("Corrections for the next pass"); + expect(corrections).toContain("b02"); +}); + +test("the panel shows the prose lines citing a clip the walk ruled incorrect", async ({ + page, +}) => { + await page.goto(`/browse/${PROJECT}`); + const cited = page.locator('[data-incorrect="b02"]'); + await expect(cited).toBeVisible(); + await expect(cited).toContainText("content.py:6"); + await expect(cited).toContainText("The claim rests on [clip:b02]"); +}); + +test("rebuild runs build.py once per content variant present", async ({ page }) => { + test.setTimeout(120_000); + await page.goto(`/browse/${PROJECT}`); + await page.locator('[data-action="deliver-rebuild"]').click(); + const { state, log } = await waitForJob(page); + expect(state, log).toBe("done"); + + // content.py -> the default stem; content_lawyer.py -> report-lawyer. Two + // steps, two files, and neither overwrote the other's. + expect(log).toContain("BUILD-OK content -> report.html"); + expect(log).toContain("BUILD-OK content_lawyer -> report-lawyer.html"); + expect(readFileSync(path.join(DIR, "report.html"), "utf8")).toContain("built from content"); + expect(readFileSync(path.join(DIR, "report-lawyer.html"), "utf8")).toContain( + "built from content_lawyer", + ); +}); + + +// --------------------------------------------------------------------------- +// STOP, and what "resume" means. +// +// One step per clip is not a presentation choice: it is what makes Stop kill +// the running ffmpeg's process group and abandon the REST, and what makes the +// next press pick up the clips that have no file rather than re-cutting the +// ones that do. Both halves are asserted on DISK -- a count, and then the +// mtimes of the files the first job wrote, which must not have moved. +// +// Deterministic rather than racy: the fixture's server sets +// UMTOOL_CUT_DELAY_MS=2000 (playwright.config.ts), the job is six clips long, +// and the click waits for the first CUT-OK. Three clips of work -- six +// seconds -- therefore cannot have happened by the time Stop lands. +// --------------------------------------------------------------------------- + +const stopCuts = () => + readdirSync(path.join(STOP_DIR, "clips")) + .filter((n) => /^s\d+\.mp4$/.test(n)) + .sort(); + +test("Stop abandons the rest of the cut, and the next one resumes without re-cutting", async ({ + page, +}) => { + test.setTimeout(240_000); + await page.goto("/browse/reports/deliver-stop-fixture"); + await expect(page.locator("[data-deliver]")).toHaveAttribute("data-deliver-need-cut", "6"); + expect(stopCuts()).toHaveLength(0); + + await page.locator('[data-action="deliver-cut"]').click(); + await expect(page.locator("[data-deliver-progress]")).toContainText(/of 6/); + // WAIT FOR REAL WORK. Stopping before anything finished would prove only + // that a job can be cancelled, not that what it had already done survives. + await expect(page.locator("[data-deliver-log]")).toContainText("CUT-OK", { timeout: 90_000 }); + await page.locator('[data-action="deliver-stop"]').click(); + + const stopped = await waitForJob(page); + // A cancelled step is a failed one, and the log says who asked. + expect(stopped.state, stopped.log).toBe("failed"); + expect(stopped.log).toContain("cancel"); + + const done = stopCuts(); + expect(done.length).toBeGreaterThanOrEqual(1); + expect(done.length).toBeLessThan(6); + const stamps = new Map( + done.map((n) => [n, statSync(path.join(STOP_DIR, "clips", n)).mtimeMs]), + ); + + // THE SERVER'S LIST IS THE RESUME. It is "confirmed, no file", so the second + // job is exactly the remainder -- nothing in the browser had to remember + // where the first one stopped. + await page.reload(); + await expect(page.locator("[data-deliver]")).toHaveAttribute( + "data-deliver-need-cut", + String(6 - done.length), + ); + await page.locator('[data-action="deliver-cut"]').click(); + const second = await waitForJob(page); + expect(second.state, second.log).toBe("done"); + expect(stopCuts()).toHaveLength(6); + + for (const [name, ms] of stamps) { + expect(statSync(path.join(STOP_DIR, "clips", name)).mtimeMs, `${name} was re-cut`).toBe(ms); + expect(second.log, `${name} was re-cut`).not.toContain(`CUT-OK ${name.replace(".mp4", "")}`); + } +}); diff --git a/umtool/e2e/fixtures/make-fixture.mjs b/umtool/e2e/fixtures/make-fixture.mjs @@ -682,6 +682,19 @@ const CUES = { [3, 6, "It has a second clip to fetch."], [6, 9, "And nothing else cites it."], ], + // The DELIVER fixture's source. Its own, for the same reason vid3/vid4 are + // their own: a cut writes clips/<id>.mp4 and packages them, and a spec that + // shared a source with the fetch specs would depend on which ran first. + vid6: [ + [0, 3, "The deliver fixture opens."], + [3, 6, "The first clip is confirmed."], + [6, 9, "And so is the second."], + [9, 12, "Which does not start on a keyframe."], + [12, 15, "The third is nowhere on this disk."], + [15, 18, "The fourth was ruled incorrect."], + [18, 21, "The fifth nobody has judged."], + [21, 24, "And that is the whole cut."], + ], // Long enough that one clip's PADDED window can contain another's. See // editor-fetch-reuse-fixture. vid5: [ @@ -953,6 +966,198 @@ writeProject( ]), ); +// -- THE DELIVER FIXTURE ------------------------------------------------------ +// +// What a walked report owes its readers, in one project: the cut clips, the +// batch, the fold-back. Its own source (vid6) and its own directory, because +// every test here WRITES -- a cut lands in clips/, a batch in share-*/, and the +// project's own scripts leave marker files. +// +// Six clips over TWO SECTIONS, which is the point: the panel folds on the id's +// letter prefix, and "4 of 6 confirmed" says nothing about the section where a +// clip was thrown out. +// +// a01 3.00- 6.00 confirmed, cached, no file -> cut; starts ON one of the +// window's keyframes +// a02 9.50-12.50 confirmed, cached, no file -> cut; the nearest keyframe +// is at 9.00, so a copy would +// be half a second long +// a03 30.00-33.00 confirmed, NOTHING cached -> "not fetched": a download, +// which this panel never does +// b01 0.00- 3.00 confirmed, cached, HAS a file AND is listed in an existing +// share-first/LIST.md -> the +// batch must skip it +// b02 15.00-18.00 INCORRECT, with a correction and a line of prose citing it +// b03 18.00-21.00 nobody has judged it -> Apply is refused without "partial" +const DELIVER = writeProject( + "deliver-fixture", + manifest("deliver-fixture", "The Deliver Fixture", { siteOrigin: "https://archive.example" }, [ + { type: "clip", id: "a01", video: "vid6", start: 3.0, end: 6.0, cite: 3, section: 0, lock: true, verdict: "confirmed", date: "2025-01-01", title: "A Fixture Stream", quote: "The first clip is confirmed." }, + { type: "clip", id: "a02", video: "vid6", start: 9.5, end: 12.5, cite: 9, section: 0, lock: true, verdict: "confirmed", date: "2025-01-02", title: "A Fixture Stream", quote: "Which does not start on a keyframe." }, + { type: "clip", id: "a03", video: "vid6", start: 30.0, end: 33.0, cite: 30, section: 0, lock: true, verdict: "confirmed", date: "2025-01-03", title: "A Fixture Stream", quote: "Nothing on this disk holds it." }, + { type: "clip", id: "b01", video: "vid6", start: 0.0, end: 3.0, cite: 0, section: 0, lock: true, verdict: "confirmed", date: "2025-01-04", title: "A Fixture Stream", quote: "The deliver fixture opens." }, + { type: "clip", id: "b02", video: "vid6", start: 15.0, end: 18.0, cite: 15, section: 0, lock: true, verdict: "incorrect", correction: "the speaker is the guest, not the host", date: "2025-01-05", title: "A Fixture Stream", quote: "The fourth was ruled incorrect." }, + { type: "clip", id: "b03", video: "vid6", start: 18.0, end: 21.0, cite: 18, section: 0, lock: true, date: "2025-01-06", title: "A Fixture Stream", quote: "The fifth nobody has judged." }, + ]), +); + +// The report's OWN scripts, which the panel runs rather than reimplements. +// +// Two-line stand-ins for ~/reports/elfpire-eva's apply-manifest.py and +// build.py: they write a marker and print what the real ones print. That is +// the whole of what can be tested about them here -- the real pair live beside +// the prose they rewrite, in an unversioned reports directory, and differ per +// report. What IS under test is that the panel finds them, runs them in order, +// refuses to run the first one over a half-walked cut, and shows their output +// verbatim. +writeFileSync( + path.join(DELIVER, "apply-manifest.py"), + `#!/usr/bin/env python3 +# Fixture stand-in: prints what the real one prints, and leaves a marker. +import json, os, re +HERE = os.path.dirname(os.path.abspath(__file__)) +m = json.load(open(os.path.join(HERE, "video.manifest.json"))) +clips = [e for e in m["timeline"] if e.get("type") == "clip"] +wrong = [e for e in clips if e.get("verdict") == "incorrect"] +content = open(os.path.join(HERE, "content.py"), encoding="utf-8").read().splitlines() +print("clips.json: 0 field change(s) across 0 clip(s)") +print("review: %d clips: %d incorrect, %d confirmed, %d not yet reviewed" % ( + len(clips), len(wrong), + sum(1 for e in clips if e.get("verdict") == "confirmed"), + sum(1 for e in clips if not e.get("verdict")))) +for e in wrong: + print("== %s" % e["id"]) + print(" correction: %s" % e.get("correction")) + pat = re.compile(r"\\[clip:%s\\]|[\\"']%s[\\"']" % (e["id"], e["id"])) + for i, line in enumerate(content): + if pat.search(line): + print(" content.py:%d: %s" % (i + 1, line.strip()[:220])) +open(os.path.join(HERE, "applied.marker"), "w").write("applied\\n") +print("APPLY-DONE") +`, + { mode: 0o755 }, +); +writeFileSync( + path.join(DELIVER, "build.py"), + `#!/usr/bin/env python3 +# Fixture stand-in for the report renderer: one file per variant, so "once per +# content module, with build.py's own stem convention" is checkable on disk. +import argparse, os +ap = argparse.ArgumentParser() +ap.add_argument("--content", default="content") +ap.add_argument("--out", default="report") +a = ap.parse_args() +HERE = os.path.dirname(os.path.abspath(__file__)) +open(os.path.join(HERE, a.out + ".html"), "w").write("<!-- built from %s -->\\n" % a.content) +print("BUILD-OK %s -> %s.html" % (a.content, a.out)) +`, + { mode: 0o755 }, +); +// The prose. Its SECTIONS' headings are what name a batch's folders (nth +// heading to nth letter), and one paragraph cites the clip the walk threw out +// -- which is the line the panel has to surface and never rewrite. +writeFileSync( + path.join(DELIVER, "content.py"), + `TITLE = "The Deliver Fixture" +SECTIONS = [ + {"id": "one", "heading": "1. The first fixture section", + "blocks": [("p", "Two clips carry this one."), ("clips", ["a01", "a02", "a03"])]}, + {"id": "two", "heading": "2. The second fixture section", + "blocks": [("p", "The claim rests on [clip:b02], which the walk threw out."), + ("clips", ["b01", "b02"])]}, +] +`, +); +writeFileSync( + path.join(DELIVER, "content_lawyer.py"), + `TITLE = "The Deliver Fixture, for lawyers" +SECTIONS = [ + {"id": "one", "heading": "1. The narrow cut", "blocks": [("clips", ["a01"])]}, +] +`, +); +// A batch that already went out. The folders ARE the record, so the exclusion +// list is read back from this rather than from a second file that would drift +// from it -- by FILE NAME for b01, and out of the prose for a02, which is the +// real project's case: six clips went out in an earlier set under different +// ids, and the sentence naming them is the only thing that ties the two. +mkdirSync(path.join(DELIVER, "share-first"), { recursive: true }); +writeFileSync( + path.join(DELIVER, "share-first", "LIST.md"), + [ + "# deliver-fixture — share batch `first`", + "", + "1 clip, sent before this fixture was born — and 1 already shared (a02) in", + "an earlier set under another id, which is the only record that it went out.", + "", + "## 2. The second fixture section (`B-The-second-fixture-section/`)", + "", + "- **b01_2025-01-04_A-Fixture-Stream.mp4** — 2025-01-04 · 3s", + "", + ].join("\n"), +); + +// The cache the cut reads: ONE window holding every clip but a03. +// +// KEYFRAMES EVERY THREE SECONDS, which is what a window fetched with +// --force-keyframes-at-cuts has. a01 starts on one (3.00) and a02 does not +// (9.50) -- and measured, NEITHER copy survives, because the aac frames do not +// end where the keyframe does and a copy of this window comes out 0.14 s long. +// That is exactly why the cut measures its result instead of predicting it +// from the keyframes, and why both clips still come out to the frame. +mkdirSync(path.join(DELIVER, "out", "clips-raw"), { recursive: true }); +ff([ + "-f", "lavfi", "-i", "testsrc=size=320x180:rate=30:duration=24", + "-f", "lavfi", "-i", "sine=frequency=440:duration=24", + "-t", "24", + "-c:v", "libx264", "-pix_fmt", "yuv420p", + "-force_key_frames", "expr:gte(t,n_forced*3)", + "-c:a", "aac", "-ar", "48000", "-ac", "2", + path.join(DELIVER, "out", "clips-raw", "vid6_0.00-24.00.mp4"), +]); +// b01 already has its file: it is confirmed AND shared, so it is the clip the +// batch must leave out rather than one the cut has to make. +mkdirSync(path.join(DELIVER, "clips"), { recursive: true }); +ff([ + "-f", "lavfi", "-i", "testsrc=size=320x180:rate=30:duration=3", + "-f", "lavfi", "-i", "sine=frequency=330:duration=3", + "-t", "3", "-c:v", "libx264", "-pix_fmt", "yuv420p", + "-c:a", "aac", "-ar", "48000", "-ac", "2", + path.join(DELIVER, "clips", "b01.mp4"), +]); + +// A SECOND deliver project, for STOP. +// +// Its own, because the cut is one job over every clip that needs one: a spec +// that stopped halfway through deliver-fixture's two would leave that project +// in a state the batch tests do not expect, and a six-clip job is what makes +// "stopped after k of n" a measurement rather than a race. +// +// Six confirmed clips, none of them cut, all inside the same cached window. +// With UMTOOL_CUT_DELAY_MS set (playwright.config.ts) each step takes about +// two seconds, so the spec can see the first CUT-OK, press Stop, and know that +// at least three clips could not possibly have been reached. +const STOP = writeProject( + "deliver-stop-fixture", + manifest("deliver-stop-fixture", "The Deliver Stop Fixture", { siteOrigin: "https://archive.example" }, [ + { type: "clip", id: "s01", video: "vid6", start: 0.0, end: 2.0, cite: 0, section: 0, lock: true, verdict: "confirmed", date: "2025-02-01", title: "A Fixture Stream", quote: "The deliver fixture opens." }, + { type: "clip", id: "s02", video: "vid6", start: 2.0, end: 4.0, cite: 2, section: 0, lock: true, verdict: "confirmed", date: "2025-02-02", title: "A Fixture Stream", quote: "The first clip is confirmed." }, + { type: "clip", id: "s03", video: "vid6", start: 4.0, end: 6.0, cite: 4, section: 0, lock: true, verdict: "confirmed", date: "2025-02-03", title: "A Fixture Stream", quote: "And so is the second." }, + { type: "clip", id: "s04", video: "vid6", start: 6.0, end: 8.0, cite: 6, section: 0, lock: true, verdict: "confirmed", date: "2025-02-04", title: "A Fixture Stream", quote: "And so is the third." }, + { type: "clip", id: "s05", video: "vid6", start: 8.0, end: 10.0, cite: 8, section: 0, lock: true, verdict: "confirmed", date: "2025-02-05", title: "A Fixture Stream", quote: "Which does not start on a keyframe." }, + { type: "clip", id: "s06", video: "vid6", start: 10.0, end: 12.0, cite: 10, section: 0, lock: true, verdict: "confirmed", date: "2025-02-06", title: "A Fixture Stream", quote: "Nor does this one." }, + ]), +); +mkdirSync(path.join(STOP, "out", "clips-raw"), { recursive: true }); +copyFileSync( + path.join(DELIVER, "out", "clips-raw", "vid6_0.00-24.00.mp4"), + path.join(STOP, "out", "clips-raw", "vid6_0.00-24.00.mp4"), +); +// An EMPTY clips/, so the spec can count files without first asking whether +// the directory exists -- and so "nothing has been cut yet" is a state the +// fixture states rather than one it leaves to chance. +mkdirSync(path.join(STOP, "clips"), { recursive: true }); + // -- STUB BINARIES, so a build is offline and deterministic -------------------- // // The pipeline shells out to yt-dlp for the availability preflight and for every @@ -1235,4 +1440,6 @@ console.log(` deep/nested/solo-fixture (collapse case), bench-fixture console.log(` walk-fixture (read-only: w01/w04 walkable, w02 unfetched, w03 judged),`); console.log(` editor-fetch-{,many-,reuse-}fixture (nothing cached — the editor fetch's subjects),`); console.log(` longform-fixture (cue gap, legacy .bak, ffmeta), longform-edit-fixture, dash-fixture`); +console.log(` deliver-fixture (writable: a01/a02 to cut, a03 unfetched, b01 shared, b02 incorrect, b03 unjudged)`); +console.log(` deliver-stop-fixture (writable: six confirmed clips to cut, for Stop and resume)`); console.log(` ${taken} candidate files copied, 2 mix tracks synthesised`); diff --git a/umtool/lib/report/cut.mjs b/umtool/lib/report/cut.mjs @@ -0,0 +1,152 @@ +// Cutting ONE clip out of the window already on disk. +// +// The deliverable a written report ships is `clips/<id>.mp4` -- the players in +// report.html read exactly that -- and the first batch of them was fetched one +// per clip with `yt-dlp --download-sections`. That is a second download of +// seconds already paid for: the bench fetches a generous window per clip into +// `out/clips-raw` (or, now, into the corpus) before anybody can judge it, and +// every confirmed clip therefore already has its own frames on this disk. +// +// So this cuts from the cache and never touches the network. A clip with no +// containing window is NOT cut here and is not an error either -- it is a clip +// nobody has fetched yet, which is the bench's own "not fetched" state and has +// its own button. +// +// The ffmpeg cut itself is build-video's: `cutArgs()` is the render's segment +// pass, exported rather than copied, so the seconds this writes and the seconds +// the video renders are the same arithmetic. +import { execFile } from "node:child_process"; +import { mkdir, rename, rm } from "node:fs/promises"; +import path from "node:path"; +import { promisify } from "node:util"; +import { FFMPEG_BIN, cutArgs } from "umtool-report-to-video/build-video"; +import { clipsOf, readManifest } from "../projects/report.mjs"; +import { ACCURATE_CUT_ARGS, probeSeconds } from "./encode.mjs"; +import { projectCache } from "./serve.mjs"; + +const execFileP = promisify(execFile); + +/** Where the report's own players look. Not configurable: build.py hardcodes it. */ +export const CLIPS_DIR = "clips"; + +/** + * How far out a stream-copied cut may land before it is re-encoded. + * + * A copy can only start on a keyframe. A window fetched with + * `--force-keyframes-at-cuts` has one exactly where the clip begins, so the + * copy is exact and free; a window fetched any other way -- or a clip whose + * edges were MOVED on the bench after the fetch -- has one wherever the encoder + * put it, and the copy silently begins seconds early. That is not a rounding + * error to tolerate: it is the wrong sentence. + */ +export const CUT_TOLERANCE = 0.05; + +/** ffmpeg gets a generous cap; a cut of a cached window is seconds of work. */ +const CUT_TIMEOUT_MS = 5 * 60_000; + +/** + * Cut `clipId` of `project` out of the cached window that contains it. + * + * @param {{ id: string, dir: string }} project + * @param {string} clipId + * @param {{ reencode?: "auto" | "always" | "never", manifest?: object | null, + * cache?: Awaited<ReturnType<typeof projectCache>> | null }} [opts] + * `reencode` is the operator's override of the keyframe test above: + * "always" for a window whose copy is known to be wrong, "never" for a + * re-cut that must not lose a generation. + * @returns {Promise<{ ok: boolean, id: string, reason?: string, error?: string, + * path?: string, rel?: string, seconds?: number, want?: number, + * mode?: "copy" | "reencode", window?: { name: string, from: number, to: number } }>} + */ +export async function cutClipFromCache( + project, + clipId, + { reencode = "auto", manifest = null, cache = null } = {}, +) { + const m = manifest ?? (await readManifest(project.dir)); + if (!m) return { ok: false, id: clipId, reason: "no-manifest", error: "no manifest" }; + const clip = clipsOf(m).find((e) => e.id === clipId); + if (!clip) return { ok: false, id: clipId, reason: "no-clip", error: "no such clip" }; + + const start = Number(clip.start); + const end = Number(clip.end); + if (!Number.isFinite(start) || !Number.isFinite(end) || !(end > start)) { + return { ok: false, id: clipId, reason: "no-window", error: "this clip has no window" }; + } + + // THE EXTENT, not the cut-to-quote. `cutStart`/`cutEnd` is what the VIDEO + // plays; a written report's player is the reviewed extent, which is what + // clips.json carries and what the caption under it describes. + const c = cache ?? (await projectCache(project, m)); + const win = c.containing(clip.video, start, end); + if (!win) { + // Not an error. Nobody has fetched this one yet, and the bench has a + // button for exactly that. + return { + ok: false, + id: clipId, + reason: "not-fetched", + error: "no cached window holds this clip end to end", + }; + } + + const want = end - start; + const a = Math.max(0, start - win.from); + const b = a + want; + const dir = path.join(project.dir, CLIPS_DIR); + await mkdir(dir, { recursive: true }); + const out = path.join(dir, `${clipId}.mp4`); + const tmp = path.join(dir, `.${clipId}.cutting.mp4`); + + const run = async (args) => { + await rm(tmp, { force: true }); + await execFileP( + FFMPEG_BIN, + ["-nostdin", "-v", "error", "-y", ...cutArgs(win.path, a, b), ...args, tmp], + { maxBuffer: 1 << 24, timeout: CUT_TIMEOUT_MS }, + ); + return probeSeconds(tmp); + }; + + let mode = reencode === "always" ? "reencode" : "copy"; + let got = null; + try { + if (mode === "copy") { + // `-avoid_negative_ts make_zero` so the copied packets' timestamps start + // at zero: without it a copy that began on an earlier keyframe carries + // the window's own clock into the file, and every player disagrees about + // how long it is. + got = await run(["-c", "copy", "-avoid_negative_ts", "make_zero", "-movflags", "+faststart"]); + const off = got == null ? Infinity : Math.abs(got - want); + if (off > CUT_TOLERANCE && reencode !== "never") { + // The window's keyframes are not where this clip's edges are, so the + // copy is the wrong seconds. Pay for one generation and get the cut + // that was asked for. + mode = "reencode"; + got = await run(ACCURATE_CUT_ARGS); + } + } else { + got = await run(ACCURATE_CUT_ARGS); + } + } catch (e) { + await rm(tmp, { force: true }); + return { + ok: false, + id: clipId, + reason: "ffmpeg", + error: e instanceof Error ? e.message : String(e), + }; + } + + await rename(tmp, out); + return { + ok: true, + id: clipId, + path: out, + rel: path.posix.join(CLIPS_DIR, `${clipId}.mp4`), + seconds: got == null ? null : Number(got.toFixed(3)), + want: Number(want.toFixed(3)), + mode, + window: { name: win.name, from: win.from, to: win.to }, + }; +} diff --git a/umtool/lib/report/deliver.mjs b/umtool/lib/report/deliver.mjs @@ -0,0 +1,421 @@ +// DELIVERY: what a walked report owes the people who will read it. +// +// The bench answers one question per clip. What happens after the last one is +// a second job entirely, and until now it lived as five shell scripts and a +// python file in ONE project's directory (~/reports/elfpire-eva): cut the +// confirmed clips, fold the rulings back into the prose, rebuild every report +// variant, and package a batch of mp4s for whoever is writing the piece. +// +// None of that is project-specific except the prose. So it is here, as a +// surface: the counts by section, the clips still missing a file, the batch, +// and the two child processes (apply-manifest.py, build.py) that are the +// project's own and are RUN rather than reimplemented. +// +// Plain ESM with no Next imports, because the batch builder is also spawned as +// a script by lib/jobs.ts -- one implementation, whether a button or a terminal +// asked for it. +import { copyFile, mkdir, readFile, readdir, stat, writeFile } from "node:fs/promises"; +import { execFile } from "node:child_process"; +import path from "node:path"; +import { promisify } from "node:util"; +import { FFMPEG_BIN } from "umtool-report-to-video/build-video"; +import { citeUrlFor, clipVerdict, clipsOf, readManifest } from "../projects/report.mjs"; +import { SHARE_PROFILES } from "./encode.mjs"; +import { CLIPS_DIR } from "./cut.mjs"; + +const execFileP = promisify(execFile); + +/** `share-<name>/` is the batch directory, and the prefix is how they are found. */ +export const SHARE_PREFIX = "share-"; + +/** + * The SECTION a clip belongs to, read off its own id. + * + * Ids in a sectioned report are letter-prefixed by section -- a01…a08, b01…b06 + * -- and that prefix is the only thing that ties a clip to a section in the + * MANIFEST, which carries no sections at all. (clips.json carries `section`, + * but the manifest is the source of truth once a walk starts.) So the prefix is + * the grouping key, and a clip with no letter prefix lands in "?" rather than + * being dropped. + */ +export const sectionOf = (id) => (/^([A-Za-z]+)/.exec(String(id ?? "")) ?? [, "?"])[1].toUpperCase(); + +/** A-Z by position: the first section is A, which is how the ids were assigned. */ +const letterAt = (i) => (i < 26 ? String.fromCharCode(65 + i) : `Z${i - 25}`); + +const slug = (s, max) => + String(s ?? "") + .normalize("NFKD") + .replace(/[^\p{L}\p{N}]+/gu, "-") + .replace(/^-+|-+$/g, "") + .slice(0, max) + .replace(/-+$/g, ""); + +/** + * The section headings, read out of the report's own content module. + * + * A report's prose is a python file the build renders; its SECTIONS carry the + * headings a reader sees, in order, and the nth of them is the nth letter. That + * is a convention rather than a schema, so it is read defensively: no content + * file, or no headings in it, and a section is named by its letter alone. A + * folder called `C` is worse than `C-Family-law-for-a-year-then-quitting-the-` + * and better than a wrong name. + */ +export async function sectionHeadings(projectDir, moduleName = "content") { + const text = await readFile(path.join(projectDir, `${moduleName}.py`), "utf8").catch(() => null); + if (!text) return new Map(); + const out = new Map(); + const re = /["']heading["']\s*:\s*(["'])((?:\\.|(?!\1)[^\\])*)\1/g; + let m; + let i = 0; + while ((m = re.exec(text))) { + // The number the writer put in front of the heading is the section's + // position, which the letter already says. + out.set(letterAt(i), m[2].replace(/^\s*\d+[.)]\s*/, "").replace(/\\(.)/g, "$1")); + i += 1; + } + return out; +} + +/** `<Letter>-<slugged heading>`, the folder a batch sorts a clip into. */ +export const sectionFolder = (letter, heading) => + heading ? `${letter}-${slug(heading, 40)}` : letter; + +/** `<id>_<date>_<title>.mp4` — the name the first batch shipped under. */ +export const clipFileName = (clip) => + [clip.id, clip.date ?? "undated", slug(clip.title ?? "untitled", 50) || "untitled"].join("_") + + ".mp4"; + +/** Every id a batch directory already shipped. */ +export async function sharedIdsIn(dir) { + const ids = new Set(); + // The LIST.md is the batch's own manifest, and the only one a hand-made + // batch is guaranteed to have. Ids are read out of the file NAMES it lists, + // which is the one part of its prose that cannot drift from the files. + const list = await readFile(path.join(dir, "LIST.md"), "utf8").catch(() => null); + if (list) { + for (const m of list.matchAll(/\b([A-Za-z]{1,3}\d{1,3})_[^\s`*]*\.mp4\b/g)) ids.add(m[1]); + const marked = /<!--\s*shared-ids:\s*([^>]*?)\s*-->/.exec(list); + if (marked) for (const id of marked[1].split(/[\s,]+/).filter(Boolean)) ids.add(id); + // AND THE IDS A LIST NAMES AS SHARED SOMEWHERE ELSE. + // + // Measured on the real project: six clips went out in an earlier set under + // DIFFERENT ids (em01, ie01 -- a separate cut of the same moments), and + // nothing on disk ties those files to b02/c08/f01/f05/h01/h02. What does + // tie them is the sentence the batch that followed wrote down: "minus the + // 6 already shared in the emancipation/Ireland set (b02 c08 f01 f05 h01 + // h02)". Without this the next batch re-ships all six. + // + // Prose, and read as narrowly as prose can be: the phrase, then a + // parenthesis within the next 120 characters, then only the tokens that + // are shaped like a clip id. The batches this writes phrase it the same + // way, so a generated list round trips through here unchanged. + // + // The 120 is what lets a HAND-WRAPPED list still be read -- the phrase and + // its parenthesis routinely end up on two lines -- while stopping "already + // shared" in one paragraph from claiming a parenthetical three paragraphs + // down. Ids inside the parenthesis are still filtered by shape, so a + // wrongly-claimed one contributes nothing unless it reads like a clip id. + for (const m of list.matchAll(/already shared[^(]{0,120}\(([^)]*)\)/gi)) { + for (const tok of m[1].split(/[\s,]+/)) { + if (/^[A-Za-z]{1,3}\d{1,3}$/.test(tok)) ids.add(tok); + } + } + } + // And the files themselves, for a batch assembled before anyone wrote a list. + const walk = async (d) => { + for (const ent of await readdir(d, { withFileTypes: true }).catch(() => [])) { + if (ent.isDirectory()) await walk(path.join(d, ent.name)); + else { + const m = /^([A-Za-z]{1,3}\d{1,3})_.*\.mp4$/.exec(ent.name); + if (m) ids.add(m[1]); + } + } + }; + await walk(dir); + return ids; +} + +/** The batches already in this project, newest name last. */ +export async function listBatches(projectDir) { + const names = (await readdir(projectDir, { withFileTypes: true }).catch(() => [])) + .filter((e) => e.isDirectory() && e.name.startsWith(SHARE_PREFIX)) + .map((e) => e.name) + .sort(); + return Promise.all( + names.map(async (name) => { + const dir = path.join(projectDir, name); + const ids = [...(await sharedIdsIn(dir))].sort(); + return { + name, + label: name.slice(SHARE_PREFIX.length), + dir, + ids, + hasList: !!(await readFile(path.join(dir, "LIST.md"), "utf8").catch(() => null)), + }; + }), + ); +} + +/** The content modules this project can render, and the flags build.py needs. */ +export async function contentVariants(projectDir) { + const names = (await readdir(projectDir).catch(() => [])) + .filter((n) => /^content(_[A-Za-z0-9_]+)?\.py$/.test(n)) + .sort(); + return names.map((file) => { + const mod = file.replace(/\.py$/, ""); + // build.py's own defaults: `content` renders to `report`, and every other + // module renders to `report-<suffix>` so two variants cannot overwrite each + // other's html. + const out = mod === "content" ? "report" : `report-${mod.slice("content_".length)}`; + return { + file, + module: mod, + out, + argv: mod === "content" ? [] : ["--content", mod, "--out", out], + }; + }); +} + +/** + * The lines of prose that cite a clip the walk ruled INCORRECT. + * + * Not rewritten, and deliberately: what a wrong clip does to an argument is a + * judgement about the argument. apply-manifest.py says the same thing in the + * terminal; this says it on the page the operator is already looking at, so + * "which paragraphs do I have to touch" is not a second command. + */ +export async function incorrectCitations(projectDir, manifest, variants) { + const wrong = clipsOf(manifest).filter((e) => clipVerdict(e) === "incorrect"); + if (!wrong.length) return []; + const files = await Promise.all( + variants.map(async (v) => ({ + file: v.file, + lines: (await readFile(path.join(projectDir, v.file), "utf8").catch(() => "")).split("\n"), + })), + ); + return wrong.map((e) => { + const id = e.id.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); + const re = new RegExp(`\\[clip:${id}\\]|["']${id}["']`); + const hits = []; + for (const f of files) { + f.lines.forEach((line, i) => { + if (re.test(line)) hits.push({ file: f.file, line: i + 1, text: line.trim().slice(0, 240) }); + }); + } + return { id, correction: String(e.correction ?? "").trim(), hits }; + }); +} + +/** + * Everything the Deliver panel shows, computed server-side in one pass. + * + * `entries` is readClipDetail's, so `fetched` (is there a window holding this + * clip) is the bench's own answer rather than a second one computed here. + * + * @param {{ id: string, dir: string }} project + * @param {{ manifest?: any, entries?: any[] }} [opts] + */ +export async function deliverStateOf(project, { manifest = null, entries = null } = {}) { + const m = manifest ?? (await readManifest(project.dir)); + if (!m) return null; + const clips = clipsOf(m); + const fetchedOf = new Map( + (entries ?? []).filter((e) => (e.kind ?? e.type) === "clip").map((e) => [e.id, !!e.fetched]), + ); + const have = new Set( + (await readdir(path.join(project.dir, CLIPS_DIR)).catch(() => [])) + .filter((n) => n.endsWith(".mp4")) + .map((n) => n.slice(0, -4)), + ); + + const headings = await sectionHeadings(project.dir); + const batches = await listBatches(project.dir); + const shared = new Set(batches.flatMap((b) => b.ids)); + const variants = await contentVariants(project.dir); + + const rows = clips.map((e) => ({ + id: e.id, + section: sectionOf(e.id), + verdict: clipVerdict(e), + date: e.date ?? null, + title: e.title ?? null, + seconds: Number((Number(e.end) - Number(e.start)).toFixed(1)), + cut: have.has(e.id), + // A clip nobody has fetched cannot be cut, and saying "cut failed" about it + // would send somebody looking for a bug instead of pressing fetch. + fetched: fetchedOf.get(e.id) ?? null, + shared: shared.has(e.id), + file: clipFileName(e), + href: citeUrlFor(m, e), + quote: String(e.quote ?? "").trim(), + })); + + const byLetter = new Map(); + for (const r of rows) { + if (!byLetter.has(r.section)) byLetter.set(r.section, []); + byLetter.get(r.section).push(r); + } + const sections = [...byLetter.entries()] + .sort(([a], [b]) => a.localeCompare(b)) + .map(([letter, list]) => ({ + letter, + heading: headings.get(letter) ?? null, + folder: sectionFolder(letter, headings.get(letter)), + total: list.length, + confirmed: list.filter((r) => r.verdict === "confirmed").length, + incorrect: list.filter((r) => r.verdict === "incorrect").length, + unreviewed: list.filter((r) => r.verdict === "unreviewed").length, + cut: list.filter((r) => r.verdict === "confirmed" && r.cut).length, + ids: list.map((r) => r.id), + })); + + const confirmed = rows.filter((r) => r.verdict === "confirmed"); + const needCut = confirmed.filter((r) => !r.cut); + const candidates = confirmed.filter((r) => r.cut && !r.shared); + + return { + project: project.id, + dir: project.dir, + sections, + review: { + total: rows.length, + confirmed: confirmed.length, + incorrect: rows.filter((r) => r.verdict === "incorrect").length, + unreviewed: rows.filter((r) => r.verdict === "unreviewed").length, + }, + cut: confirmed.length - needCut.length, + // Cuttable now: a window is already on this disk. + needCut: needCut.filter((r) => r.fetched !== false).map((r) => ({ id: r.id, seconds: r.seconds })), + // And the ones that first need a download, which is a different button. + notFetched: needCut.filter((r) => r.fetched === false).map((r) => r.id), + batches: batches.map((b) => ({ name: b.name, label: b.label, count: b.ids.length, hasList: b.hasList })), + excluded: { + shared: [...shared].sort(), + incorrect: rows.filter((r) => r.verdict === "incorrect").map((r) => r.id), + }, + candidates: candidates.map((r) => ({ id: r.id, section: r.section, file: r.file })), + nextName: `batch-${new Date().toISOString().slice(0, 10)}`, + variants, + incorrect: await incorrectCitations(project.dir, m, variants), + hasApplyScript: await exists(path.join(project.dir, "apply-manifest.py")), + hasBuildScript: await exists(path.join(project.dir, "build.py")), + }; +} + +/** Is there a file here? `stat`, not a read: an `orig/` mp4 is megabytes. */ +const exists = (p) => stat(p).then((st) => st.isFile(), () => false); + +/** The batch's own manifest, in the shape the first one shipped. */ +export function renderListMd(project, name, sections, { excluded }) { + const total = sections.reduce((n, s) => n + s.rows.length, 0); + const lines = [ + `# ${project.id} — share batch \`${name}\``, + "", + `${total} clip${total === 1 ? "" : "s"}: every clip the umtool bench CONFIRMED, minus ` + + `${excluded.shared.length} already shared${excluded.shared.length ? ` (${excluded.shared.join(" ")})` : ""}` + + ` and ${excluded.incorrect.length} ruled incorrect` + + `${excluded.incorrect.length ? ` (${excluded.incorrect.join(" ")})` : ""}.`, + "", + "Folders: `orig/` (the cut as fetched), " + + Object.entries(SHARE_PROFILES).map(([k, p]) => `\`${k}/\` (${p.label})`).join(", ") + + ". Files are `<clipId>_<date>_<title>.mp4`.", + "", + // Machine-readable, so the NEXT batch's exclusions are a read rather than a + // parse of the prose above. + `<!-- shared-ids: ${sections.flatMap((s) => s.rows.map((r) => r.id)).join(" ")} -->`, + "", + ]; + for (const s of sections) { + lines.push(`## ${s.heading ?? `Section ${s.letter}`} (\`${s.folder}/\`)`, ""); + for (const r of s.rows) { + lines.push( + `- **${r.file}** — ${r.date ?? "undated"} · ${Math.round(r.seconds)}s · ` + + `[${r.title ?? r.id}](${r.href})`, + ); + if (r.quote) lines.push(` "${r.quote}"`); + } + lines.push(""); + } + return lines.join("\n"); +} + +/** + * Build `share-<name>/{orig,std,small}/<Section>/<file>.mp4` + LIST.md. + * + * Every encode is SKIPPED when its output is already there, like reencode.py's + * own cache: a batch interrupted at clip 40 of 53 resumes rather than restarts. + * Progress is one line per file so lib/jobs.ts's log reads as work. + * + * @param {{ id: string, dir: string }} project + * @param {string} name + */ +export async function buildShareBatch(project, name, { log = console.log } = {}) { + const state = await deliverStateOf(project); + if (!state) throw new Error("no manifest"); + const m = await readManifest(project.dir); + const clips = new Map(clipsOf(m).map((e) => [e.id, e])); + const headings = await sectionHeadings(project.dir); + const chosen = state.candidates.map((c) => c.id); + if (!chosen.length) throw new Error("nothing to ship: every confirmed clip is already shared, or not cut yet"); + + const root = path.join(project.dir, `${SHARE_PREFIX}${name}`); + const bySection = new Map(); + for (const id of chosen) { + const e = clips.get(id); + const letter = sectionOf(id); + if (!bySection.has(letter)) bySection.set(letter, []); + bySection.get(letter).push({ + id, + file: clipFileName(e), + date: e.date ?? null, + title: e.title ?? null, + seconds: Number(e.end) - Number(e.start), + href: citeUrlFor(m, e), + quote: String(e.quote ?? "").trim(), + }); + } + const sections = [...bySection.entries()] + .sort(([a], [b]) => a.localeCompare(b)) + .map(([letter, rows]) => ({ + letter, + heading: headings.get(letter) ?? null, + folder: sectionFolder(letter, headings.get(letter)), + rows, + })); + + log(`BATCH-START ${chosen.length} clips into ${path.basename(root)}`); + for (const s of sections) { + for (const r of s.rows) { + const src = path.join(project.dir, CLIPS_DIR, `${r.id}.mp4`); + const orig = path.join(root, "orig", s.folder, r.file); + await mkdir(path.dirname(orig), { recursive: true }); + if (await exists(orig)) log(`ORIG-CACHED ${r.id}`); + else { + await copyFile(src, orig); + log(`ORIG-OK ${r.id}`); + } + for (const [tag, profile] of Object.entries(SHARE_PROFILES)) { + const out = path.join(root, tag, s.folder, r.file); + await mkdir(path.dirname(out), { recursive: true }); + if (await exists(out)) { + log(`${tag.toUpperCase()}-CACHED ${r.id}`); + continue; + } + await execFileP( + FFMPEG_BIN, + ["-nostdin", "-v", "error", "-y", "-i", orig, ...profile.args, out], + { maxBuffer: 1 << 24, timeout: 10 * 60_000 }, + ); + log(`${tag.toUpperCase()}-OK ${r.id}`); + } + } + } + await writeFile( + path.join(root, "LIST.md"), + renderListMd(project, name, sections, { excluded: state.excluded }), + "utf8", + ); + log(`BATCH-DONE ${chosen.length} clips · ${path.basename(root)}/LIST.md`); + return { root, count: chosen.length, sections: sections.length }; +} diff --git a/umtool/lib/report/driver.mjs b/umtool/lib/report/driver.mjs @@ -263,3 +263,132 @@ export const localFetch = () => process.env.UMTOOL_LOCAL_FETCH === "1"; export function checkSourcesSteps(projects, env = {}) { return projects.map((p) => availabilityStep(p, env)); } + +// --------------------------------------------------------------------------- +// DELIVERY: the steps that come after the walk. +// +// Same contract as every other chain here: the client sends a project and, at +// most, a name. Never a path, never an argv. What is different is that two of +// these run the PROJECT'S OWN python -- apply-manifest.py and build.py, which +// live beside the prose they rewrite and differ per report. They are RUN, not +// reimplemented: what folding a ruling back into an argument means is a +// decision the report's author already wrote down. +// --------------------------------------------------------------------------- + +/** This package's own root. bin/ lives here, and so does report-to-video/. */ +export const UMTOOL_DIR = path.resolve(process.cwd()); + +const tool = (name) => path.join(UMTOOL_DIR, "bin", name); + +/** The interpreter a project's own scripts are run with. */ +export const PYTHON = process.env.PYTHON_BIN ?? "python3"; + +/** + * Cut every named clip out of the cache: ONE STEP PER CLIP. + * + * Which is what makes the panel's `k of n` and its Stop real rather than + * decorative -- jobs.ts runs steps strictly in order, reports the index, and + * cancels by killing the running step's process group. A single step looping + * over the ids would have had none of that, and a loop of POSTs in the browser + * would have had to fight the one-job-at-a-time rule for every clip. + * + * @param {{ id: string, dir: string }} project + * @param {string[]} clipIds + * @returns {import("../trim").Step[]} + */ +export function cutSteps(project, clipIds) { + return clipIds.map((id) => ({ + cwd: UMTOOL_DIR, + env: {}, + label: `cut ${id} from the cached window`, + argv: ["node", tool("cut-from-cache.mjs"), "--project", project.id, "--clip", id], + timeoutMs: 10 * 60_000, + })); +} + +/** + * Package a share batch. One step, because its own log is per file. + * @param {{ id: string }} project + * @param {string} name + */ +export function shareBatchSteps(project, name) { + return [ + { + cwd: UMTOOL_DIR, + env: {}, + label: `package share-${name}`, + argv: ["node", tool("share-batch.mjs"), "--project", project.id, "--name", name], + // Two encodes per clip over a batch that can be fifty of them. + timeoutMs: 60 * 60_000, + }, + ]; +} + +/** + * Fold the bench's rulings back into the report's sources. + * + * Step 1 is the project's own apply-manifest.py: it syncs clips.json from the + * manifest, deletes the mp4 of every clip whose window MOVED (so the cut list + * refills and those clips are re-cut), and prints the prose lines that cite + * each clip ruled incorrect. Step 2 prints the corrections as markdown, which + * is the form the next sweep's prompt wants. + * + * Nothing rewrites prose. That is the point. + */ +export function applyRulingsSteps(project) { + return [ + { + cwd: project.dir, + env: {}, + label: "apply-manifest.py — sync clips.json, drop the mp4s of moved windows", + argv: [PYTHON, path.join(project.dir, "apply-manifest.py")], + timeoutMs: 10 * 60_000, + }, + { + cwd: project.dir, + env: {}, + label: `umtool corrections ${project.id} → corrections.md`, + // REDIRECTED, because the file is the artifact. finish-sweep.sh opens + // with `umtool corrections <id> > corrections.md` and the overnight + // review reads that file; a job log somebody has to copy out of a + // browser is not the same thing. + // + // `$0` is the destination and `"$@"` the command, both passed as + // POSITIONAL arguments rather than interpolated into the script: a + // project id and a directory can then contain anything at all without + // becoming shell. `exec` keeps the command's own exit status. + argv: [ + "sh", + "-c", + 'exec "$@" > "$0"', + path.join(project.dir, "corrections.md"), + "node", + tool("umtool.mjs"), + "corrections", + project.id, + ], + timeoutMs: 5 * 60_000, + }, + ]; +} + +/** + * Re-render every report variant the project carries. + * + * One step per content module, with build.py's own flags: bare `content` + * renders to the default stem, and `content_<x>` renders to `report-<x>` so two + * variants cannot overwrite each other's html. The list comes from the + * directory (contentVariants), never from the client. + * + * @param {{ dir: string }} project + * @param {{ module: string, out: string, argv: string[] }[]} variants + */ +export function rebuildReportSteps(project, variants) { + return variants.map((v) => ({ + cwd: project.dir, + env: {}, + label: `build.py → ${v.out}.{html,bbcode,md}`, + argv: [PYTHON, path.join(project.dir, "build.py"), ...v.argv], + timeoutMs: 15 * 60_000, + })); +} diff --git a/umtool/lib/report/encode.mjs b/umtool/lib/report/encode.mjs @@ -0,0 +1,83 @@ +// The encode profiles, in ONE place. +// +// These came off `share-report-clips/reencode.py` in ~/reports/elfpire-eva, +// which is where the first share batch was actually encoded and therefore the +// only honest source for "what the last batch looked like". Ported rather than +// spawned: a python file living beside one project's deliverables is not a +// profile the next project can reach, and two copies of an x264 command line +// is exactly the drift that makes batch 2 not match batch 1. +// +// `std` and `small` are that script's two profiles, argument for argument. +// `accurate` is neither: it is the re-encode a CUT falls back to when the cut +// cannot be stream-copied, and it deliberately keeps the source's geometry -- +// an `orig/` file is "the window as fetched", and rescaling it here would make +// the folder's own description untrue. +import { execFile } from "node:child_process"; +import { promisify } from "node:util"; +import { FFPROBE_BIN } from "umtool-report-to-video/build-video"; + +const execFileP = promisify(execFile); + +/** Scale-and-pad to an exact frame, the shape both share profiles use. */ +const box = (w, h) => + `scale=w=${w}:h=${h}:force_original_aspect_ratio=decrease,` + + `pad=${w}:${h}:(ow-iw)/2:(oh-ih)/2,format=yuv420p`; + +export const SHARE_PROFILES = { + // Exactly 1280x720, faststart, a level every phone and every forum player + // will take. This is the one people download. + std: { + label: "1280x720 · crf 23", + args: [ + "-vf", box(1280, 720), + "-c:v", "libx264", "-profile:v", "high", "-level", "4.0", + "-preset", "medium", "-crf", "23", "-r", "30", "-g", "60", + "-c:a", "aac", "-b:a", "128k", "-ar", "44100", "-ac", "2", + "-movflags", "+faststart", + ], + }, + // 640x360 under a 300k ceiling: the copy that survives an upload limit. + small: { + label: "640x360 · crf 28 · 300k cap", + args: [ + "-vf", box(640, 360), + "-c:v", "libx264", "-profile:v", "main", "-level", "3.1", + "-preset", "medium", "-crf", "28", "-maxrate", "300k", "-bufsize", "600k", + "-r", "30", "-g", "60", + "-c:a", "aac", "-b:a", "64k", "-ar", "44100", "-ac", "2", + "-movflags", "+faststart", + ], + }, +}; + +/** The encode names a batch produces, beside the untouched `orig`. */ +export const SHARE_VARIANTS = ["orig", ...Object.keys(SHARE_PROFILES)]; + +/** + * The re-encode a cut falls back to when the window's keyframes are in the + * wrong places. No scaling and no frame-rate change: the point of this file is + * that its first and last frames are the seconds that were asked for. + */ +export const ACCURATE_CUT_ARGS = [ + "-c:v", "libx264", "-preset", "veryfast", "-crf", "20", "-pix_fmt", "yuv420p", + "-c:a", "aac", "-b:a", "160k", + "-movflags", "+faststart", +]; + +/** + * A file's duration in seconds, from the container. + * + * build-video's probeDuration() counts VIDEO FRAMES, because a rendered + * segment's length has to agree with the timeline it is concatenated into. + * Nothing here is concatenated: the question is "did this cut come out the + * length it was asked for", and the container's own answer is the one a + * downloader will see. + */ +export async function probeSeconds(file) { + const { stdout } = await execFileP(FFPROBE_BIN, [ + "-v", "error", "-show_entries", "format=duration", + "-of", "default=nw=1:nk=1", file, + ]); + const n = Number(String(stdout).trim()); + return Number.isFinite(n) ? n : null; +} diff --git a/umtool/playwright.config.ts b/umtool/playwright.config.ts @@ -71,6 +71,10 @@ export default defineConfig({ // with. Set here rather than in a step's env: jobView() echoes a step's // env back to the browser, and this is a token. `ARCHILYZER_EDITOR_URL=http://127.0.0.1:${STUB_PORT} WORKER_TOKEN=umtool-e2e-token ` + + // Two seconds of pacing per CUT, so the deliver spec can press Stop in + // the middle of a job and prove that cancelling abandons the rest while + // the next run resumes. Read only by bin/cut-from-cache.mjs. + `UMTOOL_CUT_DELAY_MS=2000 ` + `NEXT_DIST_DIR=.next-e2e pnpm exec next dev --port ${PORT}`, port: PORT, reuseExistingServer: false, diff --git a/umtool/report-to-video/build-video.mjs b/umtool/report-to-video/build-video.mjs @@ -77,6 +77,12 @@ const FFMPEG = process.env.FFMPEG_BIN ?? "ffmpeg"; const FFPROBE = process.env.FFPROBE_BIN ?? "ffprobe"; const QRENCODE = process.env.QRENCODE_BIN ?? "qrencode"; +// The binaries, under the names the rest of umtool calls them by. A second +// module reading FFMPEG_BIN for itself would be a second place a fixture's +// stub has to be wired in, and the one that forgot would shell out to the real +// ffmpeg in the middle of a test. +export { FFMPEG as FFMPEG_BIN, FFPROBE as FFPROBE_BIN }; + // Cue windows and per-video metadata come from a local corpus when there is one // and from the published archive otherwise, so this runs in a clone with no // `transcripts/` directory. Built once main() has the manifest (it carries the @@ -252,7 +258,7 @@ async function videoMeta(videoId, channelSlug, hints = {}) { // The video stream's frame COUNT is the number the timeline actually runs on, // so derive the duration from it. nb_frames is absent on some demuxers; fall // back to the container rather than failing a build over a probe. -async function probeDuration(file, fps) { +export async function probeDuration(file, fps) { if (fps) { const { stdout } = await execFileP(FFPROBE, [ "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=nb_frames", @@ -345,6 +351,27 @@ export async function findContainingWindow(rawDir, video, from, to) { return tightestContaining(await cachedWindowsFor(rawDir, video), from, to); } +/** + * WHERE A CLIP IS CUT OUT OF A CACHED WINDOW, as ffmpeg input arguments. + * + * `-ss`/`-to` BEFORE `-i`, and both matter. Before the input, ffmpeg seeks the + * demuxer rather than decoding and discarding, which is the difference between + * seconds and minutes on a 45-minute window; `-to` before the input is then + * measured on the same clock as the `-ss`, i.e. in the SOURCE FILE's seconds, + * which is what an offset into a cached window is. + * + * Exported because two things cut a clip out of a window now -- the render's + * segment pass and the bench's "cut confirmed clips from cache" -- and a second + * spelling of this is a second set of seconds that can drift from the first. + * + * @param {string} raw the cached window file + * @param {number} a seconds INTO that file where the clip starts + * @param {number} b seconds into it where the clip ends + */ +export function cutArgs(raw, a, b) { + return ["-ss", a.toFixed(3), "-to", b.toFixed(3), "-i", raw]; +} + async function fetchClip(entry, meta, render, rawDir, opts) { // Deliberately over-fetch: the snapping pass below needs room on both sides to // find a silence, and a clip that has no slack can only be cut where the cue @@ -687,7 +714,7 @@ async function buildClipSegment(entry, meta, render, dirs, opts, chrome, nodes, // eof_action=repeat holds the strip's last frame, which is the parked bar. const barT = Math.max(0.2, cutB - cutA - 0.25); - const inputs = ["-ss", cutA.toFixed(3), "-to", cutB.toFixed(3), "-i", raw]; + const inputs = cutArgs(raw, cutA, cutB); let nextIdx = 1; let footerIdx, markerIdx, barIdx, qrIdx; if (hasFooter) {