// Turning a report project into a chain of steps lib/jobs.ts can run. // // SPAWN, not import. A 40-minute chain of yt-dlp and ffmpeg inside a request // handler has no cancellation story, its execFile buffers live in the server's // heap, and a runaway grandchild outlives the request that started it. The // scripts are given a progress protocol instead (`--progress ndjson`), so the // UI reads events rather than scraping prose. // // The client sends a PROJECT and a PRESET NAME. It never sends a path, an argv // or an env map -- the same contract /api/browse/build already keeps. import path from "node:path"; /** Where the pipeline lives. One place, so a move is one edit. */ export const PIPELINE_DIR = path.resolve(/* turbopackIgnore: true */ process.cwd(), "report-to-video"); const script = (name) => path.join(/* turbopackIgnore: true */ PIPELINE_DIR, name); /** * Presets, in the order somebody actually works. * * `preview` is for looking at ONE clip after moving its edges; `fast` is hard * cuts over the whole timeline, which is minutes rather than tens of minutes and * is what you watch to check the argument; `final` is the deliverable. */ export const PRESETS = { preview: { label: "preview one clip", xfade: false, chapters: false, only: true }, fast: { label: "fast pass (hard cuts)", xfade: false, chapters: true, only: false }, final: { label: "final", xfade: true, chapters: true, only: false }, }; /** A build's timeout, scaled to the work rather than to a global guess. */ export const buildTimeoutMs = (clipCount, xfade) => Math.max(15 * 60_000, clipCount * (xfade ? 120_000 : 60_000)); /** * The chain. * * Step 1 is the availability preflight and it is a STEP, not a preamble: it is * the one fact about a manifest that goes stale in both directions, it costs * seconds, and without it a dead source is discovered twenty minutes and a * dozen paid-for fetches into the build. * * Step 2 runs resolve-windows DRY. If it reports changes, the chain stops and * shows them -- a widener silently rewriting windows somebody just set in the * bench is exactly the surprise `lock` exists to prevent. Applying is a second, * explicit action. */ /** * The build options the pipeline has and the driver used to hide. * * variant `--variant sourced|full` -- which cut of a two-cut manifest * xfade: false `--no-xfade` on a preset that would crossfade * chaptersOnly `--chapters-only` -- retitle the chapters from the segments * already on disk; no fetch, no encode * preview `--preview ` -- the rail alone over a window * chromeOnly `--chrome-only` -- re-render the on-screen deck over the * segments already on disk and re-concat; no segment rebuilt * * chaptersOnly, preview and chromeOnly SKIP the preflight and the dry resolve: * none of them touches a source. The first two are seconds of work under a * five-minute cap; a deck re-render is a render plus a concat of the whole * cut, so it keeps the build's own timeout. * * @typedef {{ variant?: string, xfade?: boolean, chaptersOnly?: boolean, chromeOnly?: boolean, preview?: { at: number, dur: number } | null }} BuildOptions */ /** The step-1 preflight alone, reused by the check-sources job. */ export function availabilityStep(project, env = {}) { const manifest = path.join(project.dir, "video.manifest.json"); return { cwd: PIPELINE_DIR, env, label: `check every source of ${project.id} is still fetchable`, argv: ["node", script("check-availability.mjs"), manifest, "--out", path.join(project.dir, "out"), "--allow-missing"], timeoutMs: 10 * 60_000, }; } /** * @param {{ id: string, dir: string }} project * @param {{ preset?: string, only?: string | null, skipFetch?: boolean, env?: Record, clipCount?: number, options?: BuildOptions }} [opts] * @returns {import("../trim").Step[]} */ export function buildSteps(project, { preset = "fast", only = null, skipFetch = false, env = {}, clipCount = 20, options = {} } = {}) { const p = PRESETS[preset] ?? PRESETS.fast; const manifest = path.join(project.dir, "video.manifest.json"); const outDir = path.join(project.dir, "out"); const base = { cwd: PIPELINE_DIR, env }; const quick = !!(options.chaptersOnly || options.preview); const steps = quick || options.chromeOnly ? [] : [ { ...base, label: "check every source is still fetchable", argv: ["node", script("check-availability.mjs"), manifest, "--out", outDir], timeoutMs: 10 * 60_000, }, { ...base, label: "resolve windows (dry — nothing is written)", argv: ["node", script("resolve-windows.mjs"), manifest], timeoutMs: 5 * 60_000, }, ]; const buildArgv = [ "node", script("build-video.mjs"), manifest, "--out", outDir, "--progress", "ndjson", "--continue-on-error", ]; if (options.variant) buildArgv.push("--variant", options.variant); if (!p.xfade || options.xfade === false) buildArgv.push("--no-xfade"); if (!p.chapters) buildArgv.push("--no-chapters"); if (skipFetch) buildArgv.push("--skip-fetch"); if (p.only && only) buildArgv.push("--only", only); if (options.chaptersOnly) buildArgv.push("--chapters-only"); if (options.chromeOnly) buildArgv.push("--chrome-only"); if (options.preview) buildArgv.push("--preview", String(options.preview.at), String(options.preview.dur)); steps.push({ ...base, label: options.chaptersOnly ? "retitle the chapters (no encode)" : options.chromeOnly ? "re-render on-screen" : options.preview ? `rail preview at ${options.preview.at}s` : p.label, argv: buildArgv, ndjson: true, timeoutMs: quick ? 5 * 60_000 : buildTimeoutMs(clipCount, p.xfade), }); // A build can exit 0 and still be wrong: a concat that produced nothing, a // chapter pass that dropped markers, a timeline that lost a clip because // --continue-on-error let it. Each looks like success at the terminal. // // Skipped for a one-clip preview, which deliberately does not produce a // deliverable to measure. // // Also skipped for a rail preview: it writes .preview.mp4, which is // not the deliverable and has no chapters to count. if ((!p.only || !only) && !options.preview) { const verify = ["node", script("verify-build.mjs"), manifest, "--out", outDir]; if (options.variant) verify.push("--variant", options.variant); if (buildArgv.includes("--no-xfade")) verify.push("--no-xfade"); steps.push({ ...base, label: "verify the file that came out", argv: verify, timeoutMs: 5 * 60_000, }); } return steps; } /** * The widest pad the bench may ask for. * * Here rather than in the route, because the BENCH has to know it too: the * button reads "fetch to ±N s" and has to say "that is as wide as it goes" * rather than offering a number the server will quietly clamp. */ export const FETCH_MAX_PAD = 120; /** * Fetch ONE clip's window, wide. What the bench's "fetch more" runs. * @param {{ dir: string }} project * @param {string} clipId * @param {number | { padBefore: number, padAfter: number }} pad * @returns {import("../trim").Step[]} */ export function fetchSteps(project, clipId, pad) { // A number is the symmetric shorthand; {padBefore, padAfter} is one edge at // a time, which is what extending a window actually is. const before = typeof pad === "object" && pad ? Number(pad.padBefore) : Number(pad); const after = typeof pad === "object" && pad ? Number(pad.padAfter) : Number(pad); return [ { cwd: PIPELINE_DIR, env: {}, label: `fetch ${clipId} with −${before}s / +${after}s of pad`, argv: [ "node", script("build-video.mjs"), path.join(project.dir, "video.manifest.json"), "--out", path.join(project.dir, "out"), "--fetch-only", clipId, "--pad-before", String(before), "--pad-after", String(after), "--progress", "ndjson", ], ndjson: true, timeoutMs: 10 * 60_000, }, ]; } /** * The same fetch, ASKED OF THE EDITOR. * * Beside fetchSteps rather than replacing it: a machine with no editor to ask * still needs the local path, and UMTOOL_LOCAL_FETCH=1 is how it says so. The * step shape, the argv flags and the NDJSON events are identical, so the job * runner, the bench's progress readout and the cache predicate cannot tell the * two apart — which is the only way "which one ran" stays an operator's * decision rather than a fork in every consumer. * * The editor URL and the shared WORKER_TOKEN come from the ENVIRONMENT, never * from the request: a client that could name the editor could name any host -- * and never from the step's `env` either, which jobView() shows the browser. * * @param {{ dir: string }} project * @param {string} clipId * @param {number | { padBefore: number, padAfter: number }} pad * @returns {import("../trim").Step[]} */ export function editorFetchSteps(project, clipId, pad) { const before = typeof pad === "object" && pad ? Number(pad.padBefore) : Number(pad); const after = typeof pad === "object" && pad ? Number(pad.padAfter) : Number(pad); return [ { cwd: PIPELINE_DIR, // EMPTY, and that is the point. A step's `env` is echoed back to the // browser by jobView(), so putting WORKER_TOKEN here would print the // shared secret on the page that started the job. The child inherits // process.env, which is where both values already are. env: {}, label: `ask the editor for ${clipId} with −${before}s / +${after}s of pad`, argv: [ "node", script("fetch-via-editor.mjs"), path.join(project.dir, "video.manifest.json"), "--fetch-only", clipId, "--pad-before", String(before), "--pad-after", String(after), "--progress", "ndjson", ], ndjson: true, timeoutMs: 15 * 60_000, }, ]; } /** * THE WHOLE RECORDING, asked of the editor. * * Same script, same events, one extra flag — so the job runner and the bench's * progress readout cannot tell it from a window fetch, which is what keeps * "which one ran" an operator's decision rather than a fork in every consumer. * * There is no local twin on purpose. A full source is the expensive ask, and * the reason to route it through the editor (cookie policy, per-platform * sleeps, the 429 cooldown, and a container that lands in the store where the * next report reuses it) is strongest exactly here. UMTOOL_LOCAL_FETCH is about * a machine with no editor; such a machine has nowhere to put a full source * that anything else would find. * * The timeout is the window fetch's doubled: a whole recording can be hours of * media, and the job keeps running on the editor if we give up — nothing is * lost, and the next ask finds it cached. * * THE HEIGHT. `--max-height` is the caller's `maxHeight`, else the manifest's * `render.maxHeightSource` — the tallest source the report renders from, so a * whole recording is not fetched taller than the cut will use. At 720 or less * the editor saves its 720p H.264 preset, above it the original. A value that * is not a whole number from 144 to 2160 falls through to the next, and with * neither the flag is left off and the channel's own quality applies. * * @param {{ dir: string }} project * @param {string} clipId * @param {{ maxHeight?: unknown, manifest?: { render?: { maxHeightSource?: unknown } } | null }} [opts] * @returns {import("../trim").Step[]} */ export function editorFullSourceSteps(project, clipId, opts = {}) { const maxHeight = [opts.maxHeight, opts.manifest?.render?.maxHeightSource].find(isFetchMaxHeight); return [ { cwd: PIPELINE_DIR, // EMPTY — see editorFetchSteps: jobView() echoes a step's env to the // browser, and WORKER_TOKEN is already in process.env. env: {}, label: `ask the editor for the whole source behind ${clipId}`, argv: [ "node", script("fetch-via-editor.mjs"), path.join(project.dir, "video.manifest.json"), "--fetch-only", clipId, "--full", ...(maxHeight !== undefined ? ["--max-height", String(maxHeight)] : []), "--progress", "ndjson", ], ndjson: true, timeoutMs: 30 * 60_000, }, ]; } /** * A source height the editor accepts as a fetch cap: a whole number of pixels * from 144 to 2160 (common/lib/clipWindow.ts isFetchMaxHeight, whose bounds * fetch-via-editor.mjs checks too). * @param {unknown} v * @returns {v is number} */ function isFetchMaxHeight(v) { return typeof v === "number" && Number.isInteger(v) && v >= 144 && v <= 2160; } /** Whether this instance fetches locally with yt-dlp instead of asking. */ export const localFetch = () => process.env.UMTOOL_LOCAL_FETCH === "1"; /** * One preflight per project, in order. What the check-sources job runs: the * driver's own step 1, reused as-is, so `out/availability.json` is written by * the same script whether a build or a re-check asked for it. * * `--allow-missing`, because a dead source here is a FINDING to record, not a * failure to abort the rest of the list on. * @param {{ id: string, dir: string }[]} projects * @returns {import("../trim").Step[]} */ export function checkSourcesSteps(projects, env = {}) { return projects.map((p) => availabilityStep(p, env)); } // --------------------------------------------------------------------------- // DELIVERY: the steps that come after the walk. // // Same contract as every other chain here: the client sends a project and, at // most, a name. Never a path, never an argv. What is different is that two of // these run the PROJECT'S OWN python -- apply-manifest.py and build.py, which // live beside the prose they rewrite and differ per report. They are RUN, not // reimplemented: what folding a ruling back into an argument means is a // decision the report's author already wrote down. // --------------------------------------------------------------------------- /** This package's own root. bin/ lives here, and so does report-to-video/. */ export const UMTOOL_DIR = path.resolve(/* turbopackIgnore: true */ process.cwd()); const tool = (name) => path.join(/* turbopackIgnore: true */ UMTOOL_DIR, "bin", name); /** The interpreter a project's own scripts are run with. */ export const PYTHON = process.env.PYTHON_BIN ?? "python3"; /** * Cut every named clip out of the cache: ONE STEP PER CLIP. * * Which is what makes the panel's `k of n` and its Stop real rather than * decorative -- jobs.ts runs steps strictly in order, reports the index, and * cancels by killing the running step's process group. A single step looping * over the ids would have had none of that, and a loop of POSTs in the browser * would have had to fight the one-job-at-a-time rule for every clip. * * @param {{ id: string, dir: string }} project * @param {string[]} clipIds * @returns {import("../trim").Step[]} */ export function cutSteps(project, clipIds) { return clipIds.map((id) => ({ cwd: UMTOOL_DIR, env: {}, label: `cut ${id} from the cached window`, argv: ["node", tool("cut-from-cache.mjs"), "--project", project.id, "--clip", id], timeoutMs: 10 * 60_000, })); } /** * Package a share batch. One step, because its own log is per file. * @param {{ id: string }} project * @param {string} name */ export function shareBatchSteps(project, name) { return [ { cwd: UMTOOL_DIR, env: {}, label: `package share-${name}`, argv: ["node", tool("share-batch.mjs"), "--project", project.id, "--name", name], // Two encodes per clip over a batch that can be fifty of them. timeoutMs: 60 * 60_000, }, ]; } /** * Move a project's deliverables (`clips/`, every `share-*` directory) to the media root * or back, then set `storage.deliverables` (release 17). One step: the CLI's * own log is per directory. Run as a JOB so the app's one-job-at-a-time rule * keeps every cut and batch out while it moves; the CLI's process scan covers * what runs outside the app. * @param {{ id: string }} project * @param {"media" | "local"} to */ export function moveDeliverablesSteps(project, to) { return [ { cwd: UMTOOL_DIR, env: {}, label: `move deliverables to ${to}`, argv: ["node", tool("umtool.mjs"), "storage", "deliverables", project.id, "--to", to], // A copy, a mirror and a verify of what can be gigabytes of mp4. timeoutMs: 60 * 60_000, }, ]; } /** * Fold the bench's rulings back into the report's sources. * * Step 1 is the project's own apply-manifest.py: it syncs clips.json from the * manifest, deletes the mp4 of every clip whose window MOVED (so the cut list * refills and those clips are re-cut), and prints the prose lines that cite * each clip ruled incorrect. Step 2 prints the corrections as markdown, which * is the form the next sweep's prompt wants. * * Nothing rewrites prose. That is the point. */ export function applyRulingsSteps(project) { return [ { cwd: project.dir, env: {}, label: "apply-manifest.py — sync clips.json, drop the mp4s of moved windows", argv: [PYTHON, path.join(project.dir, "apply-manifest.py")], timeoutMs: 10 * 60_000, }, { cwd: project.dir, env: {}, label: `umtool corrections ${project.id} → corrections.md`, // REDIRECTED, because the file is the artifact. finish-sweep.sh opens // with `umtool corrections > corrections.md` and the overnight // review reads that file; a job log somebody has to copy out of a // browser is not the same thing. // // `$0` is the destination and `"$@"` the command, both passed as // POSITIONAL arguments rather than interpolated into the script: a // project id and a directory can then contain anything at all without // becoming shell. `exec` keeps the command's own exit status. argv: [ "sh", "-c", 'exec "$@" > "$0"', path.join(project.dir, "corrections.md"), "node", tool("umtool.mjs"), "corrections", project.id, ], timeoutMs: 5 * 60_000, }, ]; } /** * Re-render every report variant the project carries. * * One step per content module, with build.py's own flags: bare `content` * renders to the default stem, and `content_` renders to `report-` so two * variants cannot overwrite each other's html. The list comes from the * directory (contentVariants), never from the client. * * @param {{ dir: string }} project * @param {{ module: string, out: string, argv: string[] }[]} variants */ export function rebuildReportSteps(project, variants) { return variants.map((v) => ({ cwd: project.dir, env: {}, label: `build.py → ${v.out}.{html,bbcode,md}`, argv: [PYTHON, path.join(project.dir, "build.py"), ...v.argv], timeoutMs: 15 * 60_000, })); }