commit 3e5791ac2416dc60d6ad7debe1b049d6e929ca4d
parent 922071fc240f6564c9665f387b9dea83c2f26ec2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 20 Sep 2026 19:27:04 -0400
umtool: ask the editor for a clip window instead of running yt-dlp
The bench's "fetch more" ran build-video.mjs --fetch-only, which meant no
cookie policy, no per-platform sleeps, no 429 cooldown, and bytes in one
project's out/clips-raw that the next report could not see. It now posts to the
editor, which fetches through its managed path into the corpus beside the
video. UMTOOL_LOCAL_FETCH=1 keeps the old path for a machine with no editor to
ask; the step shape, the argv flags and the NDJSON events are identical, so
nothing downstream can tell which one ran.
The cache had to learn to look there. rawCacheOf takes extraDirs — read once,
so windows() stays synchronous and every caller is unchanged — and a corpus
clips dir is scanned with the un-prefixed <from>-<to>.mp4 shape, since that
directory is already per-video. projectCache() is the door: a bare
rawCacheOf(project.dir) sees only out/clips-raw, so a window the editor fetched
would read as "nothing fetched for this clip yet" and the bench would offer to
download it again.
CHANNELS_DIR joins READ_ROOTS and deliberately not WRITE_ROOTS: /api/report/raw
resolves through resolveInRoots, so without it a window that is right there
400s as "outside the roots" — and the corpus is production data nothing here
should be able to render over.
A window carries its own `path` now, because it is no longer always under
out/clips-raw and rebuilding the path from the name names a file that is not
there.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
9 files changed, 448 insertions(+), 22 deletions(-)
diff --git a/umtool/app/api/report/fetch/route.ts b/umtool/app/api/report/fetch/route.ts
@@ -1,6 +1,11 @@
import { getJob, jobView, runningJob, startJob } from "@/lib/jobs";
-import { FETCH_MAX_PAD, fetchSteps } from "@/lib/report/driver.mjs";
-import { resolveClip, windowsFor } from "@/lib/report/serve.mjs";
+import {
+ FETCH_MAX_PAD,
+ editorFetchSteps,
+ fetchSteps,
+ localFetch,
+} from "@/lib/report/driver.mjs";
+import { projectCache, resolveClip, windowsFor } from "@/lib/report/serve.mjs";
export const dynamic = "force-dynamic";
@@ -42,7 +47,15 @@ export async function POST(request: Request) {
// or at the pad 10 a prior CLI run used -- downloads nothing, exits 0, and
// the bench reported "fetched" over a cache that had not moved. A no-op is
// not a success, and saying so here is cheaper than a job that proves it.
- const windows = (await windowsFor(r.project, r.clip)) as { from: number; to: number }[];
+ // Through projectCache, so a window the EDITOR already fetched into the
+ // corpus counts as cached. Without it this route would queue a job to
+ // download bytes that are on the disk, and the bench would show a fetch that
+ // moved nothing.
+ const windows = (await windowsFor(
+ r.project,
+ r.clip,
+ await projectCache(r.project, r.manifest),
+ )) as { from: number; to: number }[];
const want = { from: Number(r.clip.start) - padBefore, to: Number(r.clip.end) + padAfter };
const holds = windows.find((w) => w.from <= want.from + WIN_EPS && w.to >= want.to - WIN_EPS);
if (holds) {
@@ -74,11 +87,19 @@ export async function POST(request: Request) {
);
}
- const job = startJob(
- `fetch ${projectId}/${clipId}`,
- fetchSteps(r.project, clipId, { padBefore, padAfter }),
- { project: r.project.id },
- );
+ // ASK THE EDITOR, unless this instance has been told to fetch locally.
+ //
+ // The editor's fetch is the managed one: cookie policy, per-platform sleeps,
+ // the 429 cooldown and its backoff, a job log, and bytes that land in the
+ // corpus where the next report reuses them instead of paying again. Running
+ // yt-dlp from here has none of that, which is why it is now the opt-out
+ // (UMTOOL_LOCAL_FETCH=1) rather than the default.
+ const steps = localFetch()
+ ? fetchSteps(r.project, clipId, { padBefore, padAfter })
+ : editorFetchSteps(r.project, clipId, { padBefore, padAfter });
+ const job = startJob(`fetch ${projectId}/${clipId}`, steps, {
+ project: r.project.id,
+ });
// Which side this run is growing, so the caller can say so rather than
// guessing from a file name.
return Response.json({ job: jobView(job), padBefore, padAfter }, { status: 202 });
diff --git a/umtool/components/projects/ClipBenchPage.tsx b/umtool/components/projects/ClipBenchPage.tsx
@@ -1,4 +1,3 @@
-import path from "node:path";
import { notFound } from "next/navigation";
import BrowseHeader from "@/components/BrowseHeader";
import ClipBench, { type ClipBenchData } from "./ClipBench";
@@ -11,7 +10,7 @@ import {
siblingsOf,
walkReadiness,
} from "@/lib/projects/report.mjs";
-import { rawCacheOf, segmentFor, windowsFor } from "@/lib/report/serve.mjs";
+import { projectCache, segmentFor, windowsFor } from "@/lib/report/serve.mjs";
import { FETCH_MAX_PAD } from "@/lib/report/driver.mjs";
import type { ProjectRef } from "@/lib/project-types";
@@ -56,7 +55,7 @@ export default async function ClipBenchPage({
// One listing of clips-raw for the page: readClipDetail already built one to
// answer `fetched` for every clip, and this is the same question for this one.
- const windows = await windowsFor(project, entry, await rawCacheOf(project.dir));
+ const windows = await windowsFor(project, entry, await projectCache(project));
// The mtime rides along so a re-render busts the browser's cache: the segment
// is written back to the SAME path, and without it the player would keep
// showing the cut from before the edit.
@@ -137,7 +136,9 @@ export default async function ClipBenchPage({
// spec is relative to the file it names.
mixHref: widest
? `/mix?${new URLSearchParams({
- body: path.join(project.dir, "out", "clips-raw", widest.name),
+ // The window's own path; see ReportProject — a window may live in
+ // the corpus rather than under out/clips-raw.
+ body: widest.path,
start: (entry.start - widest.from).toFixed(2),
end: (entry.end - widest.from).toFixed(2),
from: project.id,
diff --git a/umtool/components/projects/ReportProject.tsx b/umtool/components/projects/ReportProject.tsx
@@ -144,12 +144,14 @@ export default async function ReportProject({
id: string;
start: number;
end: number;
- widest: { name: string; from: number } | null;
+ widest: { name: string; path: string; from: number } | null;
segment: string | null;
}): string | null => {
const q = new URLSearchParams();
if (e.widest) {
- q.set("body", path.join(project.dir, "out", "clips-raw", e.widest.name));
+ // The window's own path, not a rebuild of it: the editor fetches into
+ // the corpus, so "out/clips-raw/<name>" is a file that is not there.
+ q.set("body", e.widest.path);
q.set("start", (e.start - e.widest.from).toFixed(2));
q.set("end", (e.end - e.widest.from).toFixed(2));
} else if (e.segment) {
diff --git a/umtool/lib/paths.mjs b/umtool/lib/paths.mjs
@@ -65,10 +65,31 @@ const dedupe = (list) => [...new Set(list.map((p) => path.resolve(p)))];
// `quartering-uh-song/videos/<song>/wide.mp4`. labelFor and resolveInRoots read
// the same ordered list, which is what keeps a label a round-trip.
// ---------------------------------------------------------------------------
+// ---------------------------------------------------------------------------
+// The CORPUS is readable, and only readable.
+//
+// The editor fetches a clip window into channels/<slug>/data/<id>/clips/, which
+// is the point: one managed, provenanced fetch that every tool reuses. But
+// /api/report/raw resolves the file it serves through resolveInRoots, so
+// without the corpus in READ_ROOTS a window that is right there 400s as
+// "outside the roots" and the bench plays nothing.
+//
+// It does NOT join WRITE_ROOTS, and that is the whole reason the two lists
+// exist: the corpus is real production data, and nothing here should be able to
+// render over it.
+//
+// Same env var the cue reader already uses (lib/projects/report.mjs
+// GLOBAL_CHANNELS_DIR), so a fixture that confines one confines both.
+// ---------------------------------------------------------------------------
+export const CHANNELS_DIR = path.resolve(
+ process.env.CHANNELS_DIR ??
+ "/home/user/Projects/yt-dlp-transcript-browser/transcripts/channels",
+);
+
export const READ_ROOTS = dedupe(
process.env.MIX_ROOTS
? process.env.MIX_ROOTS.split(":").filter(Boolean)
- : [SONG_REPORTS, REPORTS_ROOT, SONG_DATA, SONG_SCRATCH],
+ : [SONG_REPORTS, REPORTS_ROOT, SONG_DATA, SONG_SCRATCH, CHANNELS_DIR],
);
export const WRITE_ROOTS = dedupe(
diff --git a/umtool/lib/projects/report.mjs b/umtool/lib/projects/report.mjs
@@ -292,6 +292,31 @@ export async function readAvailability(dir) {
}
}
+/**
+ * The corpus directories that also hold windows of this manifest's videos.
+ *
+ * The editor fetches a clip window into channels/<slug>/data/<id>/clips/ --
+ * managed, polite, provenanced, and REUSABLE, which is the whole point: a
+ * window one report paid for is a window the next one does not. One entry per
+ * distinct (channel, video) the timeline cites; rawCacheOf reads each once.
+ */
+export const clipWindowDirs = (m, channelsDir) => {
+ const seen = new Set();
+ const out = [];
+ for (const e of clipsOf(m)) {
+ const chan = channelFor(m, e);
+ if (!chan || !e.video) continue;
+ const key = `${chan}/${e.video}`;
+ if (seen.has(key)) continue;
+ seen.add(key);
+ out.push({
+ video: e.video,
+ dir: path.join(channelsDir ?? GLOBAL_CHANNELS_DIR(), chan, "data", e.video, "clips"),
+ });
+ }
+ return out;
+};
+
export const cuePathFor = (m, e, channelsDir) => {
const chan = channelFor(m, e);
if (!chan) return null;
@@ -1007,7 +1032,7 @@ export async function readClipDetail(dir, { manifest = null } = {}) {
// ONE listing of out/clips-raw for the whole project. This used to be two
// readdirs PER CLIP -- the padded lookup and the full list -- so a forty-clip
// cut paid eighty directory reads to draw one page.
- const raw = await rawCacheOf(dir);
+ const raw = await rawCacheOf(dir, { extraDirs: clipWindowDirs(m, channelsDir) });
const segDirs = segmentDirs(path.join(dir, "out"));
const segLists = await Promise.all(segDirs.map((d) => readdir(d).catch(() => [])));
const segWhich = segLists.findIndex((l) => l.length);
@@ -1101,12 +1126,20 @@ export async function readClipDetail(dir, { manifest = null } = {}) {
endsSentence,
noPunctuation,
proposed,
- cached: cached ? { name: cached.name, from: cached.from, to: cached.to } : null,
+ // `path` rides along because a window is no longer always under
+ // out/clips-raw: the editor fetches into the corpus, and a caller that
+ // rebuilds the project-local path from the name names a file that is not
+ // there.
+ cached: cached
+ ? { name: cached.name, path: cached.path, from: cached.from, to: cached.to }
+ : null,
// `cached` is the BUILD's question (is the padded window on disk); this is
// the PLAYER's (can this clip be watched end to end right now), and it is
// what the walk skips on. Same predicate, different window.
fetched: raw.isFetched(e),
- widest: widest ? { name: widest.name, from: widest.from, to: widest.to } : null,
+ widest: widest
+ ? { name: widest.name, path: widest.path, from: widest.from, to: widest.to }
+ : null,
segment: segNames.has(`${e.id}.mp4`) ? path.posix.join(segRel, `${e.id}.mp4`) : null,
wantFrom: from,
wantTo: to,
diff --git a/umtool/lib/report/driver.mjs b/umtool/lib/report/driver.mjs
@@ -198,6 +198,57 @@ export function fetchSteps(project, clipId, pad) {
}
/**
+ * The same fetch, ASKED OF THE EDITOR.
+ *
+ * Beside fetchSteps rather than replacing it: a machine with no editor to ask
+ * still needs the local path, and UMTOOL_LOCAL_FETCH=1 is how it says so. The
+ * step shape, the argv flags and the NDJSON events are identical, so the job
+ * runner, the bench's progress readout and the cache predicate cannot tell the
+ * two apart — which is the only way "which one ran" stays an operator's
+ * decision rather than a fork in every consumer.
+ *
+ * The editor URL and the shared WORKER_TOKEN come from the ENVIRONMENT, never
+ * from the request: a client that could name the editor could name any host.
+ *
+ * @param {{ dir: string }} project
+ * @param {string} clipId
+ * @param {number | { padBefore: number, padAfter: number }} pad
+ * @returns {import("../trim").Step[]}
+ */
+export function editorFetchSteps(project, clipId, pad) {
+ const before = typeof pad === "object" && pad ? Number(pad.padBefore) : Number(pad);
+ const after = typeof pad === "object" && pad ? Number(pad.padAfter) : Number(pad);
+ return [
+ {
+ cwd: PIPELINE_DIR,
+ env: {
+ ARCHILYZER_EDITOR_URL: process.env.ARCHILYZER_EDITOR_URL ?? "",
+ WORKER_TOKEN: process.env.WORKER_TOKEN ?? "",
+ },
+ label: `ask the editor for ${clipId} with −${before}s / +${after}s of pad`,
+ argv: [
+ "node",
+ script("fetch-via-editor.mjs"),
+ path.join(project.dir, "video.manifest.json"),
+ "--fetch-only",
+ clipId,
+ "--pad-before",
+ String(before),
+ "--pad-after",
+ String(after),
+ "--progress",
+ "ndjson",
+ ],
+ ndjson: true,
+ timeoutMs: 15 * 60_000,
+ },
+ ];
+}
+
+/** Whether this instance fetches locally with yt-dlp instead of asking. */
+export const localFetch = () => process.env.UMTOOL_LOCAL_FETCH === "1";
+
+/**
* One preflight per project, in order. What the check-sources job runs: the
* driver's own step 1, reused as-is, so `out/availability.json` is written by
* the same script whether a build or a re-check asked for it.
diff --git a/umtool/lib/report/raw-cache.mjs b/umtool/lib/report/raw-cache.mjs
@@ -13,6 +13,27 @@ import {
windowsFromNames,
} from "umtool-report-to-video/build-video";
+// A window file inside a CORPUS clips dir. Un-prefixed, because that directory
+// is already per-video: channels/<slug>/data/<id>/clips/<from>-<to>.mp4. The
+// project-local cache prefixes with the video id because one flat directory
+// holds every source. Same numbers, two namings, one predicate over both.
+const BARE_WINDOW_RE = /^(\d+(?:\.\d+)?)-(\d+(?:\.\d+)?)\.mp4$/;
+
+function windowsFromBareNames(names, dir) {
+ const out = [];
+ for (const name of names) {
+ const m = BARE_WINDOW_RE.exec(name);
+ if (!m) continue;
+ const from = Number(m[1]);
+ const to = Number(m[2]);
+ if (!(to > from)) continue;
+ out.push({ name, path: path.join(dir, name), from, to });
+ }
+ return out;
+}
+
+const SOURCE_MEDIA_RE = /^source-media\.[A-Za-z0-9]+$/;
+
/**
* The cache. The directory is keyed by VIDEO and a page asks about it per clip
* -- the bench page did two readdirs per clip, and a forty-clip cut paid eighty
@@ -26,12 +47,41 @@ import {
* mere overlap -- a half-covered clip cannot be watched through, so it is not
* ready to judge.
*/
-export async function rawCacheOf(projectDir) {
+/**
+ * @param {string} projectDir
+ * @param {{ extraDirs?: { video: string, dir: string, duration?: number }[] }} [opts]
+ * Directories OUTSIDE the project that also hold windows of a given video --
+ * today, the corpus clips dir the editor fetches into. Each is read once,
+ * here, so `windows()` stays synchronous and every existing caller is
+ * unchanged. A `duration` makes a whole `source-media.*` container in that
+ * directory count as the window [0, duration]: a full source contains every
+ * window of its video, which is exactly what the predicate needs to know.
+ */
+export async function rawCacheOf(projectDir, { extraDirs = [] } = {}) {
const rawDir = path.join(projectDir, "out", "clips-raw");
const names = await listRawNames(rawDir);
+ const extra = new Map();
+ for (const e of extraDirs) {
+ if (!e?.video || !e?.dir) continue;
+ const listed = await listRawNames(e.dir);
+ const found = windowsFromBareNames(listed, e.dir);
+ if (Number.isFinite(e.duration)) {
+ const whole = listed.find((n) => SOURCE_MEDIA_RE.test(n));
+ if (whole) {
+ found.push({ name: whole, path: path.join(e.dir, whole), from: 0, to: Number(e.duration) });
+ }
+ }
+ if (!found.length) continue;
+ extra.set(e.video, [...(extra.get(e.video) ?? []), ...found]);
+ }
const byVideo = new Map();
const windows = (video) => {
- if (!byVideo.has(video)) byVideo.set(video, windowsFromNames(names, rawDir, video));
+ if (!byVideo.has(video)) {
+ byVideo.set(video, [
+ ...windowsFromNames(names, rawDir, video),
+ ...(extra.get(video) ?? []),
+ ]);
+ }
return byVideo.get(video);
};
return {
diff --git a/umtool/lib/report/serve.mjs b/umtool/lib/report/serve.mjs
@@ -11,7 +11,13 @@ import { REPORTS_ROOT, resolveInRoots } from "../paths.mjs";
import { walkProjects } from "../projects/walk.mjs";
import { DEFAULT_VARIANT, WIN_EPS } from "umtool-report-to-video/build-video";
import { rawCacheOf } from "./raw-cache.mjs";
-import { clipsOf, readManifest } from "../projects/report.mjs";
+import {
+ channelsDirFor,
+ clipWindowDirs,
+ clipsOf,
+ hasShadowChannels,
+ readManifest,
+} from "../projects/report.mjs";
export async function resolveClip(projectId, clipId) {
const projects = await walkProjects(REPORTS_ROOT);
@@ -28,6 +34,22 @@ export async function resolveClip(projectId, clipId) {
export { rawCacheOf } from "./raw-cache.mjs";
/**
+ * The cache for a project, INCLUDING the corpus clips dirs the editor fetches
+ * into.
+ *
+ * Use this rather than rawCacheOf(project.dir) anywhere a clip's cached windows
+ * are the question. A bare rawCacheOf sees only out/clips-raw, so a window the
+ * editor fetched reads as "nothing fetched for this clip yet" and the bench
+ * offers to download bytes that are already on the disk.
+ */
+export async function projectCache(project, manifest = null) {
+ const m = manifest ?? (await readManifest(project.dir));
+ const shadowExists = await hasShadowChannels(project.dir);
+ const channelsDir = channelsDirFor(project.dir, m, { shadowExists });
+ return rawCacheOf(project.dir, { extraDirs: clipWindowDirs(m, channelsDir) });
+}
+
+/**
* The cached source windows for a clip: the ones that hold ITS window, widest
* first, then any that merely overlap it.
*
@@ -52,7 +74,7 @@ export { rawCacheOf } from "./raw-cache.mjs";
export async function windowsFor(project, clip, cache = null) {
// The cache is optional and passed in by callers that already have one (the
// bench page asks for every clip), so the directory is read once per request.
- const c = cache ?? (await rawCacheOf(project.dir));
+ const c = cache ?? (await projectCache(project));
const all = c.windows(clip.video);
const width = (w) => w.to - w.from;
const { start, end } = clip;
diff --git a/umtool/report-to-video/fetch-via-editor.mjs b/umtool/report-to-video/fetch-via-editor.mjs
@@ -0,0 +1,225 @@
+#!/usr/bin/env node
+// Fetch one clip's window BY ASKING THE EDITOR, not by running yt-dlp.
+//
+// A drop-in for `build-video.mjs --fetch-only`: same argv shape, same NDJSON
+// events, same exit convention — so the driver can swap one step for the other
+// and nothing downstream knows which ran.
+//
+// WHY IT EXISTS. The operator's rule is that no fetch happens by hand. Running
+// yt-dlp from here means no cookie policy, no per-platform sleeps, no 429
+// cooldown, and bytes that land in one project's out/clips-raw where the next
+// report cannot see them. Asking the editor means all four, and the window ends
+// up in the corpus beside the video with a note saying who wanted it and why.
+//
+// It never names a file: the editor decides where the bytes go and tells us.
+// What comes back is a path inside the corpus, which lib/paths.mjs admits as a
+// READ root (and deliberately not a write one).
+//
+// The local path is still there behind UMTOOL_LOCAL_FETCH=1 — for a machine
+// with no editor to ask.
+
+import path from "node:path";
+import { readFile } from "node:fs/promises";
+
+const DEFAULT_EDITOR = "http://localhost:3001";
+const POLL_MS = 1000;
+// A window is seconds of media; a queue behind a channel sync is not. Generous,
+// and the job keeps running on the editor if we give up — nothing is lost, the
+// next ask finds it cached.
+const POLL_TIMEOUT_MS = 10 * 60_000;
+
+const argv = process.argv.slice(2);
+const flag = (name) => {
+ const i = argv.indexOf(name);
+ return i < 0 ? undefined : argv[i + 1];
+};
+
+let EMIT = (ev, fields = {}) => {
+ if (ev === "fetch" && !fields.cached) {
+ console.log(` fetch ${fields.id}: ${fields.video} ${fields.from}–${fields.to}`);
+ } else if (ev === "note") {
+ console.log(fields.message);
+ } else if (ev === "done" && fields.out) {
+ console.log(`fetched ${fields.out}`);
+ }
+};
+if (flag("--progress") === "ndjson") {
+ EMIT = (ev, fields = {}) =>
+ process.stdout.write(JSON.stringify({ ev, ...fields }) + "\n");
+}
+
+function die(message) {
+ EMIT("note", { message });
+ process.stderr.write(`${message}\n`);
+ process.exit(1);
+}
+
+const manifestPath = argv.find((a) => !a.startsWith("-") && a.endsWith(".json"));
+const clipId = flag("--fetch-only") ?? flag("--clip");
+if (!manifestPath || !clipId) {
+ die("usage: fetch-via-editor.mjs <manifest.json> --fetch-only <clipId> [--pad-before N] [--pad-after N] [--progress ndjson]");
+}
+
+const editorUrl = (process.env.ARCHILYZER_EDITOR_URL ?? DEFAULT_EDITOR).replace(/\/+$/, "");
+const token = process.env.WORKER_TOKEN ?? "";
+if (!token) {
+ // NAME THE VARIABLE. "401 Unauthorized" from an endpoint the operator has
+ // never heard of is a twenty-minute detour; this is a ten-second one.
+ die(
+ "WORKER_TOKEN is not set, so the editor cannot be asked for this window. " +
+ "Set WORKER_TOKEN to the same value the editor runs with (or set " +
+ "UMTOOL_LOCAL_FETCH=1 to fetch locally with yt-dlp instead).",
+ );
+}
+
+const whole = JSON.parse(await readFile(manifestPath, "utf8"));
+const provenance = whole.provenance ?? {};
+
+// Same resolution build-video.mjs's --fetch-only does, and for the same
+// reasons: a still has nothing to fetch, a non-clip entry is an error, and a
+// LEDGER CLAIM is a moment rather than a window (most of a ledger is cited by
+// no clip at all, and adjudicating one means listening around it).
+let entry = (whole.timeline ?? []).find((e) => e.id === clipId);
+if (entry?.type === "image") {
+ EMIT("note", { message: `${clipId} is an image entry — nothing to fetch` });
+ EMIT("done", { out: null, nothingToFetch: true });
+ process.exit(0);
+}
+if (entry && entry.type !== "clip") {
+ die(`${clipId} is a ${entry.type ?? "non-clip"} entry, not a clip`);
+}
+if (!entry) {
+ const claim = (whole.ledger ?? []).find((e) => e.id === clipId);
+ if (!claim) die(`no timeline entry or ledger claim with id ${clipId}`);
+ if (!claim.video) die(`ledger claim ${clipId} has no \`video\` to fetch`);
+ const at = Number(claim.cite);
+ if (!Number.isFinite(at)) die(`ledger claim ${clipId} has no \`cite\` second`);
+ entry = {
+ id: claim.id,
+ video: claim.video,
+ channel: claim.channel ?? null,
+ start: Math.max(0, at - 1),
+ end: at + 1,
+ note: claim.text ?? claim.claim ?? null,
+ };
+}
+
+const channelSlug = entry.channel ?? provenance.channelSlug;
+if (!channelSlug) {
+ die(`${clipId} has no channel, and the manifest's provenance names none`);
+}
+
+const pad = Number(flag("--pad") ?? 3);
+const padBefore = Number(flag("--pad-before") ?? pad);
+const padAfter = Number(flag("--pad-after") ?? pad);
+// TWO DECIMALS, matching the editor's own naming (common/lib/clipWindow.ts) and
+// the build's. The name IS the window, so a request that rounds differently
+// addresses a different file and the cache misses forever.
+const from = Number(Math.max(0, Number(entry.start) - padBefore).toFixed(2));
+const to = Number((Number(entry.end) + padAfter).toFixed(2));
+
+// WHY THESE SECONDS. Stored beside the file so a directory of windows can be
+// read back months later. The clip's own note is the closest thing the manifest
+// has to a reason; the manifest id and the clip id say the rest.
+const manifestId = provenance.manifestId ?? path.basename(path.dirname(path.resolve(manifestPath)));
+const reason =
+ entry.note ??
+ entry.quote ??
+ `clip window with ${padBefore}s before / ${padAfter}s after`;
+
+const headers = {
+ authorization: `Bearer ${token}`,
+ "content-type": "application/json",
+};
+
+async function ask(url, init) {
+ try {
+ return await fetch(url, init);
+ } catch (err) {
+ die(`could not reach the editor at ${editorUrl}: ${err.message}`);
+ }
+}
+
+EMIT("fetch", { id: entry.id, video: entry.video, from, to, cached: false });
+
+const res = await ask(`${editorUrl}/api/media/fetch-window`, {
+ method: "POST",
+ headers,
+ body: JSON.stringify({
+ channelSlug,
+ videoId: entry.video,
+ webpageUrl: entry.webpageUrl ?? undefined,
+ from,
+ to,
+ requestedBy: "umtool",
+ manifest: manifestId,
+ clipId: entry.id,
+ reason: String(reason).slice(0, 400),
+ pad: Math.max(padBefore, padAfter),
+ }),
+});
+
+const body = await res.json().catch(() => ({}));
+
+if (res.status === 200) {
+ // Already on disk — possibly WIDER than asked for, which is the point of
+ // containing-window reuse. `fetchStart` is the file's own start, because
+ // every cut downstream is expressed relative to it.
+ EMIT("fetch", {
+ id: entry.id,
+ video: entry.video,
+ from: body.from,
+ to: body.to,
+ cached: true,
+ reuse: path.basename(String(body.file ?? "")),
+ });
+ EMIT("done", { out: body.file, fetchStart: body.from, cached: true });
+ process.exit(0);
+}
+
+if (res.status === 409) {
+ // The per-platform cooldown. SAY THE SECONDS: "rate limited" with no number
+ // is advice to keep pressing the button.
+ const secs = Math.ceil(Number(body.cooldownMs ?? 0) / 1000);
+ die(
+ `the editor is in a ${body.platform ?? "platform"} rate-limit cooldown — ` +
+ `${secs}s remaining. ${body.error ?? ""}`.trim(),
+ );
+}
+
+if (res.status !== 202) {
+ // Everything else — an unreachable drive, a low-disk refusal, a bad token —
+ // is passed through VERBATIM. The editor's operator is the person who can
+ // act on it, and a paraphrase here is a sentence they cannot search for.
+ die(`the editor refused (HTTP ${res.status}): ${body.error ?? "no reason given"}`);
+}
+
+const jobId = body.jobId;
+EMIT("note", { message: ` queued on the editor as job ${jobId}` });
+
+const deadline = Date.now() + POLL_TIMEOUT_MS;
+let last = "";
+for (;;) {
+ if (Date.now() > deadline) {
+ die(
+ `gave up waiting for editor job ${jobId} after ${POLL_TIMEOUT_MS / 60_000} ` +
+ `minutes (it is still running there; ask again and it will be cached)`,
+ );
+ }
+ await new Promise((r) => setTimeout(r, POLL_MS));
+ const poll = await ask(`${editorUrl}/api/media/fetch-window/${jobId}`, { headers });
+ const j = await poll.json().catch(() => ({}));
+ if (!poll.ok) die(`polling editor job ${jobId} failed (HTTP ${poll.status}): ${j.error ?? ""}`);
+ if (j.status !== last) {
+ last = j.status;
+ EMIT("note", { message: ` editor job ${jobId}: ${j.status}` });
+ }
+ if (j.status === "done") {
+ EMIT("fetch", { id: entry.id, video: entry.video, from, to, cached: true });
+ EMIT("done", { out: j.file ?? null, fetchStart: j.from ?? from, cached: false });
+ process.exit(0);
+ }
+ if (j.status === "failed" || j.status === "cancelled") {
+ die(`editor job ${jobId} ${j.status}: ${j.error ?? "see the editor's /jobs log"}`);
+ }
+}