commit e7231d42d93739f222087cbb88c32b133b24018c
parent befc694178c5cb78eed418baa3c7c238cf8c9090
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 20 Sep 2026 19:37:43 -0400
tests: pin the no-info-json rule and the editor fetch end to end
editor/e2e/fetch-window.spec.ts fetches a window into a video that has NEVER
been downloaded and then proves the corpus still says so: no
metadata.info.json, and the channel's undownloaded count unmoved at four. That
is the load-bearing rule — the index keys a video's presence on that file — and
it is the one a well-meaning refactor would break silently.
It also pins the door (401 without the bearer), the argument checks (traversal
in a slug, a backwards window, a span past the cap, an anonymous request), the
provenance sidecar, containing-window reuse (a narrower ask answered by the
wider file, with no second invocation), and a 429 leaving no half-written file
and a cooldown the next ask is refused by.
The fake yt-dlp's new branch writes ONLY the window, deliberately: a fake that
also wrote a metadata.info.json would make the spec pass for the wrong reason.
umtool's suite gets a second webServer — the editor, stubbed. A real one would
arm its runners against whatever corpus it found, and the endpoint's job is to
spend somebody's bandwidth. What the stub records is the part a file cannot
prove: that the provenance and the token crossed the wire. The strongest
assertion in that spec is a negative one — the fixture's yt-dlp stub now logs
every invocation, and this path produces none.
Two fetch fixtures, not one: the corpus is keyed by video, so sharing vid1
would make the button's pending list depend on which spec ran first.
A step's `env` is echoed back to the browser by jobView(), so the editor step
passes none: the child inherits WORKER_TOKEN from the server's environment
instead of printing it on the page that started the job.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
9 files changed, 736 insertions(+), 9 deletions(-)
diff --git a/common/ytdlp/fetchWindowManaged.ts b/common/ytdlp/fetchWindowManaged.ts
@@ -84,6 +84,12 @@ export type FetchWindowOpts = {
videoDir: string;
videoId: string;
videoUrl: string;
+ // Working directory for the yt-dlp process. Nothing it writes is
+ // cwd-relative (`-o` is absolute and `--ignore-config` stops the operator's
+ // own config redirecting anything), so this only decides where yt-dlp's own
+ // scratch and any future cwd-relative trace land. The editor passes the
+ // channel root, which is where every other invocation in this repo runs.
+ cwd?: string;
from: number;
to: number;
provenance: FetchWindowProvenance;
@@ -187,7 +193,7 @@ export async function fetchWindowManaged(
onLog: opts.onLog,
signal: opts.signal,
},
- videoDir,
+ opts.cwd ?? videoDir,
argsWith(cookies),
);
diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts
@@ -865,6 +865,8 @@ export async function fetchWindowAction(req: {
videoDir,
videoId,
videoUrl: url,
+ // The channel root, as every other yt-dlp invocation here runs.
+ cwd: path.join(paths.channelsDir, slug),
from,
to,
provenance: req.provenance,
diff --git a/editor/e2e/fetch-window.spec.ts b/editor/e2e/fetch-window.spec.ts
@@ -0,0 +1,263 @@
+// Sourcing clip media for another tool: POST /api/media/fetch-window.
+//
+// The feature exists so umtool stops running yt-dlp itself, and the thing that
+// makes that safe is a rule the fake can't fake around: a window fetch writes
+// NO metadata.info.json, because the index keys a video's presence on that
+// file. So this spec fetches into a video that has never been downloaded and
+// then proves the corpus still says so.
+//
+// Same token as /api/worker/* (the test server runs with
+// WORKER_TOKEN=test-worker-token; see package.json dev:test).
+
+import { readFile, stat } from "node:fs/promises";
+import { test, expect, type APIRequestContext } from "@playwright/test";
+import { generateReport, resetData, resolvePath, writeSettings } from "./helpers";
+import { baseUrl } from "./baseUrl";
+
+const TOKEN = "test-worker-token";
+const AUTH = { authorization: `Bearer ${TOKEN}` };
+
+// In the playlist, with no data dir: undownloaded, which is the state the
+// no-info-json rule has to leave untouched.
+const SLUG = "test-youtube";
+const VIDEO = "fake00000002";
+const rel = (p: string) => `test-transcripts/channels/${SLUG}/${p}`;
+const clipRel = (name: string) => rel(`data/${VIDEO}/clips/${name}`);
+
+async function exists(relPath: string): Promise<boolean> {
+ try {
+ await stat(resolvePath(relPath));
+ return true;
+ } catch {
+ return false;
+ }
+}
+
+async function invocations(): Promise<string> {
+ try {
+ return await readFile(resolvePath(rel("fake-ytdlp.invocations")), "utf8");
+ } catch {
+ return "";
+ }
+}
+
+async function pollJob(
+ request: APIRequestContext,
+ jobId: string,
+): Promise<Record<string, unknown>> {
+ let last: Record<string, unknown> = {};
+ await expect
+ .poll(
+ async () => {
+ const r = await request.get(
+ `${baseUrl}/api/media/fetch-window/${jobId}`,
+ { headers: AUTH },
+ );
+ last = (await r.json()) as Record<string, unknown>;
+ return last.status as string;
+ },
+ { timeout: 30_000 },
+ )
+ .not.toMatch(/^(queued|running)$/);
+ return last;
+}
+
+test.beforeEach(async () => {
+ await resetData("youtube-with-playlist");
+ // NAME THE FLOOR. A spec that leaves it at the product default inherits 5 GB
+ // and then depends on how much room this host has left — a refusal that
+ // arrives as a returned { ok: false } no assertion reads.
+ await writeSettings({ minFreeDiskGB: 0 });
+});
+
+test("the endpoint is behind the worker token", async ({ request }) => {
+ const noAuth = await request.post(`${baseUrl}/api/media/fetch-window`, {
+ data: {
+ channelSlug: SLUG,
+ videoId: VIDEO,
+ from: 12,
+ to: 42,
+ requestedBy: "umtool",
+ },
+ });
+ expect(noAuth.status()).toBe(401);
+});
+
+test("a window is refused unless it names who asked and a sane span", async ({
+ request,
+}) => {
+ const cases: Array<[Record<string, unknown>, string]> = [
+ [{ channelSlug: "../escape", videoId: VIDEO, from: 1, to: 2, requestedBy: "umtool" }, "traversal in the slug"],
+ [{ channelSlug: SLUG, videoId: "a/b", from: 1, to: 2, requestedBy: "umtool" }, "separator in the id"],
+ [{ channelSlug: SLUG, videoId: VIDEO, from: 42, to: 12, requestedBy: "umtool" }, "backwards window"],
+ [{ channelSlug: SLUG, videoId: VIDEO, from: 0, to: 1000, requestedBy: "umtool" }, "past the span cap"],
+ [{ channelSlug: SLUG, videoId: VIDEO, from: 1, to: 2 }, "anonymous"],
+ ];
+ for (const [data, why] of cases) {
+ const r = await request.post(`${baseUrl}/api/media/fetch-window`, {
+ headers: AUTH,
+ data,
+ });
+ expect(r.status(), why).toBe(400);
+ }
+});
+
+test("a fetched window lands in clips/, carries its provenance, and leaves the video undownloaded", async ({
+ page,
+ request,
+}) => {
+ test.setTimeout(90_000);
+ const post = await request.post(`${baseUrl}/api/media/fetch-window`, {
+ headers: AUTH,
+ data: {
+ channelSlug: SLUG,
+ videoId: VIDEO,
+ from: 12,
+ to: 42,
+ requestedBy: "umtool",
+ manifest: "demo-report",
+ clipId: "c03",
+ reason: "the clip ends mid-sentence",
+ pad: 20,
+ },
+ });
+ expect(post.status()).toBe(202);
+ const queued = (await post.json()) as { jobId: string; file: string };
+ expect(queued.file).toContain(`data/${VIDEO}/clips/12.00-42.00.mp4`);
+
+ const finished = await pollJob(request, queued.jobId);
+ expect(finished.status, JSON.stringify(finished)).toBe("done");
+ expect(finished.file).toContain("12.00-42.00.mp4");
+ expect(Number(finished.bytes)).toBeGreaterThan(0);
+
+ // The bytes, under the name that IS the window.
+ expect(await exists(clipRel("12.00-42.00.mp4"))).toBe(true);
+
+ // And the note saying who wanted them. Without this a directory of windows
+ // is bytes nobody can account for in six months.
+ const sidecar = JSON.parse(
+ await readFile(resolvePath(clipRel("12.00-42.00.json")), "utf8"),
+ ) as Record<string, unknown>;
+ expect(sidecar.requestedBy).toBe("umtool");
+ expect(sidecar.manifest).toBe("demo-report");
+ expect(sidecar.clipId).toBe("c03");
+ expect(sidecar.reason).toBe("the clip ends mid-sentence");
+ expect(sidecar.pad).toBe(20);
+ expect(Number(sidecar.bytes)).toBeGreaterThan(0);
+ expect(typeof sidecar.fetchedAt).toBe("string");
+
+ // yt-dlp was asked for THAT span, with the H.264 pin, and nothing else.
+ const inv = await invocations();
+ expect(inv).toContain("download-sections:*12.00-42.00");
+ expect(inv).toContain("avc1");
+
+ // THE LOAD-BEARING RULE. A window carries no metadata, so the index has no
+ // reason to think this video arrived.
+ expect(await exists(rel(`data/${VIDEO}/metadata.info.json`))).toBe(false);
+
+ // The page says who asked.
+ await page.goto(`/channels/${SLUG}/videos/${VIDEO}`);
+ await page.getByLabel("Fetched windows stage summary").click();
+ await expect(
+ page.getByText(/requested by umtool for demo-report\/c03/),
+ ).toBeVisible();
+ await expect(
+ page.getByText(/Fetched windows do not make this video downloaded/i),
+ ).toBeVisible();
+
+ // And the corpus still counts it as undownloaded: four of the playlist's
+ // five, exactly as before the fetch.
+ await generateReport(page, SLUG);
+ await page.goto("/operations/download");
+ const row = page
+ .getByRole("region", { name: "undownloaded", exact: true })
+ .getByLabel(`undownloaded row ${SLUG}`);
+ await expect(row).toContainText("4");
+
+ // A SECOND ASK IS FREE. Containing-window reuse is what makes one generous
+ // fetch the next tool's cache rather than a second download.
+ const before = (await invocations()).length;
+ const again = await request.post(`${baseUrl}/api/media/fetch-window`, {
+ headers: AUTH,
+ data: {
+ channelSlug: SLUG,
+ videoId: VIDEO,
+ // INSIDE the fetched window, not equal to it: the point is that a
+ // narrower ask is answered by a wider file.
+ from: 15,
+ to: 30,
+ requestedBy: "umtool",
+ manifest: "demo-report",
+ clipId: "c03",
+ reason: "same seconds",
+ },
+ });
+ expect(again.status()).toBe(200);
+ const cached = (await again.json()) as {
+ cached: boolean;
+ file: string;
+ from: number;
+ to: number;
+ };
+ expect(cached.cached).toBe(true);
+ expect(cached.file).toContain("12.00-42.00.mp4");
+ // The FILE's window, not the request's — every cut downstream is expressed
+ // relative to it.
+ expect(cached.from).toBe(12);
+ expect(cached.to).toBe(42);
+ expect((await invocations()).length).toBe(before);
+});
+
+test("a 429 fails the job and puts the platform in cooldown", async ({
+ request,
+}) => {
+ test.setTimeout(60_000);
+ const post = await request.post(`${baseUrl}/api/media/fetch-window`, {
+ headers: AUTH,
+ data: {
+ channelSlug: SLUG,
+ videoId: "ratelimitvid1",
+ // The fake reads the sentinel out of the URL, so the request names one.
+ webpageUrl: "https://www.youtube.com/watch?v=ratelimitvid1",
+ from: 5,
+ to: 10,
+ requestedBy: "umtool",
+ manifest: "demo-report",
+ clipId: "c01",
+ reason: "checking the backoff",
+ },
+ });
+ expect(post.status()).toBe(202);
+ const { jobId } = (await post.json()) as { jobId: string };
+ const finished = await pollJob(request, jobId);
+ expect(finished.status).toBe("failed");
+ expect(String(finished.error)).toMatch(/429|Too Many Requests/i);
+
+ // Nothing half-written is left under the window's name.
+ expect(
+ await exists(`test-transcripts/channels/${SLUG}/data/ratelimitvid1/clips/5.00-10.00.mp4`),
+ ).toBe(false);
+
+ // The cooldown the auto-download runner and a clicked sync both honour. The
+ // next ask is refused at the door rather than re-storming the source.
+ const refused = await request.post(`${baseUrl}/api/media/fetch-window`, {
+ headers: AUTH,
+ data: {
+ channelSlug: SLUG,
+ videoId: VIDEO,
+ from: 12,
+ to: 42,
+ requestedBy: "umtool",
+ manifest: "demo-report",
+ clipId: "c03",
+ reason: "should be refused",
+ },
+ });
+ expect(refused.status()).toBe(409);
+ const body = (await refused.json()) as {
+ cooldownMs: number;
+ platform: string;
+ };
+ expect(body.platform).toBe("youtube");
+ expect(body.cooldownMs).toBeGreaterThan(0);
+});
diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs
@@ -492,6 +492,38 @@ async function main() {
process.exit(101);
}
+ // CLIP WINDOW (fetchWindowManaged): --download-sections *<from>-<to> with an
+ // absolute -o. Branched FIRST because nothing else passes --download-sections
+ // and because its -o is a dotfile .part under data/<id>/clips/, which no
+ // other mode's shape test would recognise.
+ //
+ // It writes ONLY that file: the whole point of the feature is that a window
+ // fetch leaves no metadata.info.json behind (that would put an undownloaded
+ // video into the index), and a fake that wrote one would make the spec that
+ // pins the rule pass for the wrong reason.
+ if (has("--download-sections") && arg("-o")) {
+ const dest = arg("-o");
+ const url = lastNonFlag() ?? "";
+ await appendFile(
+ "fake-ytdlp.invocations",
+ `download-sections:${arg("--download-sections")} dest=${dest} ` +
+ `fmt=${arg("-f") ?? ""} cookies=${cookieArg()}\n`,
+ );
+ if (cookieGateBlocked(url)) failCookieGate(url);
+ if (url.toLowerCase().includes("ratelimit")) {
+ process.stderr.write(
+ `ERROR: [youtube] ${url}: HTTP Error 429: Too Many Requests\n`,
+ );
+ process.exit(1);
+ }
+ await ensureDir(path.dirname(dest));
+ // Deterministic bytes, one chunk's worth, so a size assertion is stable.
+ await writeFile(dest, Buffer.alloc(CHUNK_BYTES, "w"));
+ process.stdout.write(`[download] section ${arg("--download-sections")}\n`);
+ process.stdout.write(`[fake-ytdlp] window complete\n`);
+ return;
+ }
+
if (
has("--flat-playlist") &&
has("--skip-download") &&
diff --git a/umtool/e2e/fixtures/editor-stub.mjs b/umtool/e2e/fixtures/editor-stub.mjs
@@ -0,0 +1,170 @@
+#!/usr/bin/env node
+// A stand-in for the editor's POST /api/media/fetch-window.
+//
+// The suite must not run a real editor: it would boot the runners against
+// whatever corpus it found, and the whole point of the endpoint is that it
+// spends somebody's bandwidth. So this answers the contract — bearer token,
+// 202 + jobId, poll to done — and writes a real mp4 into the fixture's corpus
+// at the path the editor would have used:
+//
+// <CHANNELS_DIR>/<slug>/data/<video>/clips/<from>-<to>.mp4
+//
+// It records every request so a spec can assert the provenance actually
+// crossed the wire, which is the part that cannot be inferred from the file.
+//
+// Modelled on e2e/fixtures/ollama-stub.mjs in the editor: a second webServer
+// entry, ready when its own endpoint answers.
+
+import http from "node:http";
+import { spawnSync } from "node:child_process";
+import { mkdirSync, writeFileSync, statSync } from "node:fs";
+import path from "node:path";
+
+const PORT = Number(process.env.EDITOR_STUB_PORT ?? 3052);
+const CHANNELS = process.env.CHANNELS_DIR ?? "";
+const TOKEN = process.env.WORKER_TOKEN ?? "";
+// Where the spec reads what was asked for.
+const LOG = process.env.EDITOR_STUB_LOG ?? path.join(process.cwd(), "editor-stub.requests.json");
+
+const requests = [];
+const jobs = new Map();
+let nextJob = 1;
+
+function record() {
+ writeFileSync(LOG, JSON.stringify(requests, null, 2) + "\n");
+}
+record();
+
+function json(res, status, body) {
+ const s = JSON.stringify(body);
+ res.writeHead(status, {
+ "content-type": "application/json",
+ "content-length": Buffer.byteLength(s),
+ });
+ res.end(s);
+}
+
+// A real container, because the bench serves it with byte ranges and the
+// project page probes it. Same generator the fixture's yt-dlp stub uses, so a
+// window fetched here and one fetched locally are the same kind of file.
+function writeWindow(dir, name, seconds) {
+ mkdirSync(dir, { recursive: true });
+ const out = path.join(dir, name);
+ const dur = Math.max(1, seconds);
+ const r = spawnSync(
+ "ffmpeg",
+ [
+ "-nostdin", "-v", "error", "-y",
+ "-f", "lavfi", "-i", `color=c=darkblue:size=320x180:rate=15:duration=${dur}`,
+ "-f", "lavfi", "-i", `sine=frequency=330:duration=${dur}`,
+ "-t", String(dur),
+ "-c:v", "libx264", "-pix_fmt", "yuv420p", "-c:a", "aac",
+ "-ar", "48000", "-ac", "2", out,
+ ],
+ { stdio: "inherit" },
+ );
+ if (r.status !== 0) throw new Error(`ffmpeg failed writing ${out}`);
+ return out;
+}
+
+const server = http.createServer((req, res) => {
+ const url = new URL(req.url ?? "/", `http://127.0.0.1:${PORT}`);
+
+ if (url.pathname === "/health") return json(res, 200, { ok: true });
+
+ const auth = req.headers.authorization ?? "";
+ if (!TOKEN || auth !== `Bearer ${TOKEN}`) {
+ req.resume();
+ return json(res, 401, { error: "missing bearer token" });
+ }
+
+ // GET /api/media/fetch-window/<jobId>
+ const poll = /^\/api\/media\/fetch-window\/(.+)$/.exec(url.pathname);
+ if (req.method === "GET" && poll) {
+ req.resume();
+ const job = jobs.get(poll[1]);
+ if (!job) return json(res, 404, { error: "no such job" });
+ // One poll of latency, so the spec exercises the queued -> done path
+ // rather than only the already-finished one.
+ if (!job.polled) {
+ job.polled = true;
+ return json(res, 200, { status: "running", jobId: poll[1] });
+ }
+ if (job.fail) {
+ return json(res, 200, { status: "failed", jobId: poll[1], error: job.fail });
+ }
+ return json(res, 200, {
+ status: "done",
+ jobId: poll[1],
+ file: job.file,
+ bytes: statSync(job.file).size,
+ from: job.from,
+ to: job.to,
+ });
+ }
+
+ if (req.method !== "POST" || url.pathname !== "/api/media/fetch-window") {
+ req.resume();
+ return json(res, 404, { error: "not found" });
+ }
+
+ let raw = "";
+ req.on("data", (c) => (raw += c));
+ req.on("end", () => {
+ let body;
+ try {
+ body = JSON.parse(raw);
+ } catch {
+ return json(res, 400, { error: "malformed JSON body" });
+ }
+ requests.push({ ...body, authorization: auth });
+ record();
+
+ if (!body.requestedBy) {
+ return json(res, 400, { error: "requestedBy is required" });
+ }
+ // A sentinel so a spec can drive the refusal path without a real 429.
+ if (String(body.clipId ?? "").includes("cooldown")) {
+ return json(res, 409, {
+ error: "youtube is in a rate-limit cooldown (42s remaining).",
+ cooldownMs: 42_000,
+ platform: "youtube",
+ });
+ }
+
+ const dir = path.join(CHANNELS, body.channelSlug, "data", body.videoId, "clips");
+ const name = `${Number(body.from).toFixed(2)}-${Number(body.to).toFixed(2)}.mp4`;
+ const id = `stub-job-${nextJob++}`;
+ let file;
+ try {
+ file = writeWindow(dir, name, Number(body.to) - Number(body.from));
+ } catch (err) {
+ jobs.set(id, { fail: String(err.message), polled: false });
+ return json(res, 202, { cached: false, jobId: id, file: null, from: body.from, to: body.to });
+ }
+ // The provenance sidecar the editor writes, so a corpus built by the stub
+ // reads back the same way a real one does.
+ writeFileSync(
+ path.join(dir, name.replace(/\.mp4$/, ".json")),
+ JSON.stringify(
+ {
+ requestedBy: body.requestedBy,
+ manifest: body.manifest,
+ clipId: body.clipId,
+ reason: body.reason,
+ pad: body.pad,
+ bytes: statSync(file).size,
+ fetchedAt: new Date().toISOString(),
+ },
+ null,
+ 2,
+ ) + "\n",
+ );
+ jobs.set(id, { file, from: body.from, to: body.to, polled: false });
+ json(res, 202, { cached: false, jobId: id, file, from: body.from, to: body.to });
+ });
+});
+
+server.listen(PORT, "127.0.0.1", () => {
+ process.stdout.write(`[editor-stub] listening on ${PORT}, corpus ${CHANNELS}\n`);
+});
diff --git a/umtool/e2e/fixtures/make-fixture.mjs b/umtool/e2e/fixtures/make-fixture.mjs
@@ -871,6 +871,37 @@ const WALK = writeProject(
);
+// -- THE EDITOR-FETCH FIXTURE -------------------------------------------------
+//
+// Two clips on vid1 and NOTHING in out/clips-raw, so both read "not fetched
+// yet" and `ready 0 of 2`. That is the state the editor fetch exists to leave:
+// after it, the window is in the CORPUS (channels/testchan/data/vid1/clips/)
+// rather than in this project's own out/ — which is the whole argument, since
+// the next report citing vid1 gets it for free.
+//
+// Its own project for the same reason bench-fixture is: every test here writes
+// (a fetch, and then a corpus file), and sharing would make one suite's result
+// depend on the other's order.
+writeProject(
+ "editor-fetch-fixture",
+ manifest("editor-fetch-fixture", "The Editor Fetch Fixture", { siteOrigin: "https://archive.example" }, [
+ { type: "clip", id: "e01", video: "vid1", start: 3.0, end: 6.0, cite: 3, section: 0, lock: true, quote: "and because", note: "the speaker names the number here" },
+ ]),
+);
+
+// The SAME state, on a different source, for the "fetch every unfetched clip"
+// button. Two projects rather than two clips in one, because the first test
+// leaves a window in the corpus and the corpus is keyed by VIDEO: sharing vid1
+// would make the button's pending list depend on which spec ran first, which is
+// the order-dependence bench-fixture already exists to avoid.
+writeProject(
+ "editor-fetch-many-fixture",
+ manifest("editor-fetch-many-fixture", "The Editor Fetch Sweep Fixture", { siteOrigin: "https://archive.example" }, [
+ { type: "clip", id: "f01", video: "vid2", start: 0.0, end: 3.0, cite: 0, section: 0, lock: true, quote: "no punctuation" },
+ { type: "clip", id: "f02", video: "vid2", start: 3.0, end: 6.0, cite: 3, section: 0, lock: true, quote: "never emitted a full stop" },
+ ]),
+);
+
// -- STUB BINARIES, so a build is offline and deterministic --------------------
//
// The pipeline shells out to yt-dlp for the availability preflight and for every
@@ -894,12 +925,16 @@ writeFileSync(
`#!/usr/bin/env node
// Fixture stub for yt-dlp. Deterministic, offline.
import { spawnSync } from "node:child_process";
-import { mkdirSync } from "node:fs";
+import { appendFileSync, mkdirSync } from "node:fs";
import path from "node:path";
const argv = process.argv.slice(2);
const all = argv.join(" ");
+// EVERY invocation is logged, because one spec's assertion is that there were
+// ZERO of them: the editor fetch must not shell out to yt-dlp here.
+appendFileSync(${JSON.stringify(path.join(BIN, "yt-dlp.invocations"))}, all + "\\n");
+
if (argv.includes("--version")) {
process.stdout.write("2026.01.01-fixture\\n");
process.exit(0);
@@ -1147,5 +1182,6 @@ console.log(` projects: report-fixture (4 clips, 1 mid-sentence), no-origin-fix
console.log(` localhost-fixture, bike-fixture (sweep), find/ (shadowed),`);
console.log(` deep/nested/solo-fixture (collapse case), bench-fixture (writable),`);
console.log(` walk-fixture (read-only: w01/w04 walkable, w02 unfetched, w03 judged),`);
+console.log(` editor-fetch-fixture + editor-fetch-many-fixture (nothing cached — the editor fetch's subjects),`);
console.log(` longform-fixture (cue gap, legacy .bak, ffmeta), longform-edit-fixture, dash-fixture`);
console.log(` ${taken} candidate files copied, 2 mix tracks synthesised`);
diff --git a/umtool/e2e/report-fetch-via-editor.spec.ts b/umtool/e2e/report-fetch-via-editor.spec.ts
@@ -0,0 +1,187 @@
+import { test, expect } from "@playwright/test";
+import { existsSync, readFileSync, statSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+
+// ---------------------------------------------------------------------------
+// Fetching a clip window BY ASKING THE EDITOR.
+//
+// The operator's rule is that no yt-dlp runs by hand, so the bench's "fetch
+// more" now posts to the editor and the bytes land in the CORPUS beside the
+// video — where the next report citing the same stream gets them for free.
+//
+// The editor is STUBBED (e2e/fixtures/editor-stub.mjs, a second webServer):
+// running a real one would arm its runners against whatever corpus they found,
+// and the endpoint's whole job is to spend somebody's bandwidth. What the stub
+// records is the part a file on disk cannot prove — that the provenance and the
+// bearer token actually crossed the wire.
+//
+// The strongest assertion here is a NEGATIVE one: the fixture's yt-dlp stub
+// logs every invocation, and this path must produce none.
+// ---------------------------------------------------------------------------
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const FIXTURE = path.join(HERE, "..", ".e2e-song");
+const PROJECT = "reports/editor-fetch-fixture";
+const MANY = "reports/editor-fetch-many-fixture";
+const STUB_LOG = path.join(FIXTURE, "editor-stub.requests.json");
+const YTDLP_LOG = path.join(FIXTURE, "bin", "yt-dlp.invocations");
+const clipsDir = (video: string) =>
+ path.join(FIXTURE, "channels", "testchan", "data", video, "clips");
+
+type StubRequest = {
+ channelSlug?: string;
+ videoId?: string;
+ from?: number;
+ to?: number;
+ requestedBy?: string;
+ manifest?: string;
+ clipId?: string;
+ reason?: string;
+ authorization?: string;
+};
+
+const stubRequests = (): StubRequest[] => {
+ try {
+ return JSON.parse(readFileSync(STUB_LOG, "utf8")) as StubRequest[];
+ } catch {
+ return [];
+ }
+};
+
+const ytdlpCalls = (): number => {
+ try {
+ return readFileSync(YTDLP_LOG, "utf8").split("\n").filter(Boolean).length;
+ } catch {
+ return 0;
+ }
+};
+
+async function waitForJob(
+ request: import("@playwright/test").APIRequestContext,
+ baseURL: string,
+): Promise<{ state: string; error: string | null }> {
+ let last = { state: "running", error: null as string | null };
+ await expect
+ .poll(
+ async () => {
+ const r = await request.get(`${baseURL}/api/report/fetch`);
+ const j = (await r.json()) as { job: typeof last | null };
+ if (j.job) last = j.job;
+ return j.job?.state ?? "done";
+ },
+ { timeout: 60_000 },
+ )
+ .not.toBe("running");
+ return last;
+}
+
+test("the fetch route asks the editor, carries the provenance, and runs no yt-dlp", async ({
+ page,
+ request,
+ baseURL,
+}) => {
+ test.setTimeout(90_000);
+ const ytdlpBefore = ytdlpCalls();
+ const asksBefore = stubRequests().length;
+
+ const post = await request.post(`${baseURL}/api/report/fetch`, {
+ data: { project: PROJECT, clip: "e01", padBefore: 2, padAfter: 2 },
+ });
+ expect(post.status(), await post.text()).toBe(202);
+
+ const job = await waitForJob(request, baseURL!);
+ expect(job.state, job.error ?? "").toBe("done");
+
+ // WHAT CROSSED THE WIRE. A window with no requester is a window nobody can
+ // account for in six months, which is the reason the field is required.
+ const asks = stubRequests().slice(asksBefore);
+ expect(asks.length).toBe(1);
+ const ask = asks[0];
+ expect(ask.authorization).toBe("Bearer umtool-e2e-token");
+ expect(ask.channelSlug).toBe("testchan");
+ expect(ask.videoId).toBe("vid1");
+ expect(ask.requestedBy).toBe("umtool");
+ expect(ask.manifest).toBe("editor-fetch-fixture");
+ expect(ask.clipId).toBe("e01");
+ // The clip's own note is the closest thing the manifest has to a reason.
+ expect(ask.reason).toContain("the speaker names the number");
+ // Two decimals, the clip's window plus the pads it asked for.
+ expect(ask.from).toBe(1);
+ expect(ask.to).toBe(8);
+
+ // In the CORPUS, not in this project's out/.
+ const file = path.join(clipsDir("vid1"), "1.00-8.00.mp4");
+ expect(existsSync(file)).toBe(true);
+ expect(statSync(file).size).toBeGreaterThan(0);
+
+ // AND NO yt-dlp. This is the whole point: the fetch went through the
+ // editor's managed path, not through a process this server spawned.
+ expect(ytdlpCalls()).toBe(ytdlpBefore);
+
+ // The page reads the corpus window as cached, so the walk stops skipping it.
+ await page.goto(`/browse/${PROJECT}`);
+ await expect(page.locator('[data-entry="e01"]')).toHaveAttribute(
+ "data-fetched",
+ "1",
+ );
+ await expect(page.locator("[data-ready-count]")).toContainText("ready 1 of 1");
+
+ // And /api/report/raw serves it: resolveInRoots admits the corpus as a READ
+ // root, which is what keeps a window that is right there from 400ing.
+ const raw = await request.get(
+ `${baseURL}/api/report/raw?project=${encodeURIComponent(PROJECT)}&clip=e01`,
+ );
+ expect(raw.status()).toBe(200);
+ expect(raw.headers()["x-window"]).toBe("1.00-8.00.mp4");
+ expect(raw.headers()["x-fetch-start"]).toBe("1");
+});
+
+test("the project page fetches the unfetched clips one at a time, and Stop halts it", async ({
+ page,
+ request,
+ baseURL,
+}) => {
+ test.setTimeout(120_000);
+ const ytdlpBefore = ytdlpCalls();
+
+ await page.goto(`/browse/${MANY}`);
+ await expect(page.locator("[data-ready-count]")).toContainText("ready 0 of 2");
+ const button = page.getByRole("button", { name: /fetch 2 unfetched clips/i });
+ await expect(button).toBeVisible();
+
+ // STOP ABANDONS, IT DOES NOT KILL: the in-flight fetch is already paid for,
+ // so what Stop means is "start no more". One clip is asked for, the second
+ // never is — which is also what proves the loop is serial rather than a
+ // burst, since a burst would have both requests already on the wire.
+ await button.click();
+ await page.getByRole("button", { name: /^stop$/i }).click();
+ await expect(page.locator("[data-fetch-unfetched-msg]")).toContainText(
+ /stopped after 1 of 2/,
+ { timeout: 90_000 },
+ );
+
+ const asks = stubRequests().filter((a) => a.manifest === "editor-fetch-many-fixture");
+ expect(asks.map((a) => a.clipId)).toEqual(["f01"]);
+ // f01 is 0–3 s with the route's default ±20 s of pad.
+ expect(existsSync(path.join(clipsDir("vid2"), "0.00-23.00.mp4"))).toBe(true);
+
+ // AND f02 IS NOW FETCHED TOO, without anybody asking for it. A window is
+ // deliberately generous and the predicate is containment, so one fetch is
+ // the next clip's cache — which is the entire argument for putting these
+ // bytes in the corpus instead of in one project's out/.
+ await page.reload();
+ await expect(page.locator("[data-ready-count]")).toContainText("ready 2 of 2");
+ await expect(page.locator('[data-entry="f02"]')).toHaveAttribute(
+ "data-fetched",
+ "1",
+ );
+ await expect(
+ page.getByRole("button", { name: /unfetched clip/i }),
+ ).toHaveCount(0);
+
+ // Still nothing shelled out to yt-dlp.
+ expect(ytdlpCalls()).toBe(ytdlpBefore);
+ // Leave the runner quiet for the next spec.
+ await waitForJob(request, baseURL!);
+});
diff --git a/umtool/lib/report/driver.mjs b/umtool/lib/report/driver.mjs
@@ -208,7 +208,8 @@ export function fetchSteps(project, clipId, pad) {
* decision rather than a fork in every consumer.
*
* The editor URL and the shared WORKER_TOKEN come from the ENVIRONMENT, never
- * from the request: a client that could name the editor could name any host.
+ * from the request: a client that could name the editor could name any host --
+ * and never from the step's `env` either, which jobView() shows the browser.
*
* @param {{ dir: string }} project
* @param {string} clipId
@@ -221,10 +222,11 @@ export function editorFetchSteps(project, clipId, pad) {
return [
{
cwd: PIPELINE_DIR,
- env: {
- ARCHILYZER_EDITOR_URL: process.env.ARCHILYZER_EDITOR_URL ?? "",
- WORKER_TOKEN: process.env.WORKER_TOKEN ?? "",
- },
+ // EMPTY, and that is the point. A step's `env` is echoed back to the
+ // browser by jobView(), so putting WORKER_TOKEN here would print the
+ // shared secret on the page that started the job. The child inherits
+ // process.env, which is where both values already are.
+ env: {},
label: `ask the editor for ${clipId} with −${before}s / +${after}s of pad`,
argv: [
"node",
diff --git a/umtool/playwright.config.ts b/umtool/playwright.config.ts
@@ -3,6 +3,10 @@ import path from "node:path";
import { fileURLToPath } from "node:url";
const PORT = Number(process.env.UMTOOL_E2E_PORT ?? 3051);
+// The editor stub's port. Derived from the suite's own, so a worktree that gets
+// a different port block gets a different stub port too and two checkouts can
+// never drive each other's stub.
+const STUB_PORT = PORT + 1;
// The package is "type": "module", so there is no __dirname here.
const FIXTURE = path.join(path.dirname(fileURLToPath(import.meta.url)), ".e2e-song");
@@ -26,7 +30,27 @@ export default defineConfig({
trace: "retain-on-failure",
},
projects: [{ name: "chromium", use: { ...devices["Desktop Chrome"] } }],
- webServer: {
+ webServer: [
+ {
+ // THE EDITOR, STUBBED.
+ //
+ // A clip window is now fetched by ASKING the editor, and the suite must
+ // not run a real one: booting it arms the runners against whatever
+ // corpus they find, and the endpoint's whole job is to spend somebody's
+ // bandwidth. The stub answers the contract and writes a real mp4 into
+ // the fixture's own corpus.
+ command: `node ${path.join(path.dirname(fileURLToPath(import.meta.url)), "e2e", "fixtures", "editor-stub.mjs")}`,
+ url: `http://127.0.0.1:${STUB_PORT}/health`,
+ timeout: 30_000,
+ reuseExistingServer: false,
+ env: {
+ EDITOR_STUB_PORT: String(STUB_PORT),
+ CHANNELS_DIR: `${FIXTURE}/channels`,
+ WORKER_TOKEN: "umtool-e2e-token",
+ EDITOR_STUB_LOG: `${FIXTURE}/editor-stub.requests.json`,
+ },
+ },
+ {
// NEXT_DIST_DIR keeps this server's build directory -- and so its dev lock
// -- separate from a dev server someone is judging clips in. Without it
// Next refuses to start and every spec fails with ERR_CONNECTION_REFUSED.
@@ -41,9 +65,14 @@ export default defineConfig({
// Stub binaries, so a build spec is offline and deterministic. The
// pipeline already reads both as overrides; the fixture writes them.
`YTDLP_BIN=${FIXTURE}/bin/yt-dlp QRENCODE_BIN=${FIXTURE}/bin/qrencode ` +
+ // Where a clip window is fetched FROM, and the shared secret it is asked
+ // with. Set here rather than in a step's env: jobView() echoes a step's
+ // env back to the browser, and this is a token.
+ `ARCHILYZER_EDITOR_URL=http://127.0.0.1:${STUB_PORT} WORKER_TOKEN=umtool-e2e-token ` +
`NEXT_DIST_DIR=.next-e2e pnpm exec next dev --port ${PORT}`,
port: PORT,
reuseExistingServer: false,
timeout: 120_000,
- },
+ },
+ ],
});