commit a1004f19ac8db79b212af82bb86e53b254d76b1f
parent ca9616c8344331150b995ef4fb25aba58766f55b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 20 Sep 2026 20:10:05 -0400
Merge branch 'main' into storage/locations-s5-s6
# Conflicts:
# plans/FACTS.md
Diffstat:
36 files changed, 2500 insertions(+), 75 deletions(-)
diff --git a/RUNNING_IN_DOCKER.md b/RUNNING_IN_DOCKER.md
@@ -224,6 +224,55 @@ boot self-updates it. Or, once:
docker compose exec editor yt-dlp -U
```
+### Driving the editor without a browser
+
+Every editor gesture is a server action, which is fine for a person and hostile
+to a script: there is no URL to POST to. `/api/ops/*` is a thin layer over the
+**same actions** — one route per gesture, no rule of its own — so a shell, a cron
+job or an agent can run the archive without Playwright.
+
+It is gated by the **same `WORKER_TOKEN`** as `/api/worker/*`, deliberately: that
+variable already means "this instance takes instructions from something that is
+not the browser in front of it". Unset on the server and every route answers
+**503** (the surface is off until you opt in); wrong or missing on the caller and
+it answers **401**.
+
+```sh
+export ARCHILYZER_EDITOR_URL=http://localhost:3001
+export WORKER_TOKEN=<the same secret the editor is running with>
+
+pnpm ops sync --json '{"slug":"the-quartering"}' --wait
+pnpm ops metadata-scan --json '{"slug":"the-quartering"}'
+pnpm ops channel-config --json '{"slug":"the-quartering","patch":{"downloadFilterExclude":"rerun"}}'
+pnpm ops channel-priority --json '{"slugs":["the-quartering"],"operation":"download","tier":"paused"}'
+pnpm ops lane --json '{"lane":"download","held":true}'
+pnpm ops refresh-report --json '{"all":true}'
+pnpm ops get channel the-quartering
+pnpm ops list # every action name
+```
+
+Three things to know before you script against it:
+
+- **A job-starting action returns a `jobId` and does not stream.** The job may
+ sit in a platform queue behind other work for hours, so "started" is the
+ honest answer; `--wait` follows `/api/jobs/<id>/log` to the end and exits with
+ the job's status.
+- **Unknown body keys are a 400.** A misspelled `downloadFilterExclude` would
+ otherwise save cleanly and leave a channel downloading everything.
+- **`channel-config` patch keys are the CONFIGURE FORM's field names**, not
+ `config.json`'s — `downloadFilterInclude` / `downloadFilterExclude` rather than
+ a `downloadFilter` object. That is what routes them through the form's own
+ validators, so a bad regex is refused here with the sentence the form shows.
+ `""` clears a field, exactly as clearing the input does.
+
+The read side needs no new routes for jobs: `/api/jobs/active`,
+`/api/jobs/<id>/log`, `/api/scheduler/status` and `/api/auto-queue/status`
+already exist. `GET /api/ops/channel/<slug>` is the one addition — config,
+report totals, bucket sizes, priority and, the part no directory listing can
+tell you, whether the channel's media is actually **reachable**. Add `--counts`
+(`?counts=1`) for the live on-disk counts; it is opt-in because it walks every
+video directory, eleven thousand of them on the largest channel here.
+
### Booting without resuming work
`editor/instrumentation.ts` arms the sync heartbeat and every enabled auto-queue
diff --git a/SETUP.md b/SETUP.md
@@ -273,7 +273,7 @@ any of them via environment variables before launching:
| `PARAKEET_CLI` / `PARAKEET_MODEL` / `PARAKEET_STITCH_BIN` | `parakeet-cli` / — / `scripts/parakeet-stitch.mjs` | parakeet.cpp CLI, model, and wrapper. |
| `FFMPEG_BIN` / `FFPROBE_BIN` | `ffmpeg` / `ffprobe` (PATH) | Audio transcode + duration checks. |
| `RSYNC_BIN` | `rsync` (PATH) | Saved-video backup. |
-| `WORKER_TOKEN` | — | Bearer token for the remote-worker transcription API (set on both ends when used). |
+| `WORKER_TOKEN` | — | Bearer token for the remote-worker transcription API (set on both ends when used), and for the `/api/ops/*` HTTP layer over the editor's actions — see [RUNNING_IN_DOCKER.md](RUNNING_IN_DOCKER.md#driving-the-editor-without-a-browser) and `pnpm ops`. Unset means both surfaces are off. |
Feature-area docs cover their own env vars: [SCHEDULED_SYNC.md](SCHEDULED_SYNC.md)
(`SYNC_HEARTBEAT_SECONDS`, `SYNC_TICK_URL`, `SYNC_TICK_TOKEN`) and
diff --git a/common/jobs/jobSpec.ts b/common/jobs/jobSpec.ts
@@ -53,6 +53,13 @@ const REPLAY_BUCKETS: ReadonlySet<string> = new Set<ReplayBucket>([
"supersededAutoSubs",
]);
+// Is this bucket name one a replayed job can be re-derived from? Exported so a
+// caller naming a bucket (the ops API's retry-bucket route) can decide whether
+// the job is replayable without re-spelling the set.
+export function isReplayBucket(value: unknown): value is ReplayBucket {
+ return typeof value === "string" && REPLAY_BUCKETS.has(value);
+}
+
// Defensive parse for a spec read back from JSON (the <id>.meta.json sidecar).
// Returns null on anything malformed so a hand-edited or stale file can't
// crash a reader. Mirrors the tolerance of readJobMeta / readWorkerDefaults.
diff --git a/editor/app/api/ops/_lib.ts b/editor/app/api/ops/_lib.ts
@@ -0,0 +1,209 @@
+import { NextResponse } from "next/server";
+import { authorizeWorkerRequest } from "yt-dlp-transcript-common/lib/workerToken";
+import { isValidChannelSlug } from "yt-dlp-transcript-common/controller/channels";
+import type { StreamActionResult } from "yt-dlp-transcript-common/jobs/streamCommand";
+import type { QueueOutcome } from "../../channels/lib/queueForSlugs";
+
+// THE OPS API IS ADAPTERS, AND NOTHING ELSE.
+//
+// Every route under /api/ops is ~5 lines that validate a JSON body and call ONE
+// existing server action. No route may contain a rule the UI does not already
+// enforce: the point of the layer is that an agent driving the editor over HTTP
+// and an operator clicking the same button get the same refusal, with the same
+// sentence, from the same code. A check written here would be a second opinion
+// nobody maintains.
+//
+// THE TOKEN IS THE WORKER TOKEN, ON PURPOSE. `WORKER_TOKEN` already gates the
+// LAN worker protocol and already means "this instance accepts instructions
+// from something that is not the browser in front of it". A second secret would
+// be a second thing to distribute, rotate and leave unset; the failure modes are
+// identical, so the gate is. Unset => 503 (the surface is off, you opt in),
+// wrong => 401.
+//
+// A JOB-STARTING ROUTE RETURNS A jobId AND NEVER STREAMS. runManagedFunction
+// hands back a ReadableStream the browser consumes; an HTTP caller wants to
+// disconnect and poll. So every adapter cancels the stream (which stops pushing
+// into the controller and leaves the on-disk log running — see streamCommand's
+// `cancel()` note) and returns the id. Follow it with /api/jobs/<id>/log.
+
+export type OpsBody = Record<string, unknown>;
+
+// Thrown by the field readers below; caught by `ops()` and rendered as a 400.
+export class OpsInputError extends Error {}
+
+export function opsFail(
+ error: string,
+ status = 400,
+ extra?: Record<string, unknown>,
+): NextResponse {
+ return NextResponse.json({ ok: false, error, ...extra }, { status });
+}
+
+// Auth + body parse + unknown-key rejection, wrapped around one handler.
+//
+// UNKNOWN KEYS ARE A 400, not a silent ignore. A caller that misspells
+// `downloadFilterExclude` would otherwise get a cheerful `{ ok: true }` and a
+// channel that still downloads everything. The allow-list IS the route's
+// documented body shape.
+export async function ops(
+ request: Request,
+ allowedKeys: readonly string[],
+ run: (body: OpsBody) => Promise<NextResponse>,
+): Promise<NextResponse> {
+ const auth = authorizeWorkerRequest(request.headers.get("authorization"));
+ if (!auth.ok) return opsFail(auth.error, auth.status);
+ let body: unknown;
+ try {
+ body = await request.json();
+ } catch {
+ return opsFail("malformed JSON body");
+ }
+ if (typeof body !== "object" || body === null || Array.isArray(body)) {
+ return opsFail("body must be a JSON object");
+ }
+ const unknown = Object.keys(body as OpsBody).filter(
+ (k) => !allowedKeys.includes(k),
+ );
+ if (unknown.length) {
+ return opsFail(
+ `unknown key(s): ${unknown.join(", ")} — this route accepts ${
+ allowedKeys.length ? allowedKeys.join(", ") : "no keys"
+ }`,
+ );
+ }
+ try {
+ return await run(body as OpsBody);
+ } catch (e) {
+ if (e instanceof OpsInputError) return opsFail(e.message);
+ return opsFail((e as Error).message, 500);
+ }
+}
+
+// A GET route's gate. Same token, no body.
+export function opsAuth(request: Request): NextResponse | null {
+ const auth = authorizeWorkerRequest(request.headers.get("authorization"));
+ return auth.ok ? null : opsFail(auth.error, auth.status);
+}
+
+// --- field readers ----------------------------------------------------------
+
+export function reqString(body: OpsBody, key: string): string {
+ const v = body[key];
+ if (typeof v !== "string" || !v.trim()) {
+ throw new OpsInputError(`"${key}" is required and must be a non-empty string`);
+ }
+ return v.trim();
+}
+
+// A CHANNEL SLUG, NOT MERELY A STRING. Every slug below reaches a `path.join`
+// under `channelsDir`, and the readers swallow their own errors — so a
+// traversing segment would fail SILENTLY (an empty config, an "empty channel")
+// rather than loudly. `isValidChannelSlug` is CHANNEL_SLUG_RE, which forbids
+// "/" and "..", and is what every other slug-taking surface in the app uses.
+//
+// One reader for every route rather than a check per route: a route added later
+// gets this for free by calling reqSlug instead of reqString, and there is one
+// place to be wrong.
+export function reqSlug(body: OpsBody, key: string): string {
+ const v = reqString(body, key);
+ if (!isValidChannelSlug(v)) {
+ throw new OpsInputError(
+ `"${v}" is not a valid channel slug (letters, digits, ".", "_", "-"; must start with a letter or digit)`,
+ );
+ }
+ return v;
+}
+
+export function reqSlugs(body: OpsBody, key: string): string[] {
+ const values = reqStringArray(body, key);
+ for (const v of values) {
+ if (!isValidChannelSlug(v)) {
+ throw new OpsInputError(
+ `"${v}" is not a valid channel slug (letters, digits, ".", "_", "-"; must start with a letter or digit)`,
+ );
+ }
+ }
+ return values;
+}
+
+export function optString(body: OpsBody, key: string): string | undefined {
+ const v = body[key];
+ if (v === undefined) return undefined;
+ if (typeof v !== "string") {
+ throw new OpsInputError(`"${key}" must be a string`);
+ }
+ return v;
+}
+
+export function optBool(body: OpsBody, key: string): boolean | undefined {
+ const v = body[key];
+ if (v === undefined) return undefined;
+ if (typeof v !== "boolean") {
+ throw new OpsInputError(`"${key}" must be a boolean`);
+ }
+ return v;
+}
+
+export function reqStringArray(body: OpsBody, key: string): string[] {
+ const v = body[key];
+ if (
+ !Array.isArray(v) ||
+ v.length === 0 ||
+ v.some((s) => typeof s !== "string" || !s.trim())
+ ) {
+ throw new OpsInputError(
+ `"${key}" is required and must be a non-empty array of strings`,
+ );
+ }
+ return (v as string[]).map((s) => s.trim());
+}
+
+export function oneOf<T extends string>(
+ body: OpsBody,
+ key: string,
+ values: readonly T[],
+): T {
+ const v = reqString(body, key);
+ if (!(values as readonly string[]).includes(v)) {
+ throw new OpsInputError(`"${key}" must be one of ${values.join(", ")}`);
+ }
+ return v as T;
+}
+
+// --- result mapping ---------------------------------------------------------
+
+// StreamActionResult -> { ok: true, jobId } | { ok: false, error, info? }.
+// The stream is cancelled, never returned: see the header.
+export function jobResponse(result: StreamActionResult): NextResponse {
+ if (!result.ok) {
+ return opsFail(result.error, 400, result.info ? { info: true } : undefined);
+ }
+ void result.stream.cancel();
+ return NextResponse.json({ ok: true, jobId: result.jobId });
+}
+
+// The `{ error } | undefined` shape every /channels action returns.
+export function actionResponse(
+ result: { error: string } | undefined,
+): NextResponse {
+ if (result?.error) return opsFail(result.error);
+ return NextResponse.json({ ok: true });
+}
+
+// The `{ ok } | { ok: false, error }` shape the lane/pause actions return.
+export function okResponse(
+ result: { ok: boolean; error?: string },
+): NextResponse {
+ if (!result.ok) return opsFail(result.error ?? "action failed");
+ return NextResponse.json({ ok: true });
+}
+
+// A bulk fan-out's { queued, skipped }. A skip is not a failure — the caller
+// gets both lists and decides, exactly as the bulk bar in the UI does.
+export function queueResponse(outcome: QueueOutcome): NextResponse {
+ return NextResponse.json({
+ ok: true,
+ queued: outcome.queued,
+ skipped: outcome.skipped,
+ });
+}
diff --git a/editor/app/api/ops/build-deploy/route.ts b/editor/app/api/ops/build-deploy/route.ts
@@ -0,0 +1,33 @@
+import { NextResponse } from "next/server";
+import {
+ buildAndDeployAction,
+ buildAndDeployAllSitesAction,
+} from "../../../sites/lib/buildAction";
+import { jobResponse, OpsInputError, ops, optBool, optString } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { siteId: string, skipArchives? } | { all: true, skipArchives? }
+// -> { ok: true, jobId }
+//
+// One managed job either way (build then deploy, one log, one Cancel), so the
+// caller polls /api/jobs/<jobId>/log exactly as for a single site.
+export async function POST(request: Request) {
+ return ops(request, ["siteId", "all", "skipArchives"], async (body) => {
+ const skipArchives = optBool(body, "skipArchives");
+ const all = optBool(body, "all");
+ const siteId = optString(body, "siteId");
+ if (all) {
+ if (siteId) throw new OpsInputError('send either "siteId" or "all", not both');
+ return jobResponse(await buildAndDeployAllSitesAction(skipArchives));
+ }
+ if (!siteId?.trim()) {
+ throw new OpsInputError('"siteId" is required (or send { "all": true })');
+ }
+ return jobResponse(await buildAndDeployAction(siteId.trim(), skipArchives));
+ });
+}
+
+export function GET() {
+ return NextResponse.json({ ok: false, error: "POST only" }, { status: 405 });
+}
diff --git a/editor/app/api/ops/build-index/route.ts b/editor/app/api/ops/build-index/route.ts
@@ -0,0 +1,11 @@
+import { buildIndexAction } from "../../../sites/lib/buildAction";
+import { jobResponse, ops, optString } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { queueKey? } -> { ok: true, jobId }. Rebuilds the LMDB corpus index.
+export async function POST(request: Request) {
+ return ops(request, ["queueKey"], async (body) =>
+ jobResponse(await buildIndexAction(optString(body, "queueKey"))),
+ );
+}
diff --git a/editor/app/api/ops/build-site/route.ts b/editor/app/api/ops/build-site/route.ts
@@ -0,0 +1,59 @@
+import { NextResponse } from "next/server";
+import {
+ buildAllSitesAction,
+ buildExportAction,
+} from "../../../sites/lib/buildAction";
+import { jobResponse, OpsInputError, ops, optBool } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { siteIds: string[], skipData?, skipArchives? } | { all: true, skipArchives? }
+//
+// BUILD WITHOUT DEPLOYING. `siteIds` queues one build-export job per site on the
+// shared build queue (they run one at a time, as they do from /sites) and
+// returns `{ jobs: [{ siteId, jobId }] }` — a list, because there is a job per
+// site and a caller waiting on them needs all the ids. `{ all: true }` is the
+// single build-all job instead, which is the docker fan-out.
+export async function POST(request: Request) {
+ return ops(
+ request,
+ ["siteIds", "all", "skipData", "skipArchives"],
+ async (body) => {
+ const skipArchives = optBool(body, "skipArchives");
+ if (optBool(body, "all")) {
+ if (body.siteIds !== undefined) {
+ throw new OpsInputError('send either "siteIds" or "all", not both');
+ }
+ return jobResponse(await buildAllSitesAction(skipArchives));
+ }
+ const raw = body.siteIds;
+ if (
+ !Array.isArray(raw) ||
+ raw.length === 0 ||
+ raw.some((s) => typeof s !== "string" || !s.trim())
+ ) {
+ throw new OpsInputError(
+ '"siteIds" is required and must be a non-empty array of strings (or send { "all": true })',
+ );
+ }
+ const skipData = optBool(body, "skipData");
+ const jobs: { siteId: string; jobId: string }[] = [];
+ const skipped: { siteId: string; reason: string }[] = [];
+ for (const siteId of (raw as string[]).map((s) => s.trim())) {
+ const result = await buildExportAction(
+ siteId,
+ undefined,
+ skipData,
+ skipArchives,
+ );
+ if (!result.ok) {
+ skipped.push({ siteId, reason: result.error });
+ continue;
+ }
+ void result.stream.cancel();
+ jobs.push({ siteId, jobId: result.jobId });
+ }
+ return NextResponse.json({ ok: true, jobs, skipped });
+ },
+ );
+}
diff --git a/editor/app/api/ops/channel-config/route.ts b/editor/app/api/ops/channel-config/route.ts
@@ -0,0 +1,47 @@
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels";
+import {
+ applyChannelFormPatch,
+ channelConfigToFormData,
+ validateChannelFormPatch,
+} from "../../../channels/components/channelConfigToForm";
+import { updateChannelAction } from "../../../channels/actions";
+import { actionResponse, OpsInputError, ops, reqSlug } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug: string, patch: { <Configure-form field>: string|number|boolean|null } }
+//
+// The patch keys are the FORM's field names, not ChannelConfig's, because the
+// form is what validates them: `downloadFilterInclude` / `downloadFilterExclude`
+// rather than a `downloadFilter` object, `ytdlpExtraArgs` as a string or an
+// array of lines. See channelConfigToForm.ts for why the patch is laid over the
+// channel's current form representation instead of being written directly.
+//
+// `""` (or null) clears a field, exactly as clearing the input does.
+export async function POST(request: Request) {
+ return ops(request, ["slug", "patch"], async (body) => {
+ const slug = reqSlug(body, "slug");
+ const patch = body.patch;
+ if (
+ typeof patch !== "object" ||
+ patch === null ||
+ Array.isArray(patch)
+ ) {
+ throw new OpsInputError('"patch" must be a JSON object');
+ }
+ // KEYS FIRST, CHANNEL SECOND. A misspelled field is a fact about the
+ // request; reporting "channel not found" for it would hide the real error.
+ try {
+ validateChannelFormPatch(patch as Record<string, unknown>);
+ } catch (e) {
+ throw new OpsInputError((e as Error).message);
+ }
+ const paths = getPaths();
+ const existing = await readChannelConfig(paths, slug);
+ if (!existing) throw new OpsInputError(`Channel "${slug}" not found`);
+ const fd = channelConfigToFormData(existing);
+ applyChannelFormPatch(fd, patch as Record<string, unknown>);
+ return actionResponse(await updateChannelAction(slug, undefined, fd));
+ });
+}
diff --git a/editor/app/api/ops/channel-priority/route.ts b/editor/app/api/ops/channel-priority/route.ts
@@ -0,0 +1,91 @@
+import { NextResponse } from "next/server";
+import {
+ STORED_CHANNEL_TIERS,
+ PRIORITY_OPERATIONS,
+ type PriorityOperation,
+ type StoredChannelTier,
+} from "yt-dlp-transcript-common/lib/channelPriority";
+import {
+ applyChannelPriorityPresetAction,
+ setChannelOperationTierAction,
+ setChannelTierAction,
+} from "../../../channels/actions";
+import {
+ actionResponse,
+ OpsInputError,
+ ops,
+ optString,
+ reqSlugs,
+} from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slugs: string[], tier?: "normal"|"low"|"paused"|null, operation?: one
+// of sync|transcription|download|digest|backfill, preset?: "sync-only"|"clear" }
+//
+// Three named gestures, one route, because they are one gesture in the UI: the
+// /channels deck's tier control. `operation` present pins that operation's
+// override (null clears it back to the base tier); absent sets the BASE tier.
+// `preset` is the deck's two shortcuts and takes no tier.
+export async function POST(request: Request) {
+ return ops(
+ request,
+ ["slugs", "tier", "operation", "preset"],
+ async (body) => {
+ const slugs = reqSlugs(body, "slugs");
+ const preset = optString(body, "preset");
+ if (preset !== undefined) {
+ if (preset !== "sync-only" && preset !== "clear") {
+ throw new OpsInputError('"preset" must be "sync-only" or "clear"');
+ }
+ return actionResponse(
+ await applyChannelPriorityPresetAction(slugs, preset),
+ );
+ }
+ const rawTier = body.tier;
+ if (rawTier !== null && typeof rawTier !== "string") {
+ throw new OpsInputError(
+ `"tier" must be one of ${STORED_CHANNEL_TIERS.join(", ")} (or null with an "operation", to clear the override)`,
+ );
+ }
+ if (
+ rawTier !== null &&
+ !(STORED_CHANNEL_TIERS as readonly string[]).includes(rawTier)
+ ) {
+ throw new OpsInputError(
+ `"tier" must be one of ${STORED_CHANNEL_TIERS.join(", ")}`,
+ );
+ }
+ const operation = optString(body, "operation");
+ if (operation !== undefined) {
+ if (!(PRIORITY_OPERATIONS as readonly string[]).includes(operation)) {
+ throw new OpsInputError(
+ `"operation" must be one of ${PRIORITY_OPERATIONS.join(", ")}`,
+ );
+ }
+ return actionResponse(
+ await setChannelOperationTierAction(
+ slugs,
+ operation as PriorityOperation,
+ rawTier as StoredChannelTier | null,
+ ),
+ );
+ }
+ if (rawTier === null) {
+ throw new OpsInputError(
+ 'a null "tier" clears an operation override — name an "operation", or send a stored tier',
+ );
+ }
+ return actionResponse(
+ await setChannelTierAction(slugs, rawTier as StoredChannelTier),
+ );
+ },
+ );
+}
+
+export function GET() {
+ return NextResponse.json(
+ { ok: false, error: "POST only" },
+ { status: 405 },
+ );
+}
diff --git a/editor/app/api/ops/channel/[slug]/route.ts b/editor/app/api/ops/channel/[slug]/route.ts
@@ -0,0 +1,100 @@
+import { NextResponse } from "next/server";
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import {
+ isValidChannelSlug,
+ readChannelConfig,
+ readChannelSnapshot,
+ readChannelStat,
+} from "yt-dlp-transcript-common/controller/channels";
+import { inspectChannelMedia } from "yt-dlp-transcript-common/lib/channelMedia";
+import {
+ overridesOf,
+ rankOf,
+ tierOf,
+} from "yt-dlp-transcript-common/lib/channelPriority";
+import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { opsAuth } from "../../_lib";
+
+export const dynamic = "force-dynamic";
+
+// GET /api/ops/channel/<slug>
+//
+// THE READ SIDE, so a caller never opens transcripts/ itself. Everything a
+// script needs to decide what to do next about one channel: its stored config,
+// its report totals and bucket SIZES (not the id lists — a bucket can hold tens
+// of thousands of ids and a caller deciding "is there work" only needs the
+// count; /api/ops/retry-bucket resolves the ids server-side anyway), its
+// priority tier, and — the one that cannot be inferred from the filesystem —
+// whether its media is reachable.
+//
+// `media.status` is `inspectChannelMedia`'s, which is the ONE module that can
+// tell an unmounted drive from an empty channel. Every enumerator else swallows
+// ENOENT on data/ as "no videos", so a script reading counts alone would read an
+// unmounted platter as a channel that has downloaded nothing.
+//
+// `?counts=1` ADDS THE LIVE ON-DISK COUNTS, and it is opt-in because
+// readChannelStat walks every video directory — eleven thousand readdirs on the
+// largest channel here. The report's totals answer the same question from a
+// file, and a poll loop asking for them every few seconds must not be the thing
+// that hammers the platter.
+export async function GET(
+ request: Request,
+ { params }: { params: Promise<{ slug: string }> },
+) {
+ const denied = opsAuth(request);
+ if (denied) return denied;
+ const { slug } = await params;
+ // The same shape check every POST route applies through reqSlug (_lib.ts);
+ // spelled out here only because the slug arrives as a route param, not in a
+ // body. readChannelConfig swallows its own errors, so a traversing segment
+ // would fail silently rather than loudly.
+ if (!isValidChannelSlug(slug)) {
+ return NextResponse.json(
+ { ok: false, error: `"${slug}" is not a valid channel slug` },
+ { status: 400 },
+ );
+ }
+ const paths = getPaths();
+ const config = await readChannelConfig(paths, slug);
+ if (!config) {
+ return NextResponse.json(
+ { ok: false, error: `Channel "${slug}" not found` },
+ { status: 404 },
+ );
+ }
+ const priority = getSettings().channelPriority;
+ const wantCounts = new URL(request.url).searchParams.get("counts") === "1";
+ const [snapshot, stat, media] = await Promise.all([
+ readChannelSnapshot(paths, slug),
+ wantCounts ? readChannelStat(paths, slug) : Promise.resolve(null),
+ inspectChannelMedia(paths, slug, config),
+ ]);
+ const buckets = snapshot
+ ? Object.fromEntries(
+ Object.entries(snapshot.buckets).map(([k, v]) => [
+ k,
+ Array.isArray(v) ? v.length : v,
+ ]),
+ )
+ : null;
+ return NextResponse.json({
+ ok: true,
+ slug,
+ config,
+ priority: {
+ tier: tierOf(priority, slug),
+ overrides: overridesOf(priority, slug),
+ rank: rankOf(priority, slug),
+ },
+ // null unless ?counts=1 — see the header.
+ counts: stat,
+ media,
+ report: snapshot
+ ? {
+ generatedAt: snapshot.generatedAt,
+ totals: snapshot.totals,
+ buckets,
+ }
+ : null,
+ });
+}
diff --git a/editor/app/api/ops/download-missing/route.ts b/editor/app/api/ops/download-missing/route.ts
@@ -0,0 +1,21 @@
+import { downloadMissingAction } from "../../../channels/[slug]/pipelineActions";
+import { jobResponse, ops, optBool, optString, reqSlug } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, queueKey?, ignoreArchive?, abortOnError? } -> { ok: true, jobId }
+export async function POST(request: Request) {
+ return ops(
+ request,
+ ["slug", "queueKey", "ignoreArchive", "abortOnError"],
+ async (body) =>
+ jobResponse(
+ await downloadMissingAction(
+ reqSlug(body, "slug"),
+ optString(body, "queueKey"),
+ optBool(body, "ignoreArchive"),
+ optBool(body, "abortOnError"),
+ ),
+ ),
+ );
+}
diff --git a/editor/app/api/ops/import-video/route.ts b/editor/app/api/ops/import-video/route.ts
@@ -0,0 +1,17 @@
+import { importVideoAction } from "../../../channels/[slug]/pipelineActions";
+import { jobResponse, ops, optString, reqSlug, reqString } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug: string, url: string, queueKey?: string } -> { ok: true, jobId }
+export async function POST(request: Request) {
+ return ops(request, ["slug", "url", "queueKey"], async (body) =>
+ jobResponse(
+ await importVideoAction(
+ reqSlug(body, "slug"),
+ reqString(body, "url"),
+ optString(body, "queueKey"),
+ ),
+ ),
+ );
+}
diff --git a/editor/app/api/ops/lane/route.ts b/editor/app/api/ops/lane/route.ts
@@ -0,0 +1,80 @@
+import { NextResponse } from "next/server";
+import { LANES, type AutoQueueKind } from "yt-dlp-transcript-common/lib/autoQueueTypes";
+import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { drainAutoRunner } from "yt-dlp-transcript-common/controller/autoRunner";
+import {
+ pauseLaneAction,
+ resumeLaneAction,
+ saveAutoQueueAction,
+ startAutoQueueAction,
+ stopAutoQueueAction,
+} from "../../../operations/actions";
+import { OpsInputError, okResponse, ops, oneOf, optBool } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { lane: transcription|download|digest|backfill,
+// held?: boolean, enabled?: boolean, action?: "start"|"stop"|"drain" }
+//
+// One lane, up to three independent changes, applied in the order the operator
+// would: enable the policy, set the gate, then bring the runner up or down.
+// Each is the existing action and nothing more.
+//
+// `held` IS NOT `enabled`. A hold is a dispatch gate (the runner idle-waits and
+// keeps its place); disabling is the policy's master switch. The pair is what
+// /operations draws as two separate controls and this route keeps them separate.
+//
+// `action` mirrors /api/auto-queue/control's body — same three verbs, same
+// meanings — so a caller that already drives that route needs nothing new.
+export async function POST(request: Request) {
+ return ops(request, ["lane", "held", "enabled", "action"], async (body) => {
+ const lane = oneOf(body, "lane", LANES) as AutoQueueKind;
+ const enabled = optBool(body, "enabled");
+ const held = optBool(body, "held");
+ const action = body.action;
+ if (
+ action !== undefined &&
+ action !== "start" &&
+ action !== "stop" &&
+ action !== "drain"
+ ) {
+ throw new OpsInputError('"action" must be "start", "stop" or "drain"');
+ }
+ if (enabled === undefined && held === undefined && action === undefined) {
+ throw new OpsInputError(
+ 'nothing to do — send at least one of "enabled", "held" or "action"',
+ );
+ }
+ if (enabled !== undefined) {
+ // EVERY OTHER POLICY FIELD IS CARRIED FROM THE STORED POLICY, including
+ // `root`: saveAutoQueueAction refuses a tree edit while the priority
+ // model compiles the roots, and handing it back the stored tree is what
+ // makes this a pure enable/disable rather than a tree write.
+ const policy = getSettings().autoQueue[lane];
+ const saved = await saveAutoQueueAction(lane, {
+ enabled,
+ maxWorkers: policy.maxWorkers,
+ replaceAutoSubs: policy.replaceAutoSubs === true,
+ order: policy.order ?? "listed",
+ root: policy.root,
+ });
+ if (!saved.ok) return okResponse(saved);
+ }
+ if (held !== undefined) {
+ const gated = held
+ ? await pauseLaneAction(lane)
+ : await resumeLaneAction(lane);
+ if (!gated.ok) return okResponse(gated);
+ }
+ if (action === "start") {
+ const started = await startAutoQueueAction(lane);
+ if (!started.ok) return okResponse(started);
+ } else if (action === "stop") {
+ const stopped = await stopAutoQueueAction(lane);
+ if (!stopped.ok) return okResponse(stopped);
+ } else if (action === "drain") {
+ drainAutoRunner(lane);
+ }
+ return NextResponse.json({ ok: true, lane });
+ });
+}
diff --git a/editor/app/api/ops/metadata-scan/route.ts b/editor/app/api/ops/metadata-scan/route.ts
@@ -0,0 +1,16 @@
+import { runMetadataScanAction } from "../../../channels/[slug]/pipelineActions";
+import { jobResponse, ops, optString, reqSlug } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug: string, queueKey?: string } -> { ok: true, jobId }
+export async function POST(request: Request) {
+ return ops(request, ["slug", "queueKey"], async (body) =>
+ jobResponse(
+ await runMetadataScanAction(
+ reqSlug(body, "slug"),
+ optString(body, "queueKey"),
+ ),
+ ),
+ );
+}
diff --git a/editor/app/api/ops/refresh-report/route.ts b/editor/app/api/ops/refresh-report/route.ts
@@ -0,0 +1,43 @@
+import { NextResponse } from "next/server";
+import {
+ refreshAllChannelSnapshotsAction,
+ refreshChannelSnapshotAction,
+} from "../../../channels/actions";
+import {
+ actionResponse,
+ OpsInputError,
+ ops,
+ optBool,
+ optString,
+ reqSlug,
+} from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug: string } | { all: true }
+//
+// The single-channel form REGENERATES SYNCHRONOUSLY (it is a filesystem scan,
+// not a job) and returns `{ ok: true }` once snapshot.json is on disk. The
+// `all` form queues one refresh-report job per channel and returns the bulk
+// { queued, skipped } the /channels header button shows.
+export async function POST(request: Request) {
+ return ops(request, ["slug", "all"], async (body) => {
+ const all = optBool(body, "all");
+ const slug = optString(body, "slug");
+ if (all) {
+ if (slug) {
+ throw new OpsInputError('send either "slug" or "all", not both');
+ }
+ const result = await refreshAllChannelSnapshotsAction();
+ return NextResponse.json({ ok: true, ...result });
+ }
+ if (!slug?.trim()) {
+ throw new OpsInputError('"slug" is required (or send { "all": true })');
+ }
+ // Re-read through reqSlug now that we know it is the single-channel form:
+ // the shape check belongs on the value that reaches a path.join.
+ return actionResponse(
+ await refreshChannelSnapshotAction(reqSlug(body, "slug")),
+ );
+ });
+}
diff --git a/editor/app/api/ops/relocate-back/route.ts b/editor/app/api/ops/relocate-back/route.ts
@@ -0,0 +1,20 @@
+import { moveChannelMediaBackAction } from "../../../channels/[slug]/storageActions";
+import { ops, queueResponse, reqSlugs } from "../_lib";
+import { queueForSlugs } from "../../../channels/lib/queueForSlugs";
+
+export const dynamic = "force-dynamic";
+
+// POST { slugs: string[] } -> { ok: true, queued, skipped }
+//
+// One job per slug on the shared relocation queue, and the per-channel action's
+// own busy guard decides each one — the same fan-out loop the bulk move uses,
+// so a refusal reads identically whether it came from the panel or from here.
+export async function POST(request: Request) {
+ return ops(request, ["slugs"], async (body) =>
+ queueResponse(
+ await queueForSlugs(reqSlugs(body, "slugs"), {
+ run: (slug) => moveChannelMediaBackAction(slug),
+ }),
+ ),
+ );
+}
diff --git a/editor/app/api/ops/relocate/route.ts b/editor/app/api/ops/relocate/route.ts
@@ -0,0 +1,39 @@
+import { bulkRelocateChannelMediaAction } from "../../../channels/bulkStorageActions";
+import {
+ OpsInputError,
+ ops,
+ optString,
+ queueResponse,
+ reqSlugs,
+} from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slugs: string[], locationId?: string, root?: string }
+// -> { ok: true, queued, skipped }
+//
+// THE DESTINATION IS A LOCATION ID WHEREVER POSSIBLE — the root is resolved on
+// the server from settings.storage.locations, so a caller holding a stale root
+// cannot aim a batch somewhere a re-point has moved. `root` is the one-off
+// escape hatch the panel also offers. A skip is not a failure: every slug that
+// did not queue comes back with the same sentence the bulk bar shows.
+export async function POST(request: Request) {
+ return ops(request, ["slugs", "locationId", "root"], async (body) => {
+ const slugs = reqSlugs(body, "slugs");
+ const locationId = optString(body, "locationId");
+ const root = optString(body, "root");
+ if ((locationId ? 1 : 0) + (root ? 1 : 0) !== 1) {
+ throw new OpsInputError(
+ 'send exactly one of "locationId" (a location configured on /storage) or "root" (an absolute path)',
+ );
+ }
+ return queueResponse(
+ await bulkRelocateChannelMediaAction(
+ slugs,
+ locationId
+ ? { kind: "location", locationId }
+ : { kind: "custom", root: root as string },
+ ),
+ );
+ });
+}
diff --git a/editor/app/api/ops/retry-bucket/route.ts b/editor/app/api/ops/retry-bucket/route.ts
@@ -0,0 +1,72 @@
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { readChannelSnapshot } from "yt-dlp-transcript-common/controller/channels";
+import { isReplayBucket } from "yt-dlp-transcript-common/jobs/jobSpec";
+import { retryBucketAction } from "../../../channels/[slug]/pipelineActions";
+import {
+ jobResponse,
+ OpsInputError,
+ ops,
+ optBool,
+ optString,
+ reqSlug,
+ reqString,
+} from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, bucket, queueKey?, abortOnError?, handlingOverride?,
+// forceCookies?, replaceAutoSubs? } -> { ok: true, jobId }
+//
+// THE BUCKET IS RESOLVED FROM THE SNAPSHOT HERE, and that is not new logic: the
+// bucket controls in the UI pass `snapshot.buckets[key]` to the same action, and
+// the action's own argument is a list of ids. An HTTP caller naming a bucket
+// rather than pasting ids is the same gesture — and `bucketKey` travelling with
+// it is what makes the job replayable against the CURRENT bucket, exactly as a
+// clicked one is.
+export async function POST(request: Request) {
+ return ops(
+ request,
+ [
+ "slug",
+ "bucket",
+ "queueKey",
+ "abortOnError",
+ "handlingOverride",
+ "forceCookies",
+ "replaceAutoSubs",
+ ],
+ async (body) => {
+ const slug = reqSlug(body, "slug");
+ const bucket = reqString(body, "bucket");
+ const snapshot = await readChannelSnapshot(getPaths(), slug);
+ if (!snapshot) {
+ throw new OpsInputError(
+ `Channel "${slug}" has no report yet — run /api/ops/refresh-report first.`,
+ );
+ }
+ const ids = (snapshot.buckets as Record<string, unknown>)[bucket];
+ if (!Array.isArray(ids)) {
+ throw new OpsInputError(
+ `"${bucket}" is not a bucket on this channel's report — known buckets: ${Object.keys(
+ snapshot.buckets,
+ ).join(", ")}`,
+ );
+ }
+ return jobResponse(
+ await retryBucketAction(
+ slug,
+ ids as string[],
+ optString(body, "queueKey"),
+ optBool(body, "abortOnError"),
+ optString(body, "handlingOverride"),
+ // Replayable only for the buckets a replay can re-derive; an
+ // ad-hoc bucket name still runs, it just carries no spec — the
+ // same distinction the UI's named vs checkbox controls make.
+ isReplayBucket(bucket) ? bucket : undefined,
+ optBool(body, "forceCookies"),
+ optBool(body, "replaceAutoSubs"),
+ ),
+ );
+ },
+ );
+}
diff --git a/editor/app/api/ops/sync/route.ts b/editor/app/api/ops/sync/route.ts
@@ -0,0 +1,20 @@
+import { syncAction } from "../../../channels/[slug]/pipelineActions";
+import { jobResponse, ops, optBool, optString, reqSlug } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug: string, full?: boolean, queueKey?: string } -> { ok: true, jobId }
+//
+// `full` forces the periodic whole-listing sweep now; without it the job
+// decides for itself from the configured cadence.
+export async function POST(request: Request) {
+ return ops(request, ["slug", "full", "queueKey"], async (body) =>
+ jobResponse(
+ await syncAction(
+ reqSlug(body, "slug"),
+ optString(body, "queueKey"),
+ optBool(body, "full"),
+ ),
+ ),
+ );
+}
diff --git a/editor/app/channels/components/channelConfigToForm.ts b/editor/app/channels/components/channelConfigToForm.ts
@@ -0,0 +1,186 @@
+import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig";
+
+// THE INVERSE OF parseChannelForm, and it exists for exactly one caller:
+// /api/ops/channel-config.
+//
+// WHY A ROUND TRIP THROUGH FormData rather than a direct config write. The form
+// parser is where a download-filter regex is refused, where a sync cadence is
+// bounded and where a cleared field is turned into an absent key — and
+// `updateChannelAction` DELETES every CHANNEL_FORM_FIELDS key from the stored
+// config before layering the parse result on top, so that a cleared input
+// actually clears. Feed it a FormData carrying only a patch and it would clear
+// everything the patch did not name. So the patch is applied ON TOP of the
+// channel's current form representation, which is what this builds, and the
+// action then behaves EXACTLY as it does for the browser form. One validator,
+// one writer, one set of error sentences.
+//
+// Every key below is a field name `parseChannelForm` reads. Checkbox fields
+// follow the browser: present means checked, absent means unchecked — which is
+// why they are emitted only when true.
+
+// The checkbox-shaped fields (presence, not value).
+export const CHANNEL_FORM_FLAGS = [
+ "keepSourceVideo",
+ "downloadFilterIncludeLivestreams",
+ "audioCheckEnabled",
+ "audioCheckResumeDuringProbe",
+] as const;
+
+// The value-shaped fields. A patch may set any of these to "" to clear it.
+export const CHANNEL_FORM_VALUES = [
+ "name",
+ "handling",
+ "url",
+ "platform",
+ "sourceKind",
+ "postFetcher",
+ "socialHandle",
+ "audioFormat",
+ "downloadFormat",
+ "keepLatest",
+ "extractionMode",
+ "savedVideosDir",
+ "ytdlpExtraArgs",
+ "sleepBetweenDownloadsSeconds",
+ "syncIntervalMinutes",
+ "fullSweepIntervalMinutes",
+ "cookiesFromBrowser",
+ "cookieMode",
+ "downloadFilterInclude",
+ "downloadFilterExclude",
+ "audioCheckIntervalSeconds",
+ "audioCheckMaxRollbacks",
+ "audioCheckCopyTimeoutSeconds",
+] as const;
+
+export type ChannelFormFlag = (typeof CHANNEL_FORM_FLAGS)[number];
+export type ChannelFormValue = (typeof CHANNEL_FORM_VALUES)[number];
+export type ChannelFormField = ChannelFormFlag | ChannelFormValue;
+
+export const CHANNEL_FORM_FIELD_NAMES: readonly ChannelFormField[] = [
+ ...CHANNEL_FORM_VALUES,
+ ...CHANNEL_FORM_FLAGS,
+];
+
+function put(fd: FormData, key: string, value: string | undefined | null): void {
+ if (value === undefined || value === null || value === "") return;
+ fd.set(key, value);
+}
+
+function putNum(fd: FormData, key: string, value: number | undefined): void {
+ if (value === undefined || value === null) return;
+ fd.set(key, String(value));
+}
+
+// The channel's stored config as the Configure form would post it.
+export function channelConfigToFormData(config: ChannelConfig): FormData {
+ const fd = new FormData();
+ // Both are REQUIRED by parseChannelForm and are not CHANNEL_FORM_FIELDS, so
+ // they always travel.
+ fd.set("name", config.name ?? "");
+ fd.set("handling", config.handling ?? "transcribe");
+ put(fd, "url", config.url);
+ put(fd, "platform", config.platform);
+ put(fd, "sourceKind", config.sourceKind);
+ put(fd, "postFetcher", config.postFetcher);
+ put(fd, "socialHandle", config.socialHandle);
+ put(fd, "audioFormat", config.audioFormat);
+ put(fd, "downloadFormat", config.downloadFormat);
+ if (config.keepSourceVideo) fd.set("keepSourceVideo", "on");
+ putNum(fd, "keepLatest", config.keepLatest);
+ put(fd, "extractionMode", config.extractionMode);
+ put(fd, "savedVideosDir", config.savedVideosDir);
+ if (config.ytdlpExtraArgs?.length) {
+ fd.set("ytdlpExtraArgs", config.ytdlpExtraArgs.join("\n"));
+ }
+ putNum(fd, "sleepBetweenDownloadsSeconds", config.sleepBetweenDownloadsSeconds);
+ putNum(fd, "syncIntervalMinutes", config.syncIntervalMinutes);
+ putNum(fd, "fullSweepIntervalMinutes", config.fullSweepIntervalMinutes);
+ put(fd, "cookiesFromBrowser", config.cookiesFromBrowser);
+ put(fd, "cookieMode", config.cookieMode);
+ put(fd, "downloadFilterInclude", config.downloadFilter?.include);
+ put(fd, "downloadFilterExclude", config.downloadFilter?.exclude);
+ if (config.downloadFilter?.includeLivestreams) {
+ fd.set("downloadFilterIncludeLivestreams", "on");
+ }
+ if (config.audioCheck?.enabled) {
+ fd.set("audioCheckEnabled", "on");
+ putNum(fd, "audioCheckIntervalSeconds", config.audioCheck.intervalSeconds);
+ putNum(fd, "audioCheckMaxRollbacks", config.audioCheck.maxRollbacks);
+ putNum(
+ fd,
+ "audioCheckCopyTimeoutSeconds",
+ config.audioCheck.copyTimeoutSeconds,
+ );
+ if (config.audioCheck.resumeDuringProbe) {
+ fd.set("audioCheckResumeDuringProbe", "on");
+ }
+ }
+ return fd;
+}
+
+export type ChannelConfigPatch = Record<string, unknown>;
+
+// Lay a JSON patch over a FormData built above. Values: a string (or number,
+// stringified) sets the field, `""` clears it; a flag takes a boolean.
+// `ytdlpExtraArgs` additionally accepts an array of lines, because that is the
+// shape it has in config.json and a caller should not have to know the textarea
+// is newline-separated.
+//
+// Throws on a key this form has no field for — the allow-list IS the documented
+// body shape, and a misspelled key that silently did nothing would look like a
+// successful save.
+export function applyChannelFormPatch(
+ fd: FormData,
+ patch: ChannelConfigPatch,
+): void {
+ validateChannelFormPatch(patch);
+ for (const [key, value] of Object.entries(patch)) {
+ if ((CHANNEL_FORM_FLAGS as readonly string[]).includes(key)) {
+ if (value) fd.set(key, "on");
+ else fd.delete(key);
+ continue;
+ }
+ if (key === "ytdlpExtraArgs" && Array.isArray(value)) {
+ const joined = (value as string[]).join("\n");
+ if (joined.trim()) fd.set(key, joined);
+ else fd.delete(key);
+ continue;
+ }
+ if (value === null || value === "") {
+ fd.delete(key);
+ continue;
+ }
+ fd.set(key, typeof value === "number" ? String(value) : (value as string));
+ }
+}
+
+// SHAPE ONLY, AND SEPARATE ON PURPOSE: a misspelled field name is a fact about
+// the REQUEST, so the route checks it before it looks the channel up. Otherwise
+// `{ slug: "typo", patch: { downloadFilterExcluded: … } }` reports the channel
+// and never mentions the key that was actually wrong.
+export function validateChannelFormPatch(patch: ChannelConfigPatch): void {
+ for (const [key, value] of Object.entries(patch)) {
+ if ((CHANNEL_FORM_FLAGS as readonly string[]).includes(key)) {
+ if (typeof value !== "boolean") {
+ throw new Error(`"${key}" must be a boolean`);
+ }
+ continue;
+ }
+ if (!(CHANNEL_FORM_VALUES as readonly string[]).includes(key)) {
+ throw new Error(
+ `"${key}" is not a channel config field — accepted: ${CHANNEL_FORM_FIELD_NAMES.join(", ")}`,
+ );
+ }
+ if (key === "ytdlpExtraArgs" && Array.isArray(value)) {
+ if (value.some((v) => typeof v !== "string")) {
+ throw new Error(`"${key}" must be an array of strings`);
+ }
+ continue;
+ }
+ if (value === null || typeof value === "number" || typeof value === "string") {
+ continue;
+ }
+ throw new Error(`"${key}" must be a string, a number, or "" to clear it`);
+ }
+}
diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts
@@ -0,0 +1,494 @@
+// /api/ops — the HTTP door onto the editor's server actions.
+//
+// What this spec is really pinning is the claim the layer rests on: that an
+// agent driving the editor over HTTP and an operator clicking the same button
+// get the SAME answer from the SAME code. So the assertions are deliberately
+// about the shared sentences — the download-filter message that
+// title-filter.spec.ts reads off the form, the busy-channel refusal the Storage
+// panel shows — and about the files on disk, not about the routes' own shapes.
+//
+// THE 503 BRANCH IS NOT REACHABLE FROM HERE. The test server runs with
+// WORKER_TOKEN=test-worker-token (editor/package.json, dev:test) and there is one
+// server for the whole suite, so no spec can observe the endpoint disabled. 401
+// (missing and wrong) is covered below; the 503 is authorizeWorkerRequest's own
+// first branch, shared with /api/worker/* and unit-tested by nothing else
+// either. See plans/FACTS.md.
+
+import { readdir, rm } from "node:fs/promises";
+import { test, expect, type APIRequestContext } from "@playwright/test";
+import { baseUrl } from "./baseUrl";
+import {
+ generateReport,
+ pathExists,
+ readJson,
+ resetData,
+ resolvePath,
+ writeChannelConfig,
+ writeSettings,
+} from "./helpers";
+
+const TOKEN = "test-worker-token";
+const AUTH = { authorization: `Bearer ${TOKEN}` };
+
+type OpsResponse = {
+ ok?: boolean;
+ error?: string;
+ jobId?: string;
+ queued?: string[];
+ skipped?: { slug: string; reason: string }[];
+};
+
+async function ops(
+ request: APIRequestContext,
+ action: string,
+ data: Record<string, unknown>,
+): Promise<{ status: number; body: OpsResponse }> {
+ const res = await request.post(`${baseUrl}/api/ops/${action}`, {
+ headers: AUTH,
+ data,
+ });
+ return { status: res.status(), body: (await res.json()) as OpsResponse };
+}
+
+// EVERY SPEC HERE NAMES THE DISK FLOOR. Without it the merged fixture default
+// applies, and a spec that starts a download-shaped job on a nearly-full host
+// would be refused by lowDiskError() with a returned { ok: false } no assertion
+// reads. 0 is the fixture's own value; naming it makes that a decision.
+async function settings(): Promise<void> {
+ await writeSettings({ minFreeDiskGB: 0 });
+}
+
+test("the token gate answers 401 for a missing and for a wrong bearer", async ({
+ request,
+}) => {
+ await resetData("empty");
+ for (const headers of [undefined, { authorization: "Bearer wrong" }]) {
+ const post = await request.post(`${baseUrl}/api/ops/refresh-report`, {
+ ...(headers ? { headers } : {}),
+ data: { all: true },
+ });
+ expect(post.status(), JSON.stringify(headers)).toBe(401);
+ // The READ side is gated by the same token, not merely the write side.
+ const get = await request.get(`${baseUrl}/api/ops/channel/anything`, {
+ ...(headers ? { headers } : {}),
+ });
+ expect(get.status(), JSON.stringify(headers)).toBe(401);
+ }
+});
+
+test("an unknown body key is a 400 that names the accepted keys", async ({
+ request,
+}) => {
+ await resetData("empty");
+ // A misspelled key would otherwise get a cheerful { ok: true } and a channel
+ // that did not change.
+ const { status, body } = await ops(request, "sync", {
+ slug: "x",
+ fullSweep: true,
+ });
+ expect(status).toBe(400);
+ expect(body.ok).toBe(false);
+ expect(body.error).toContain("unknown key(s): fullSweep");
+ expect(body.error).toContain("full");
+
+ // So is a nested one, on the route whose body carries an object.
+ const patch = await ops(request, "channel-config", {
+ slug: "x",
+ patch: { downloadFilterExcluded: "rerun" },
+ });
+ expect(patch.status).toBe(400);
+ expect(patch.body.error).toContain("downloadFilterExcluded");
+});
+
+test("a traversing slug is refused at the door, on every route that takes one", async ({
+ request,
+}) => {
+ await resetData("title-filter-channel");
+ const channelsDir = resolvePath("test-transcripts/channels");
+ const before = (await readdir(channelsDir)).sort();
+ expect(before).toEqual(["test-filter"]);
+
+ // EVERY SLUG BELOW REACHES A path.join UNDER channelsDir, and the readers
+ // swallow their own errors — so an unchecked traversing segment would fail
+ // SILENTLY (an empty config read as "channel not found") rather than loudly,
+ // and any future writer on that path would land outside the corpus. reqSlug
+ // is one check for all of them; this is the assertion that it is wired to
+ // each.
+ const cases: [string, Record<string, unknown>][] = [
+ ["metadata-scan", { slug: "../../escape" }],
+ ["sync", { slug: "../../escape" }],
+ ["download-missing", { slug: "../../escape" }],
+ ["import-video", { slug: "../../escape", url: "https://example.com/v" }],
+ ["retry-bucket", { slug: "../../escape", bucket: "noTranscript" }],
+ ["refresh-report", { slug: "../../escape" }],
+ ["channel-config", { slug: "../../escape", patch: { cookieMode: "always" } }],
+ ["channel-priority", { slugs: ["../../escape"], tier: "paused" }],
+ ["relocate", { slugs: ["../../escape"], root: "/tmp/ops-api-never" }],
+ ["relocate-back", { slugs: ["../../escape"] }],
+ ];
+ for (const [action, data] of cases) {
+ const { status, body } = await ops(request, action, data);
+ expect(status, action).toBe(400);
+ expect(body.error, action).toMatch(/is not a valid channel slug/);
+ }
+
+ // The read route takes its slug as a path SEGMENT rather than in a body, so
+ // it spells the same check out. A slash-bearing value is not the case to
+ // assert here — the router never matches one to a single dynamic segment, so
+ // it 404s before the handler exists. What DOES reach the handler is a
+ // one-segment name CHANNEL_SLUG_RE still refuses, and a leading dot is the
+ // one that matters: it is how a dotfile beside the channels dir would be
+ // named at.
+ const read = await request.get(`${baseUrl}/api/ops/channel/.escape`, {
+ headers: AUTH,
+ });
+ expect(read.status()).toBe(400);
+ expect(((await read.json()) as OpsResponse).error).toMatch(
+ /is not a valid channel slug/,
+ );
+
+ // NOTHING WAS TOUCHED: the corpus still holds exactly the fixture channel,
+ // and the fixture's own config is byte-identical.
+ expect((await readdir(channelsDir)).sort()).toEqual(before);
+ expect(
+ await readJson<Record<string, unknown>>(
+ "test-transcripts/channels/test-filter/config.json",
+ ),
+ ).toEqual({
+ handling: "youtube",
+ name: "Test Title Filter",
+ url: "https://www.youtube.com/@example/videos",
+ downloadFilter: { include: "guest" },
+ });
+});
+
+test("channel-config round-trips a download filter and refuses a bad regex", async ({
+ page,
+ request,
+}) => {
+ test.setTimeout(120_000);
+ await resetData("title-filter-channel");
+ await settings();
+ const SLUG = "test-filter";
+ const CONFIG = `test-transcripts/channels/${SLUG}/config.json`;
+ // TWO FIELDS THE CLEAR-THEN-LAYER PATH ACTUALLY THREATENS. name/handling/url
+ // survive a patch trivially — they are re-posted by the serializer because
+ // parseChannelForm requires the first two and CHANNEL_FORM_FIELDS spares
+ // none of the rest. `cookieMode` and `keepLatest` are in CHANNEL_FORM_FIELDS,
+ // so updateChannelAction deletes them from the baseline before layering, and
+ // a patch that failed to re-post them would silently clear both.
+ //
+ // `keepLatest: 0` is the sharp one: 0 is the explicit "disabled" sentinel, and
+ // a serializer that tested the value for truthiness rather than for
+ // null/undefined would drop it and read as "inherit" on the next save.
+ await writeChannelConfig(SLUG, {
+ handling: "youtube",
+ name: "Test Title Filter",
+ url: "https://www.youtube.com/@example/videos",
+ downloadFilter: { include: "guest" },
+ cookieMode: "always",
+ keepLatest: 0,
+ });
+ const before = await readJson<Record<string, unknown>>(CONFIG);
+ expect(before.downloadFilter).toEqual({ include: "guest" });
+ expect(before.cookieMode).toBe("always");
+ expect(before.keepLatest).toBe(0);
+
+ // THE SAME SENTENCE THE FORM SHOWS. title-filter.spec.ts reads this off the
+ // page after typing "elf(" into the exclude input; the route reaches it
+ // through the same parseChannelForm, which is the whole point of routing a
+ // patch through a FormData rather than writing config.json directly.
+ const bad = await ops(request, "channel-config", {
+ slug: SLUG,
+ patch: { downloadFilterExclude: "elf(" },
+ });
+ expect(bad.status).toBe(400);
+ expect(bad.body.error).toMatch(
+ /Download filter exclude is not a valid regular expression/,
+ );
+ // And nothing was written.
+ expect(
+ (await readJson<Record<string, unknown>>(CONFIG)).downloadFilter,
+ ).toEqual({ include: "guest" });
+
+ const good = await ops(request, "channel-config", {
+ slug: SLUG,
+ patch: { downloadFilterInclude: "guest|special", downloadFilterExclude: "rerun" },
+ });
+ expect(good.body).toEqual({ ok: true });
+ const after = await readJson<Record<string, unknown>>(CONFIG);
+ expect(after.downloadFilter).toEqual({
+ include: "guest|special",
+ exclude: "rerun",
+ });
+ // A PATCH IS A PATCH. updateChannelAction clears every form-managed key
+ // before layering the parse result on, so a route that posted only the patch
+ // would have silently dropped every one of these.
+ expect(after.name).toBe(before.name);
+ expect(after.handling).toBe(before.handling);
+ expect(after.url).toBe(before.url);
+ expect(after.cookieMode).toBe("always");
+ expect(after.keepLatest).toBe(0);
+
+ // "" clears a field, exactly as clearing the input does.
+ const cleared = await ops(request, "channel-config", {
+ slug: SLUG,
+ patch: { downloadFilterInclude: "", downloadFilterExclude: "" },
+ });
+ expect(cleared.body).toEqual({ ok: true });
+ const emptied = await readJson<Record<string, unknown>>(CONFIG);
+ expect("downloadFilter" in emptied).toBe(false);
+ // Clearing one field clears ONLY that field.
+ expect(emptied.cookieMode).toBe("always");
+ expect(emptied.keepLatest).toBe(0);
+
+ // The read route sees the same config, and says whether the media is there.
+ await generateReport(page, SLUG);
+ const read = await request.get(`${baseUrl}/api/ops/channel/${SLUG}`, {
+ headers: AUTH,
+ });
+ expect(read.status()).toBe(200);
+ const view = (await read.json()) as {
+ ok: boolean;
+ config: { name: string };
+ media: { status: string };
+ report: { totals: { videos: number } } | null;
+ };
+ expect(view.ok).toBe(true);
+ expect(view.config.name).toBe(before.name);
+ expect(view.media.status).toBe("in-place");
+ expect(view.report?.totals.videos).toBeGreaterThanOrEqual(0);
+
+ const missing = await request.get(`${baseUrl}/api/ops/channel/nope`, {
+ headers: AUTH,
+ });
+ expect(missing.status()).toBe(404);
+});
+
+test("channel-priority writes a per-operation override", async ({ request }) => {
+ await resetData("title-filter-channel");
+ await settings();
+ const SLUG = "test-filter";
+
+ const pinned = await ops(request, "channel-priority", {
+ slugs: [SLUG],
+ operation: "download",
+ tier: "paused",
+ });
+ expect(pinned.body).toEqual({ ok: true });
+ await expect
+ .poll(async () => {
+ const s = await readJson<{
+ channelPriority?: {
+ channels?: Record<string, { overrides?: Record<string, string> }>;
+ };
+ }>("test-settings.json");
+ return s.channelPriority?.channels?.[SLUG]?.overrides?.download ?? null;
+ })
+ .toBe("paused");
+
+ // null clears the override — and only with an operation named, because a bare
+ // null tier has no meaning for the BASE tier.
+ const bare = await ops(request, "channel-priority", {
+ slugs: [SLUG],
+ tier: null,
+ });
+ expect(bare.status).toBe(400);
+ expect(bare.body.error).toMatch(/clears an operation override/);
+
+ const cleared = await ops(request, "channel-priority", {
+ slugs: [SLUG],
+ operation: "download",
+ tier: null,
+ });
+ expect(cleared.body).toEqual({ ok: true });
+ await expect
+ .poll(async () => {
+ const s = await readJson<{
+ channelPriority?: {
+ channels?: Record<string, { overrides?: Record<string, string> }>;
+ };
+ }>("test-settings.json");
+ return s.channelPriority?.channels?.[SLUG]?.overrides?.download ?? null;
+ })
+ .toBe(null);
+
+ const bogus = await ops(request, "channel-priority", {
+ slugs: [SLUG],
+ tier: "urgent",
+ });
+ expect(bogus.status).toBe(400);
+ expect(bogus.body.error).toMatch(/normal, low, paused/);
+});
+
+test("metadata-scan starts a job, and the job says it is a metadata-scan", async ({
+ request,
+}) => {
+ test.setTimeout(120_000);
+ await resetData("title-filter-channel");
+ await settings();
+ const SLUG = "test-filter";
+
+ const { status, body } = await ops(request, "metadata-scan", { slug: SLUG });
+ expect(status).toBe(200);
+ expect(body.ok).toBe(true);
+ expect(body.jobId).toBeTruthy();
+
+ // THE ROUTE DOES NOT STREAM, so the id is the whole contract: the caller
+ // follows the same log endpoint the editor's own panel polls.
+ // The sidecar is written with a `void` promise right after enqueue, so poll.
+ const metaPath = `test-transcripts/.jobs/${body.jobId}.meta.json`;
+ await expect
+ .poll(async () => {
+ const meta = await readJson<{ kind: string; channelSlug: string }>(
+ metaPath,
+ ).catch(() => null);
+ return meta ? `${meta.kind}/${meta.channelSlug}` : null;
+ })
+ .toBe(`metadata-scan/${SLUG}`);
+
+ const log = await request.get(
+ `${baseUrl}/api/jobs/${body.jobId}/log?from=0`,
+ );
+ expect(log.status()).toBe(200);
+ const payload = (await log.json()) as { status: string };
+ expect(
+ ["queued", "running", "done", "failed", "cancelled"].includes(
+ payload.status,
+ ),
+ ).toBe(true);
+
+ const unknownChannel = await ops(request, "metadata-scan", { slug: "nope" });
+ expect(unknownChannel.status).toBe(400);
+ expect(unknownChannel.body.error).toContain('Channel "nope" not found');
+});
+
+test("refresh-report regenerates snapshot.json", async ({ request }) => {
+ test.setTimeout(120_000);
+ await resetData("title-filter-channel");
+ await settings();
+ const SLUG = "test-filter";
+ const SNAP = `test-transcripts/channels/${SLUG}/snapshot.json`;
+ // Removed rather than assumed absent: a debounced regen armed by the previous
+ // spec can land between resetData's copy and here, and the point of the
+ // assertion below is the ROUTE's effect, not the scheduler's.
+ await rm(resolvePath(SNAP), { force: true });
+ expect(await pathExists(SNAP)).toBe(false);
+
+ // No page, no click: the route IS the refresh. It regenerates SYNCHRONOUSLY
+ // (a filesystem scan, not a job), so { ok: true } means the file is there.
+ const first = await ops(request, "refresh-report", { slug: SLUG });
+ expect(first.body).toEqual({ ok: true });
+ const snapshot = await readJson<{ generatedAt: string; totals: { videos: number } }>(SNAP);
+ expect(snapshot.generatedAt).toBeTruthy();
+ expect(snapshot.totals.videos).toBeGreaterThanOrEqual(0);
+
+ // Re-running REWRITES it. Polled through the action itself because two scans
+ // of a six-video fixture can land in the same millisecond.
+ await expect
+ .poll(async () => {
+ await ops(request, "refresh-report", { slug: SLUG });
+ return (await readJson<{ generatedAt: string }>(SNAP)).generatedAt;
+ })
+ .not.toBe(snapshot.generatedAt);
+
+ // The bulk form queues a job per channel and reports both lists.
+ const all = await ops(request, "refresh-report", { all: true });
+ expect(all.status).toBe(200);
+ expect(all.body.ok).toBe(true);
+ expect([...(all.body.queued ?? []), ...(all.body.skipped ?? []).map((s) => s.slug)]).toContain(
+ SLUG,
+ );
+
+ const both = await ops(request, "refresh-report", { slug: SLUG, all: true });
+ expect(both.status).toBe(400);
+ expect(both.body.error).toMatch(/not both/);
+});
+
+test("relocate refuses a busy channel with the sentence the panel shows", async ({
+ request,
+}, testInfo) => {
+ test.setTimeout(120_000);
+ await resetData("slow-pipeline-channel");
+ await settings();
+ const SLUG = "slow-channel";
+
+ // --test-slow makes the fake yt-dlp sleep 30s, so the channel is genuinely
+ // busy for the length of this assertion rather than racily so.
+ const started = await ops(request, "sync", { slug: SLUG });
+ expect(started.body.ok).toBe(true);
+ await expect
+ .poll(async () => {
+ const res = await request.get(`${baseUrl}/api/jobs/active`);
+ const body = (await res.json()) as
+ | { channelSlug?: string; status: string }[]
+ | { jobs?: { channelSlug?: string; status: string }[] };
+ const jobs = Array.isArray(body) ? body : (body.jobs ?? []);
+ return jobs.filter(
+ (j) =>
+ j.channelSlug === SLUG &&
+ (j.status === "running" || j.status === "queued"),
+ ).length;
+ })
+ .toBeGreaterThan(0);
+
+ const root = testInfo.outputPath("media-root");
+ const refused = await ops(request, "relocate", { slugs: [SLUG], root });
+ expect(refused.status).toBe(200);
+ // A SKIP IS NOT A FAILURE — the bulk bar renders both numbers, and so does
+ // this. The reason is channelMediaBusyReason's, word for word.
+ expect(refused.body.queued).toEqual([]);
+ expect(refused.body.skipped?.[0]?.slug).toBe(SLUG);
+ expect(refused.body.skipped?.[0]?.reason).toMatch(
+ /running\/queued job\(s\) for this channel/,
+ );
+
+ // Exactly one destination, and it must be absolute.
+ const neither = await ops(request, "relocate", { slugs: [SLUG] });
+ expect(neither.status).toBe(400);
+ expect(neither.body.error).toMatch(/exactly one of "locationId".*or "root"/);
+});
+
+test("lane flips a hold, and /api/auto-queue/status agrees", async ({
+ request,
+}) => {
+ await resetData("empty");
+ await settings();
+
+ const held = async (): Promise<boolean> => {
+ const res = await request.get(`${baseUrl}/api/auto-queue/status`);
+ const body = (await res.json()) as Record<string, { held?: boolean }>;
+ return body.download?.held === true;
+ };
+ expect(await held()).toBe(false);
+
+ const hold = await ops(request, "lane", { lane: "download", held: true });
+ expect(hold.body).toEqual({ ok: true, lane: "download" });
+ await expect.poll(held).toBe(true);
+
+ const release = await ops(request, "lane", { lane: "download", held: false });
+ expect(release.body).toEqual({ ok: true, lane: "download" });
+ await expect.poll(held).toBe(false);
+
+ // `enabled` is the policy's master switch and is NOT the gate — two controls
+ // in the UI, two keys here.
+ const enable = await ops(request, "lane", { lane: "download", enabled: true });
+ expect(enable.body.ok).toBe(true);
+ await expect
+ .poll(async () => {
+ const s = await readJson<{
+ autoQueue?: Record<string, { enabled?: boolean }>;
+ }>("test-settings.json");
+ return s.autoQueue?.download?.enabled ?? null;
+ })
+ .toBe(true);
+ await ops(request, "lane", { lane: "download", enabled: false });
+
+ const nothing = await ops(request, "lane", { lane: "download" });
+ expect(nothing.status).toBe(400);
+ expect(nothing.body.error).toMatch(/nothing to do/);
+
+ const bogus = await ops(request, "lane", { lane: "transcode", held: true });
+ expect(bogus.status).toBe(400);
+ expect(bogus.body.error).toMatch(/transcription, download, digest, backfill/);
+});
diff --git a/package.json b/package.json
@@ -22,7 +22,8 @@
"wt": "node scripts/worktree.mjs",
"e2e:sharded": "node scripts/run-sharded-e2e.mjs",
"test:scripts": "node --test scripts/*.test.mjs umtool/report-to-video/*.test.mjs",
- "lint": "pnpm --filter export run lint"
+ "lint": "pnpm --filter export run lint",
+ "ops": "node scripts/archilyzer-ops.mjs"
},
"devDependencies": {
"tsx": "^4.21.0"
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -4313,3 +4313,65 @@ Six things, in the order they were built. Trust these over re-deriving them.
refuses work.
- The flag: `autoPauseReasonOf` is the one sentence, rendered as a "storage" chip
by `ChannelTierSelect` (`aria-label="auto-paused reason for <slug>"`).
+
+---
+
+## The ops API (`/api/ops/*`) — added 2026-09-20
+
+**It is adapters, and nothing else.** Each route under
+`editor/app/api/ops/<action>/route.ts` is ~5 lines: validate a JSON body, call
+ONE existing server action, map its result. No route contains a rule the UI does
+not already enforce — the point of the layer is that an agent over HTTP and an
+operator clicking the same button get the same refusal, with the same sentence,
+from the same code. A check written in a route would be a second opinion nobody
+maintains. `editor/app/api/ops/_lib.ts` holds auth, body parsing and the three
+result mappers (`jobResponse` / `actionResponse` / `queueResponse`).
+
+**`WORKER_TOKEN` is shared with `/api/worker/*` by design.** It already means
+"this instance accepts instructions from something that is not the browser in
+front of it", and the failure modes are identical, so the gate is: unset → 503
+(the surface is off until you opt in), wrong → 401. `authorizeWorkerRequest`
+(`common/lib/workerToken.ts`) is the single implementation; nothing new was
+written. A second secret would be a second thing to distribute, rotate and leave
+unset.
+
+**A job-starting route returns `{ ok: true, jobId }` and NEVER streams.**
+`runManagedFunction` hands back a `ReadableStream` the browser consumes; an HTTP
+caller wants to hang up and poll. Every job adapter calls `result.stream.cancel()`
+— which stops pushing into the controller and leaves the on-disk log running
+(`streamCommand.ts`'s `cancel()` note) — and the caller follows
+`/api/jobs/<id>/log`. Returning the id is also the honest answer: the queue may
+hold the job behind other work for hours, so "started" is not "running".
+
+**Unknown body keys are a 400, never a silent ignore.** The allow-list passed to
+`ops()` IS the route's documented body shape. A caller that misspells
+`downloadFilterExclude` would otherwise get `{ ok: true }` and a channel that
+still downloads everything.
+
+**`ops/channel-config`'s patch keys are the FORM's field names, not
+`ChannelConfig`'s** — `downloadFilterInclude` / `downloadFilterExclude` rather
+than a `downloadFilter` object, `ytdlpExtraArgs` as a string or an array of
+lines. That is what routes them through `parseChannelForm`'s validators. The
+patch is laid over the channel's CURRENT form representation
+(`editor/app/channels/components/channelConfigToForm.ts`) rather than posted
+alone, because `updateChannelAction` deletes every `CHANNEL_FORM_FIELDS` key from
+the stored config before layering the parse result on — a FormData carrying only
+a patch would clear everything the patch did not name.
+
+**`GET /api/ops/channel/<slug>`'s live counts are opt-in (`?counts=1`).**
+`readChannelStat` walks every video directory — eleven thousand readdirs on the
+largest channel here — so a poll loop asking for a channel's state would be the
+thing hammering the platter. The report's totals answer the same question from a
+file. The field is `counts` and is null without the flag.
+
+**No action needed a refactor to be callable from a route.** `revalidatePath` is
+supported in Route Handlers (Next 16 —
+`docs/01-app/03-api-reference/04-functions/revalidatePath.md:10`), no adapted
+action calls `cookies()`, and the only actions that `redirect()`
+(`createChannelAction`, `gotoVideoAction`) are deliberately not exposed.
+
+**The 503-when-unset branch is not covered by e2e.** The editor test server runs
+with `WORKER_TOKEN=test-worker-token` in `editor/package.json`'s `dev:test`, one
+server for the whole suite, so no spec can observe the disabled state.
+`editor/e2e/ops-api.spec.ts` covers 401 (missing and wrong); the 503 is
+`authorizeWorkerRequest`'s own first branch, shared with `/api/worker/*`.
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -0,0 +1,245 @@
+#!/usr/bin/env node
+// archilyzer-ops — drive a running editor over HTTP, without a browser.
+//
+// Every editor gesture used to be reachable only as a server action, which meant
+// an agent that wanted to sync a channel or fix a download filter had to drive
+// Playwright. /api/ops is a thin adapter layer over those same actions, and this
+// is its client.
+//
+// USAGE
+//
+// pnpm ops <action> [--json '<body>'] [--wait] [--quiet]
+// pnpm ops get channel <slug> [--counts]
+// pnpm ops list
+//
+// ARCHILYZER_EDITOR_URL editor base URL (default http://localhost:3001)
+// WORKER_TOKEN the shared secret the editor is running with.
+// Unset on the SERVER => every route 503s; unset here
+// => every route 401s.
+//
+// EXAMPLES
+//
+// pnpm ops sync --json '{"slug":"the-quartering"}' --wait
+// pnpm ops metadata-scan --json '{"slug":"the-quartering"}'
+// pnpm ops channel-config --json '{"slug":"x","patch":{"downloadFilterExclude":"rerun"}}'
+// pnpm ops channel-priority --json '{"slugs":["x"],"operation":"download","tier":"paused"}'
+// pnpm ops lane --json '{"lane":"download","held":true}'
+// pnpm ops refresh-report --json '{"all":true}'
+// pnpm ops relocate --json '{"slugs":["x"],"locationId":"platter"}'
+// pnpm ops get channel the-quartering
+//
+// --wait follows /api/jobs/<jobId>/log to the end for a job-starting action and
+// exits 0 only if the job finished `done`. Without it the command returns as
+// soon as the job is QUEUED, which is the honest answer: the queue may hold it
+// behind other work for hours.
+//
+// The response JSON is printed verbatim on stdout (log lines from --wait go to
+// stderr), so `pnpm ops … | jq` works.
+
+const DEFAULT_URL = "http://localhost:3001";
+
+// The read-side routes, reachable as `get <noun> <arg>`. Kept tiny and explicit:
+// an ops API that let a caller assemble arbitrary GET paths would be a proxy,
+// not an adapter.
+const GETTERS = {
+ // --counts adds the LIVE on-disk counts, which walk every video directory —
+ // opt-in for the same reason the route makes it opt-in.
+ channel: (slug, counts) =>
+ `/api/ops/channel/${encodeURIComponent(slug)}${counts ? "?counts=1" : ""}`,
+};
+
+const ACTIONS = [
+ "channel-priority",
+ "channel-config",
+ "metadata-scan",
+ "import-video",
+ "refresh-report",
+ "sync",
+ "download-missing",
+ "retry-bucket",
+ "build-index",
+ "build-deploy",
+ "build-site",
+ "relocate",
+ "relocate-back",
+ "lane",
+];
+
+export function parseArgs(argv) {
+ const positional = [];
+ let json = null;
+ let wait = false;
+ let quiet = false;
+ let counts = false;
+ for (let i = 0; i < argv.length; i++) {
+ const arg = argv[i];
+ if (arg === "--wait") {
+ wait = true;
+ } else if (arg === "--quiet") {
+ quiet = true;
+ } else if (arg === "--counts") {
+ counts = true;
+ } else if (arg === "--json") {
+ json = argv[++i];
+ if (json === undefined) {
+ return { error: "--json needs a JSON object argument" };
+ }
+ } else if (arg.startsWith("--json=")) {
+ json = arg.slice("--json=".length);
+ } else if (arg === "--help" || arg === "-h") {
+ return { help: true };
+ } else if (arg.startsWith("-")) {
+ return { error: `unknown flag: ${arg}` };
+ } else {
+ positional.push(arg);
+ }
+ }
+ if (positional.length === 0) return { help: true };
+ let body = {};
+ if (json !== null) {
+ try {
+ body = JSON.parse(json);
+ } catch (e) {
+ return { error: `--json is not valid JSON: ${e.message}` };
+ }
+ if (typeof body !== "object" || body === null || Array.isArray(body)) {
+ return { error: "--json must be a JSON object" };
+ }
+ }
+ if (positional[0] === "list") {
+ return { list: true };
+ }
+ if (positional[0] === "get") {
+ const noun = positional[1];
+ if (!noun || !GETTERS[noun]) {
+ return {
+ error: `get: unknown noun "${noun ?? ""}" — known: ${Object.keys(GETTERS).join(", ")}`,
+ };
+ }
+ if (!positional[2]) return { error: `get ${noun}: needs an argument` };
+ return {
+ method: "GET",
+ path: GETTERS[noun](positional[2], counts),
+ wait: false,
+ quiet,
+ };
+ }
+ const action = positional[0];
+ if (!ACTIONS.includes(action)) {
+ return {
+ error: `unknown action "${action}" — known: ${ACTIONS.join(", ")}`,
+ };
+ }
+ if (positional.length > 1) {
+ return {
+ error: `"${action}" takes no positional arguments — pass its body with --json`,
+ };
+ }
+ return { method: "POST", path: `/api/ops/${action}`, body, wait, quiet };
+}
+
+export function usage() {
+ return [
+ "Usage: pnpm ops <action> [--json '<body>'] [--wait]",
+ " pnpm ops get channel <slug> [--counts]",
+ " pnpm ops list",
+ "",
+ `Actions: ${ACTIONS.join(", ")}`,
+ "",
+ "Env: ARCHILYZER_EDITOR_URL (default http://localhost:3001), WORKER_TOKEN",
+ ].join("\n");
+}
+
+function baseUrl() {
+ return (process.env.ARCHILYZER_EDITOR_URL ?? DEFAULT_URL).replace(/\/+$/, "");
+}
+
+function authHeaders() {
+ const token = process.env.WORKER_TOKEN ?? "";
+ return token ? { authorization: `Bearer ${token}` } : {};
+}
+
+// Follow a job's log to its terminal state. Returns the status string.
+// Deliberately polls the SAME endpoint the editor's own log panel does, so a
+// job started here and a job started by a click are observed identically.
+async function followJob(jobId, quiet) {
+ let from = 0;
+ for (;;) {
+ const res = await fetch(
+ `${baseUrl()}/api/jobs/${encodeURIComponent(jobId)}/log?from=${from}`,
+ { headers: authHeaders() },
+ );
+ if (!res.ok) throw new Error(`log poll failed: HTTP ${res.status}`);
+ const payload = await res.json();
+ if (payload.content && !quiet) process.stderr.write(payload.content);
+ from = payload.nextOffset ?? from;
+ const status = payload.status;
+ if (status !== "queued" && status !== "running") return status;
+ await new Promise((r) => setTimeout(r, 1000));
+ }
+}
+
+async function main() {
+ const parsed = parseArgs(process.argv.slice(2));
+ if (parsed.help) {
+ console.log(usage());
+ return 0;
+ }
+ if (parsed.error) {
+ console.error(parsed.error);
+ console.error("");
+ console.error(usage());
+ return 2;
+ }
+ if (parsed.list) {
+ console.log(ACTIONS.join("\n"));
+ return 0;
+ }
+ const url = `${baseUrl()}${parsed.path}`;
+ const res = await fetch(url, {
+ method: parsed.method,
+ headers: {
+ ...authHeaders(),
+ ...(parsed.method === "POST" ? { "content-type": "application/json" } : {}),
+ },
+ ...(parsed.method === "POST" ? { body: JSON.stringify(parsed.body) } : {}),
+ });
+ const text = await res.text();
+ let payload;
+ try {
+ payload = JSON.parse(text);
+ } catch {
+ console.error(`HTTP ${res.status}: ${text.slice(0, 500)}`);
+ return 1;
+ }
+ console.log(JSON.stringify(payload, null, 2));
+ if (!res.ok || payload.ok === false) return 1;
+ if (!parsed.wait) return 0;
+ const jobIds = payload.jobId
+ ? [payload.jobId]
+ : Array.isArray(payload.jobs)
+ ? payload.jobs.map((j) => j.jobId)
+ : [];
+ if (jobIds.length === 0) {
+ // Not a job-starting action (or it queued nothing). --wait is satisfied.
+ return 0;
+ }
+ let worst = 0;
+ for (const jobId of jobIds) {
+ const status = await followJob(jobId, parsed.quiet);
+ console.error(`[${jobId}] ${status}`);
+ if (status !== "done") worst = 1;
+ }
+ return worst;
+}
+
+// Importable for the arg-parsing tests; only the CLI entry point runs main().
+if (process.argv[1] && import.meta.url === `file://${process.argv[1]}`) {
+ main().then(
+ (code) => process.exit(code),
+ (e) => {
+ console.error(e.message);
+ process.exit(1);
+ },
+ );
+}
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -0,0 +1,72 @@
+// Arg parsing for scripts/archilyzer-ops.mjs. No network: parseArgs is pure and
+// returns the request it WOULD make, which is the whole surface worth pinning —
+// the routes themselves are covered by editor/e2e/ops-api.spec.ts.
+//
+// Run with: pnpm test:scripts
+import assert from "node:assert/strict";
+import test from "node:test";
+import { parseArgs, usage } from "./archilyzer-ops.mjs";
+
+test("no arguments prints usage", () => {
+ assert.equal(parseArgs([]).help, true);
+ assert.match(usage(), /pnpm ops <action>/);
+});
+
+test("an action becomes a POST to its route", () => {
+ const p = parseArgs(["sync", "--json", '{"slug":"x","full":true}']);
+ assert.equal(p.method, "POST");
+ assert.equal(p.path, "/api/ops/sync");
+ assert.deepEqual(p.body, { slug: "x", full: true });
+ assert.equal(p.wait, false);
+});
+
+test("--wait and --json= are both accepted", () => {
+ const p = parseArgs(["metadata-scan", '--json={"slug":"x"}', "--wait"]);
+ assert.equal(p.wait, true);
+ assert.deepEqual(p.body, { slug: "x" });
+});
+
+test("an action with no body posts an empty object", () => {
+ const p = parseArgs(["build-index"]);
+ assert.deepEqual(p.body, {});
+});
+
+test("an unknown action is refused by name, with the list", () => {
+ const p = parseArgs(["sinc"]);
+ assert.match(p.error, /unknown action "sinc"/);
+ assert.match(p.error, /metadata-scan/);
+});
+
+test("malformed --json is refused before any request", () => {
+ assert.match(parseArgs(["sync", "--json", "{"]).error, /not valid JSON/);
+ assert.match(parseArgs(["sync", "--json", "[1]"]).error, /must be a JSON object/);
+ assert.match(parseArgs(["sync", "--json"]).error, /needs a JSON object/);
+});
+
+test("positional arguments after an action are refused", () => {
+ // `pnpm ops sync the-quartering` reads naturally and would otherwise be a
+ // silent no-op body, so it is an error that names the fix.
+ assert.match(parseArgs(["sync", "the-quartering"]).error, /--json/);
+});
+
+test("get channel becomes a GET on the read route", () => {
+ const p = parseArgs(["get", "channel", "the quartering"]);
+ assert.equal(p.method, "GET");
+ assert.equal(p.path, "/api/ops/channel/the%20quartering");
+});
+
+test("get refuses an unknown noun and a missing argument", () => {
+ assert.match(parseArgs(["get", "site", "x"]).error, /unknown noun/);
+ assert.match(parseArgs(["get", "channel"]).error, /needs an argument/);
+});
+
+test("an unknown flag is refused", () => {
+ assert.match(parseArgs(["sync", "--force"]).error, /unknown flag/);
+});
+
+test("get channel --counts asks for the live on-disk counts", () => {
+ const p = parseArgs(["get", "channel", "x", "--counts"]);
+ assert.equal(p.path, "/api/ops/channel/x?counts=1");
+ // Off by default: the counts walk every video directory.
+ assert.equal(parseArgs(["get", "channel", "x"]).path, "/api/ops/channel/x");
+});
diff --git a/umtool/components/projects/ClipBench.tsx b/umtool/components/projects/ClipBench.tsx
@@ -183,7 +183,13 @@ export type ClipBenchData = {
uploadDate: string | null;
/** Who the header will name, already resolved server-side. Heads the line. */
sourceChannel: string | null;
- /** Neighbours in the cut, computed server-side. Null at either end. */
+ /**
+ * Neighbours ON THE WALK, computed server-side: the nearest clip either side
+ * that still needs judgement and is fetched. Null when there is none that
+ * way, which is what "first" / "last" say. Not the neighbours in the cut --
+ * a clip with nothing to play is a dead end, and one already answered is the
+ * round trip the walk exists to remove.
+ */
prev: string | null;
next: string | null;
/** Where this clip sits in the cut, 1-based, and how long the cut is. */
@@ -195,6 +201,15 @@ export type ClipBenchData = {
* -- and without this component having to guess at the rest of the cut.
*/
reviewedOthers: number;
+ /**
+ * The walk's own reading: `ready` clips are fetched AND still need judgement,
+ * out of `needing` that need it at all. The gap between the two numbers is
+ * what is waiting on a download rather than on you.
+ */
+ ready: number;
+ needing: number;
+ /** Is there a cached file holding this clip's window, end to end? */
+ fetched: boolean;
cues: Cue[];
token: string | null;
fetchPad: number;
@@ -614,6 +629,32 @@ export default function ClipBench({ data }: { data: ClipBenchData }) {
[draft, clip, save, needNote],
);
+ // ---- the window patch, one rule, two callers ------------------------------
+ //
+ // `save window` writes it on demand; a confirmation carries it when the edges
+ // have actually moved. Both need the same rule about a cut the new extent no
+ // longer contains, so the rule lives here rather than in each of them.
+ const windowMoved =
+ Math.abs(round2(sel.from) - clip.start) > 0.02 || Math.abs(round2(sel.to) - clip.end) > 0.02;
+
+ const windowPatch = useCallback((): {
+ start: number;
+ end: number;
+ cutStart?: string;
+ cutEnd?: string;
+ } => {
+ const start = round2(sel.from);
+ const end = round2(sel.to);
+ const cutOutside =
+ clip.cutStart != null &&
+ clip.cutEnd != null &&
+ (clip.cutStart < start - 0.02 || clip.cutEnd > end + 0.02);
+ // The extent is the judgement being made right now; the cut was derived
+ // from a wider one and is no longer inside it. Clearing it in the SAME
+ // patch is what keeps the writer's rule and the screen agreeing.
+ return cutOutside ? { start, end, cutStart: "", cutEnd: "" } : { start, end };
+ }, [sel.from, sel.to, clip.cutStart, clip.cutEnd]);
+
// ---- the walk's verdict ---------------------------------------------------
//
// "Is this clip what the report says it is" is the question the walk exists
@@ -622,12 +663,23 @@ export default function ClipBench({ data }: { data: ClipBenchData }) {
// case; saying no stays put, because the note has to be typed -- and then
// re-read, which is why it still stays put once it is saved.
const confirmClip = useCallback(async () => {
+ // A MOVED WINDOW RIDES ALONG. Confirming a clip whose edges you just nudged
+ // is a judgement about THAT window, and the advance would otherwise walk
+ // away from it -- so the edges go in the SAME patch as the verdict rather
+ // than needing `save window` pressed first. One write, one token.
+ const win = windowMoved ? windowPatch() : null;
// A note survives a confirmation. It stops being a complaint and becomes
// what it now says it is: why this clip is here in the shape it is in.
- const ok = await save({ verdict: "confirmed" });
+ const ok = await save({ verdict: "confirmed", ...(win ?? {}) });
if (ok) setNeedNote(false);
+ if (ok && win)
+ setNote(
+ win.cutStart !== undefined
+ ? "saved — the window moved with it, and the cut no longer fitted and was cleared"
+ : "saved — the window you moved was saved with it",
+ );
if (ok && data.next) router.push(`/browse/${data.project}/clip/${data.next}`);
- }, [save, router, data.project, data.next]);
+ }, [save, router, data.project, data.next, windowMoved, windowPatch]);
const rejectClip = useCallback(() => {
setNeedNote(true);
@@ -910,23 +962,12 @@ export default function ClipBench({ data }: { data: ClipBenchData }) {
/** Save the window, and drop a cut the new extent no longer contains. */
const saveWindow = useCallback(() => {
- const start = round2(sel.from);
- const end = round2(sel.to);
- const cutOutside =
- clip.cutStart != null &&
- clip.cutEnd != null &&
- (clip.cutStart < start - 0.02 || clip.cutEnd > end + 0.02);
- if (cutOutside) {
- // The extent is the judgement being made right now; the cut was derived
- // from a wider one and is no longer inside it. Clearing it in the SAME
- // patch is what keeps the writer's rule and the screen agreeing.
- void save({ start, end, cutStart: "", cutEnd: "" }).then((ok) => {
- if (ok) setNote("saved — the cut no longer fitted this window and was cleared");
- });
- return;
- }
- void save({ start, end });
- }, [sel.from, sel.to, clip.cutStart, clip.cutEnd, save]);
+ const patch = windowPatch();
+ void save(patch).then((ok) => {
+ if (ok && patch.cutStart !== undefined)
+ setNote("saved — the cut no longer fitted this window and was cleared");
+ });
+ }, [windowPatch, save]);
// ---- the warnings --------------------------------------------------------
const endCue = cues.find((c) => sel.to >= c.start - 0.02 && sel.to <= c.end + 0.02) ?? null;
@@ -1061,6 +1102,16 @@ export default function ClipBench({ data }: { data: ClipBenchData }) {
<span className="micro" data-review-progress="">
{reviewed} reviewed
</span>
+ {/* What the WALK has left, which is not the same number: a clip nobody
+ has fetched still needs judgement and cannot get it today. */}
+ <span className="micro" data-ready-count="">
+ ready {data.ready} of {data.needing} needing judgement
+ </span>
+ {!data.fetched && (
+ <span className={badgeVariants({ variant: "open", size: "sm" })} data-clip-unfetched="">
+ not fetched yet
+ </span>
+ )}
{data.prev ? (
<Link
data-clip-nav="prev"
diff --git a/umtool/components/projects/ClipBenchPage.tsx b/umtool/components/projects/ClipBenchPage.tsx
@@ -9,8 +9,9 @@ import {
readClipDetail,
readManifest,
siblingsOf,
+ walkReadiness,
} from "@/lib/projects/report.mjs";
-import { segmentFor, windowsFor } from "@/lib/report/serve.mjs";
+import { rawCacheOf, segmentFor, windowsFor } from "@/lib/report/serve.mjs";
import { FETCH_MAX_PAD } from "@/lib/report/driver.mjs";
import type { ProjectRef } from "@/lib/project-types";
@@ -38,7 +39,24 @@ export default async function ClipBenchPage({
const clips = detail.entries.filter((e: { kind: string }) => e.kind === "clip");
const i = clips.findIndex((e: { id: string }) => e.id === clipId);
- const windows = await windowsFor(project, entry);
+ // ---- where `p` and `n` go ------------------------------------------------
+ //
+ // The walk visits the clips that still NEED JUDGEMENT and are FETCHED, in
+ // timeline order. A clip nobody can play is a dead end -- there is nothing to
+ // judge and the only move is to press `n` again -- and one already answered is
+ // the round trip a walk exists to remove. A URL still opens ANY clip: this
+ // page renders it, says what it has (or has not) got, and points its links at
+ // the fetched, unjudged neighbours on either side.
+ const readiness = walkReadiness(clips);
+ const inWalk = new Set<string>(readiness.readyIds as string[]);
+ const onWalk = (e: { id: string }) => e.id !== clipId && inWalk.has(e.id);
+ const prev =
+ [...clips.slice(0, Math.max(0, i))].reverse().find(onWalk)?.id ?? null;
+ const next = clips.slice(i + 1).find(onWalk)?.id ?? null;
+
+ // One listing of clips-raw for the page: readClipDetail already built one to
+ // answer `fetched` for every clip, and this is the same question for this one.
+ const windows = await windowsFor(project, entry, await rawCacheOf(project.dir));
// The mtime rides along so a re-render busts the browser's cache: the segment
// is written back to the SAME path, and without it the player would keep
// showing the cut from before the edit.
@@ -91,12 +109,17 @@ export default async function ClipBenchPage({
sourceTitle: entry.sourceTitleClean ?? null,
uploadDate: entry.uploadDate ?? null,
sourceChannel: entry.sourceChannel ?? null,
- prev: i > 0 ? String(clips[i - 1].id) : null,
- next: i >= 0 && i < clips.length - 1 ? String(clips[i + 1].id) : null,
+ prev: prev,
+ next: next,
index: i + 1,
total: clips.length,
// The OTHERS, so the bench can add this clip's own answer to the count
// without a reload and without knowing anything about the rest of the cut.
+ /** The walk's own reading: fetched clips that still need an answer. */
+ ready: readiness.ready,
+ needing: readiness.needing,
+ /** Is there a cached file holding THIS clip's window, end to end? */
+ fetched: !!entry.fetched,
reviewedOthers: clips.filter(
(e: { id: string }) => e.id !== clipId && clipVerdict(e) !== "unreviewed",
).length,
diff --git a/umtool/components/projects/ReportProject.tsx b/umtool/components/projects/ReportProject.tsx
@@ -6,7 +6,13 @@ import CopyButton from "@/components/CopyButton";
import CheckSourcesButton from "@/components/dashboard/CheckSourcesButton";
import { fmtAgo, fmtBytes } from "@/lib/format";
import { Markdown } from "@/lib/markdown";
-import { correctionsOf, readClipDetail, reviewOf, sourcesOf } from "@/lib/projects/report.mjs";
+import {
+ correctionsOf,
+ readClipDetail,
+ reviewOf,
+ sourcesOf,
+ walkReadiness,
+} from "@/lib/projects/report.mjs";
import { EXPORT_FORMATS, exportableVariants } from "@/lib/report/export.mjs";
import { diffManifests, formatChange } from "@/lib/report/manifest-diff.mjs";
import { listSnapshots, readSnapshot } from "@/lib/report/snapshots.mjs";
@@ -85,6 +91,11 @@ export default async function ReportProject({
// How much of the cut has been walked. One definition, shared with the bench
// header and `umtool corrections`.
const review = reviewOf(m);
+ // And how much of it can be walked TODAY: the walk visits the clips that
+ // still need judgement and have something to play, so the gap between these
+ // two numbers is what is waiting on a download rather than on somebody.
+ const readiness = walkReadiness(clips);
+ const walkStart = (readiness.readyIds[0] as string | undefined) ?? clips[0]?.id;
const nonClips = entries.filter((e) => e.kind !== "clip");
const runtime = clips.reduce((n, e) => n + Math.max(0, e.end - e.start), 0);
const showAll = search.all === "1";
@@ -184,16 +195,24 @@ export default async function ReportProject({
sources and the build chain. */}
{clips.length > 0 && (
<div data-walk-cut="" className="flex flex-col items-start gap-1">
+ {/* Where the walk STARTS is where the walk goes: the first clip
+ that still needs an answer and can be played. A cut whose
+ first clip is already judged (or not fetched) would otherwise
+ open on a clip `n` immediately leaves. */}
<Link
- href={`/browse/${project.id}/clip/${clips[0].id}`}
- data-walk-start={clips[0].id}
+ href={`/browse/${project.id}/clip/${walkStart}`}
+ data-walk-start={walkStart}
className={buttonVariants({ variant: "primary", size: "lg" })}
>
- Walk the cut → start at {clips[0].id}
+ Walk the cut → start at {walkStart}
</Link>
<span className="text-[11px] text-[var(--color-dim)]" data-walk-progress="">
<kbd>p</kbd> / <kbd>n</kbd> move between clips · {review.reviewed} of {review.total}{" "}
reviewed{review.incorrect > 0 ? ` · ${review.incorrect} incorrect` : ""}
+ {" · "}
+ <span data-ready-count="">
+ ready {readiness.ready} of {readiness.needing} needing judgement
+ </span>
</span>
</div>
)}
@@ -412,6 +431,7 @@ export default async function ReportProject({
data-entry={e.id}
data-kind="clip"
data-cached={e.cached ? "1" : "0"}
+ data-fetched={e.fetched ? "1" : "0"}
data-mid-sentence={midSentence ? "1" : "0"}
className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] px-3 py-1.5"
>
@@ -423,11 +443,15 @@ export default async function ReportProject({
{e.lock && <Pill>locked</Pill>}
{!e.lock && e.lockStart && <Pill>start pinned</Pill>}
{!e.lock && e.lockEnd && <Pill>end pinned</Pill>}
- {e.cached ? (
+ {/* The PLAYER's question, and the walk's: is this clip
+ watchable end to end right now. `data-cached` keeps the
+ BUILD's -- the padded window it would fetch -- because
+ the build chain reads it and they are different. */}
+ {e.fetched ? (
<Pill>cached</Pill>
) : (
<Pill tone="open">
- not fetched
+ not fetched yet
</Pill>
)}
{e.segment && <Pill>segment built</Pill>}
diff --git a/umtool/docs/clip-bench.md b/umtool/docs/clip-bench.md
@@ -182,13 +182,37 @@ before the edit.
## Walking the cut
`p` and `n` (and the links in the header line) move to the previous and next
-clip, computed server-side from the timeline's own order. Reviewing a cut is
-watching nineteen clips in a row, and going back to the project page between each
-one is nineteen round trips to re-find your place. The project page starts the
-walk: one **"Walk the cut → start at `<first clip>`"** button above the rows,
+clip **on the walk**, computed server-side. Reviewing a cut is watching nineteen
+clips in a row, and going back to the project page between each one is nineteen
+round trips to re-find your place. The project page starts the walk: one
+**"Walk the cut → start at `<first clip on it>`"** button above the rows,
because the per-row links are the right control for "go to that one" and the
wrong one for "start".
+### What the walk skips
+
+The walk visits the clips that still **need judgement** and are **fetched**, in
+timeline order — so `p`, `n`, both header links, the start button and the jump
+<kbd>y</kbd> makes after it confirms all go to the nearest one of those.
+
+- **Needs judgement** means no verdict: the same rule `N reviewed` counts by.
+ Walking back onto a clip somebody already answered is the round trip the walk
+ exists to remove.
+- **Fetched** means some file in `out/clips-raw` holds this clip's own window
+ end to end. Not the *padded* window a build would download, and not an
+ overlap: a half-covered clip cannot be watched through, so there is nothing
+ to judge. Landing on one is a dead end whose only move is `n` again.
+
+A **URL still opens any clip**, judged or unfetched — the bench renders it, says
+*"nothing fetched for this clip yet"* and carries a **not fetched yet** pill,
+and its two links point at the fetched, unjudged neighbours either side. The
+per-row link on the project page is that URL, so "go to that one" is unchanged.
+
+Both pages read **`ready N of M needing judgement`**: `M` is the clips with no
+verdict, `N` the ones among them that can be judged today. The gap between the
+two is what is waiting on a download rather than on you — every row the project
+page marks **not fetched yet**.
+
### Is this accurate?
A clip is a **claim** — the report said somebody said this, here — so the right
@@ -196,13 +220,19 @@ column opens with what this clip is *supposed to be* (its `quote`, and its
`note`: why it is in the cut) and then asks the one question the walk exists to
answer.
-<kbd>y</kbd> confirms and **advances**, because in the yes case the next clip is
-what you want and a walk of sixty clips should be one finger. <kbd>x</kbd> puts
+<kbd>y</kbd> confirms and **advances** — to the next clip on the walk, which is
+the next one that needs an answer and has something to play — because in the yes
+case that clip is what you want and a walk of sixty clips should be one finger. <kbd>x</kbd> puts
the cursor in the `correction` box and marks it **required**, because the note
*is* the no answer — and saving it stays put, so it can be read back against the
clip. That save is ONE patch, `{correction, verdict: "incorrect"}`, so the
manifest never holds a complaint with no verdict.
+An edge you nudged but never saved is confirmed WITH the clip: if the selection
+differs from the stored window, <kbd>y</kbd> carries `start`/`end` (and clears a
+cut the new extent no longer contains) in the same patch as the verdict, so
+walking on never loses the window you just chose.
+
A note on a clip that is already **confirmed** is just a note — why its window
moved, a caveat for the writers — and writing one there leaves the verdict
alone. The state line reads *not yet reviewed*, *confirmed*, *confirmed · with
diff --git a/umtool/docs/report-video.md b/umtool/docs/report-video.md
@@ -205,7 +205,9 @@ to let somebody undo a note they wrote before the rule existed would be a trap.
`reviewOf()` is the one definition of coverage — the bench header's
`N reviewed`, the project page's "Walk the cut" button and `umtool corrections
-<project>` all read it, and the last opens with
+<project>` all read it, and `walkReadiness()` reads the same verdict rule to
+decide what the walk VISITS (unjudged and fetched — see
+[clip-bench.md](clip-bench.md#what-the-walk-skips)), and the last opens with
`N clips: A incorrect, B confirmed (C with a note), D not yet reviewed (ids: …)`
so the walk's coverage is visible beside the defects it found.
diff --git a/umtool/e2e/clip-bench.spec.ts b/umtool/e2e/clip-bench.spec.ts
@@ -36,11 +36,16 @@ const FIXTURE = path.join(UMTOOL, ".e2e-song");
// index and decision specs assert what report-fixture's windows are -- sharing
// one project made the suite pass or fail on which spec file ran first.
const PROJECT = "reports/bench-fixture";
+// The WALK's fixture, and read-only: see the block below, and the comment on
+// walk-fixture in make-fixture.mjs.
+const WALK = "reports/walk-fixture";
const MANIFEST = path.join(FIXTURE, "reports", "bench-fixture", "video.manifest.json");
const bench = (clip: string) => `/browse/${PROJECT}/clip/${clip}`;
+const walkBench = (clip: string) => `/browse/${WALK}/clip/${clip}`;
-const readClip = (id: string) => {
- const m = JSON.parse(readFileSync(MANIFEST, "utf8")) as {
+const readClipIn = (project: string, id: string) => {
+ const file = path.join(FIXTURE, ...project.split("/"), "video.manifest.json");
+ const m = JSON.parse(readFileSync(file, "utf8")) as {
timeline: {
id: string;
start: number;
@@ -56,6 +61,7 @@ const readClip = (id: string) => {
};
return m.timeline.find((e) => e.id === id)!;
};
+const readClip = (id: string) => readClipIn(PROJECT, id);
/**
* Type into an attribution field and let it save the way a blur does.
@@ -100,8 +106,12 @@ const keyboardLive = async (page: import("@playwright/test").Page) => {
await box.blur();
};
-const token = async (request: { get: (u: string) => Promise<{ json: () => Promise<unknown> }> }, clip: string) => {
- const r = await request.get(`/api/report/clip?project=${encodeURIComponent(PROJECT)}&clip=${clip}`);
+const token = async (
+ request: { get: (u: string) => Promise<{ json: () => Promise<unknown> }> },
+ clip: string,
+ project = PROJECT,
+) => {
+ const r = await request.get(`/api/report/clip?project=${encodeURIComponent(project)}&clip=${clip}`);
return (await r.json()) as { token: string };
};
@@ -337,23 +347,93 @@ test("a date that is not a real day is refused, and the bench says so", async ({
await expect(page.locator("[data-attrib-field=date]")).toHaveValue("2025-02-31");
});
-test("prev and next walk the cut, by link and by key", async ({ page }) => {
- await page.goto(bench("c02"));
+// ---------------------------------------------------------------------------
+// The walk, and the two things it skips.
+//
+// Reviewing a cut is watching every clip in order and answering one question
+// about each, so the walk goes to the clips that still NEED an answer and can
+// be WATCHED today. A clip with nothing cached is a dead end -- there is
+// nothing to judge and the only move is to press `n` again -- and one already
+// answered is the round trip the walk exists to remove.
+//
+// These run against reports/walk-fixture, which is read-only for exactly this
+// reason: every other test in this file writes verdicts into bench-fixture, so
+// "ready 2 of 3" would be true there only until one of them ran. Its four
+// clips are w01 (fetched, unjudged), w02 (in the gap between two cached files,
+// so nothing to play at all), w03 (fetched, already confirmed) and w04
+// (fetched, unjudged).
+// ---------------------------------------------------------------------------
+
+test("the walk visits only the clips that need judgement and are fetched", async ({ page }) => {
+ await page.goto(walkBench("w01"));
+ // w02 has nothing to play and w03 has already been answered, so the one move
+ // forward from w01 is w04 -- named in the link, not just arrived at.
+ await expect(page.locator("[data-clip-nav=next]")).toHaveText(/w04/);
await page.locator("[data-clip-nav=next]").click();
- await expect(page.locator("[data-bench=c03]")).toBeVisible();
+ await expect(page.locator("[data-bench=w04]")).toBeVisible();
- // The same move from the keyboard. Reviewing a cut is watching every clip in
- // order, and the project page in between is a round trip to re-find your place.
+ // The same move from the keyboard, backwards over the same two clips.
+ await keyboardLive(page);
await page.locator("body").press("p");
- await expect(page.locator("[data-bench=c02]")).toBeVisible();
+ await expect(page.locator("[data-bench=w01]")).toBeVisible();
- // The ends say so rather than offering a link into nothing.
- await page.goto(bench("c01"));
+ // The ends of the WALK, which are not the ends of the cut.
await expect(page.locator("[data-clip-nav=prev]")).toHaveCount(0);
- await page.goto(bench("c04"));
+ await page.goto(walkBench("w04"));
await expect(page.locator("[data-clip-nav=next]")).toHaveCount(0);
});
+test("a clip off the walk still opens by URL, and says why it is off it", async ({ page }) => {
+ // Nothing fetched: it renders, it says so, and its links point at the
+ // neighbours on either side rather than at nothing.
+ await page.goto(walkBench("w02"));
+ await expect(page.locator("[data-bench=w02]")).toBeVisible();
+ await expect(page.locator("[data-clip-unfetched]")).toBeVisible();
+ await expect(page.getByText("nothing fetched for this clip yet")).toBeVisible();
+ await expect(page.locator("[data-clip-nav=prev]")).toHaveText(/w01/);
+ await expect(page.locator("[data-clip-nav=next]")).toHaveText(/w04/);
+
+ // Already judged: it opens too, with what the walk said about it, and it is
+ // fetched -- so no pill.
+ await page.goto(walkBench("w03"));
+ await expect(page.locator("[data-bench=w03]")).toHaveAttribute("data-verdict", "confirmed");
+ await expect(page.locator("[data-clip-unfetched]")).toHaveCount(0);
+});
+
+test("`ready N of M` counts the fetched clips that still need judgement", async ({ page }) => {
+ await page.goto(`/browse/${WALK}`);
+ await expect(page.locator("[data-ready-count]")).toHaveText("ready 2 of 3 needing judgement");
+ // The pill is the same question the walk asks, per row.
+ await expect(page.locator("[data-entry=w02]")).toHaveAttribute("data-fetched", "0");
+ await expect(page.locator("[data-entry=w02]")).toContainText("not fetched yet");
+ await expect(page.locator("[data-entry=w01]")).toHaveAttribute("data-fetched", "1");
+ // And the walk starts where the walk actually goes.
+ await expect(page.locator("[data-walk-start=w01]")).toHaveAttribute(
+ "href",
+ `/browse/${WALK}/clip/w01`,
+ );
+
+ await page.goto(walkBench("w01"));
+ await expect(page.locator("[data-ready-count]")).toHaveText("ready 2 of 3 needing judgement");
+});
+
+test("`y` walks on OVER a clip that has already been judged", async ({ page, request }) => {
+ await page.goto(walkBench("w01"));
+ await keyboardLive(page);
+ await page.locator("body").press("y");
+ // Not w02 (nothing to play) and not w03 (answered in the fixture): w04.
+ await expect(page.locator("[data-bench=w04]")).toBeVisible();
+ await expect.poll(() => readClipIn(WALK, "w01").verdict).toBe("confirmed");
+
+ // Put it back. This is the ONE test that writes to walk-fixture, and the
+ // counter above is only stable because it does.
+ const { token: t } = await token(request, "w01", WALK);
+ await request.put("/api/report/window", {
+ data: { project: WALK, clip: "w01", verdict: "", token: t },
+ });
+ await expect.poll(() => readClipIn(WALK, "w01").verdict).toBeUndefined();
+});
+
test("the segment route serves a built clip with ranges, and 404s when there is none", async ({
page,
request,
@@ -509,13 +589,22 @@ test("the bench says what the clip is supposed to be", async ({ page }) => {
});
test("`y` confirms the clip and walks on; the manifest says so", async ({ page, request }) => {
+ // c04 is the only OTHER clip here the walk can visit -- c02 and c03 have
+ // nothing cached that holds them -- so the state it is in decides where this
+ // lands. Said out loud rather than inherited from whichever test ran before.
+ const { token: t0 } = await token(request, "c04");
+ await request.put("/api/report/window", {
+ data: { project: PROJECT, clip: "c04", verdict: "", token: t0 },
+ });
+
await page.goto(bench("c01"));
await keyboardLive(page);
await page.locator("body").press("y");
- // Confirming ADVANCES: in the yes case the next clip is what you want, and
- // a walk of sixty clips is one finger.
- await expect(page.locator("[data-bench=c02]")).toBeVisible();
+ // Confirming ADVANCES: in the yes case the next clip you can judge is what
+ // you want, and a walk of sixty clips is one finger. Over c02 and c03, which
+ // are not fetched, to c04.
+ await expect(page.locator("[data-bench=c04]")).toBeVisible();
await expect.poll(() => readClip("c01").verdict).toBe("confirmed");
await page.goto(bench("c01"));
@@ -530,6 +619,46 @@ test("`y` confirms the clip and walks on; the manifest says so", async ({ page,
await expect.poll(() => readClip("c01").verdict).toBeUndefined();
});
+test("`y` saves a window you nudged but never saved, in the same write", async ({
+ page,
+ request,
+}) => {
+ // Known ground on both clips the walk touches, so the only thing that moves
+ // an edge here is the key press below.
+ const { token: t0 } = await token(request, "c01");
+ await request.put("/api/report/window", {
+ data: { project: PROJECT, clip: "c01", start: 3, end: 6, verdict: "", token: t0 },
+ });
+ const { token: t1 } = await token(request, "c04");
+ await request.put("/api/report/window", {
+ data: { project: PROJECT, clip: "c04", verdict: "", token: t1 },
+ });
+
+ await page.goto(bench("c01"));
+ await keyboardLive(page);
+ // 6.00 -> 6.05, unsaved: `save window` is lit and nobody has pressed it.
+ await page.locator("body").press(".");
+ await expect(page.getByRole("button", { name: "save window" })).toBeEnabled();
+
+ await page.locator("body").press("y");
+
+ // ONE write carries both. Confirming a clip whose edge you just nudged is a
+ // judgement about THAT window, and the advance would otherwise walk away
+ // from it.
+ await expect.poll(() => readClip("c01").end).toBe(6.05);
+ expect(readClip("c01").start).toBe(3);
+ expect(readClip("c01").verdict).toBe("confirmed");
+ // And it still advances, over c02 and c03, which are not fetched.
+ await expect(page.locator("[data-bench=c04]")).toBeVisible();
+
+ // Put c01 back where the rest of this file found it.
+ const { token: t2 } = await token(request, "c01");
+ await request.put("/api/report/window", {
+ data: { project: PROJECT, clip: "c01", start: 3, end: 9, verdict: "", token: t2 },
+ });
+ await expect.poll(() => readClip("c01").verdict).toBeUndefined();
+});
+
test("`x` requires the note, writes both fields at once, and stays put", async ({ page }) => {
await page.goto(bench("c03"));
// The key IS the assertion: `x` answers "no" by putting the cursor where the
@@ -698,8 +827,14 @@ test("the playback speed is this browser's, and it survives a reload", async ({
.toBe(1.5);
});
-test("auto-audition plays the clip you walk onto", async ({ page }) => {
- // c03 -> c04, because c04 is the neighbour with cached material.
+test("auto-audition plays the clip you walk onto", async ({ page, request }) => {
+ // c03 -> c04: c03 itself has nothing cached (it is reached here by URL, which
+ // still works), and c04 is the next clip the walk can visit. Its verdict is
+ // set explicitly, because an answered clip is not one the walk goes to.
+ const { token: t } = await token(request, "c04");
+ await request.put("/api/report/window", {
+ data: { project: PROJECT, clip: "c04", verdict: "", token: t },
+ });
await page.goto(bench("c03"));
await page.locator("[data-auto-audition=off]").click();
await expect(page.locator("[data-auto-audition=on]")).toBeVisible();
diff --git a/umtool/e2e/fixtures/make-fixture.mjs b/umtool/e2e/fixtures/make-fixture.mjs
@@ -838,6 +838,39 @@ const BENCH = writeProject(
]),
);
+// A FOURTH copy, for the WALK -- and this one is READ-ONLY.
+//
+// The walk visits the clips that still need judgement and are fetched, so what
+// it skips has to be pinned down by the fixture rather than by whichever spec
+// ran last. bench-fixture cannot do it: every test in that file writes verdicts
+// into it, so "ready 2 of 3" would be true only until somebody pressed `y`.
+//
+// Four clips, all on vid1 (whose cues run 0-21, so none of them outruns its
+// transcript), locked so none of them raises a decision:
+//
+// w01 3.00- 6.00 fetched, unjudged -> on the walk
+// w02 9.20-11.20 NOT fetched -> skipped; nothing to play
+// w03 12.00-15.00 fetched, CONFIRMED -> skipped; already answered
+// w04 15.00-18.00 fetched, unjudged -> on the walk
+//
+// w02 sits in the GAP between two cached files rather than merely outrunning
+// one: the bench plays the best OVERLAPPING file it has, so a clip half-held by
+// a neighbour's download still shows a picture. That is a different state from
+// "nothing to play", and the walk skips both -- neither can be watched through.
+//
+// So the walk is w01 <-> w04, over two different reasons, and the counter reads
+// `ready 2 of 3`: three clips need judgement, two of them can have it today.
+const WALK = writeProject(
+ "walk-fixture",
+ manifest("walk-fixture", "The Walk Fixture", { siteOrigin: "https://archive.example" }, [
+ { type: "clip", id: "w01", video: "vid1", start: 3.0, end: 6.0, cite: 3, section: 0, lock: true, quote: "and because" },
+ { type: "clip", id: "w02", video: "vid1", start: 9.2, end: 11.2, cite: 9, section: 0, lock: true, quote: "another whole sentence" },
+ { type: "clip", id: "w03", video: "vid1", start: 12.0, end: 15.0, cite: 12, section: 0, lock: true, quote: "a fourth one", verdict: "confirmed" },
+ { type: "clip", id: "w04", video: "vid1", start: 15.0, end: 18.0, cite: 15, section: 0, lock: true, quote: "trailing off" },
+ ]),
+);
+
+
// -- STUB BINARIES, so a build is offline and deterministic --------------------
//
// The pipeline shells out to yt-dlp for the availability preflight and for every
@@ -1022,6 +1055,19 @@ for (const name of ["vid1_10.00-16.00.mp4", "vid1_14.50-18.50.mp4"]) {
path.join(BENCH, "out", "clips-raw", name),
);
}
+// walk-fixture's cache: a file holding w01, one holding w03 and one holding
+// w04, and NOTHING touching w02 -- 9.20-11.20 falls in the gap between
+// 0.00-9.00 and 11.50-15.50, so it has no picture at all rather than half of
+// one. The content is the same nine seconds of tone in all three; what is being
+// fixed here is which windows are held, not what is in them.
+mkdirSync(path.join(WALK, "out", "clips-raw"), { recursive: true });
+for (const name of ["vid1_0.00-9.00.mp4", "vid1_11.50-15.50.mp4", "vid1_14.50-18.50.mp4"]) {
+ copyFileSync(
+ path.join(REPORT, "out", "clips-raw", "vid1_0.00-9.00.mp4"),
+ path.join(WALK, "out", "clips-raw", name),
+ );
+}
+
copyFileSync(
path.join(REPORT, "out", "clips-raw", "vid1_0.00-9.00.mp4"),
path.join(reports, "no-origin-fixture", "out", "no-origin-fixture.mp4"),
@@ -1100,5 +1146,6 @@ console.log(` CHANNELS_DIR=${CHANNELS} (testchan/vid1 punctuated, vid2 not)`);
console.log(` projects: report-fixture (4 clips, 1 mid-sentence), no-origin-fixture,`);
console.log(` localhost-fixture, bike-fixture (sweep), find/ (shadowed),`);
console.log(` deep/nested/solo-fixture (collapse case), bench-fixture (writable),`);
+console.log(` walk-fixture (read-only: w01/w04 walkable, w02 unfetched, w03 judged),`);
console.log(` longform-fixture (cue gap, legacy .bak, ffmeta), longform-edit-fixture, dash-fixture`);
console.log(` ${taken} candidate files copied, 2 mix tracks synthesised`);
diff --git a/umtool/lib/projects/report.mjs b/umtool/lib/projects/report.mjs
@@ -7,7 +7,8 @@
// and neither parses JSON.
import { readdir, readFile, stat } from "node:fs/promises";
import path from "node:path";
-import { DEFAULT_VARIANT, cachedWindowsFor, findContainingWindow } from "umtool-report-to-video/build-video";
+import { DEFAULT_VARIANT, cachedWindowsFor } from "umtool-report-to-video/build-video";
+import { rawCacheOf } from "../report/raw-cache.mjs";
import { channelName, cleanTitle } from "umtool-report-to-video/attribution";
/**
@@ -184,6 +185,37 @@ export function reviewOf(m) {
}
/**
+ * What the WALK will actually visit, and how much of it can be visited today.
+ *
+ * The walk is somebody at a desk answering one question per clip, so it goes to
+ * the clips that still need an answer AND can be watched end to end right now:
+ *
+ * needing -> nobody has judged it (`clipVerdict` is "unreviewed"). The same
+ * rule reviewOf() counts by, so "12 of 19 reviewed" and "ready 3
+ * of 7" cannot disagree about the same clip.
+ * ready -> needing AND `fetched`, which readClipDetail computes from the
+ * clips-raw cache: a file holding the clip's own window.
+ *
+ * Walking onto a clip with nothing to play is a dead end -- there is nothing to
+ * judge and the only move is to press `n` again -- and walking back onto one
+ * already answered is the same round trip a walk exists to remove. `readyIds`
+ * is in TIMELINE ORDER, because the cut's order is the order you watch it in.
+ *
+ * @param {{id: string, kind?: string, fetched?: boolean}[]} entries readClipDetail's entries
+ */
+export function walkReadiness(entries) {
+ const clips = (entries ?? []).filter((e) => (e.kind ?? e.type) === "clip");
+ const needing = clips.filter((e) => clipVerdict(e) === "unreviewed");
+ const ready = needing.filter((e) => !!e.fetched);
+ return {
+ needing: needing.length,
+ ready: ready.length,
+ needingIds: needing.map((e) => e.id),
+ readyIds: ready.map((e) => e.id),
+ };
+}
+
+/**
* The OTHER clips in the cut that come from this same recording.
*
* "Does this clip need more context, or is the context already coming up as
@@ -972,7 +1004,10 @@ export async function readClipDetail(dir, { manifest = null } = {}) {
const shadowExists = await hasShadowChannels(dir);
const channelsDir = channelsDirFor(dir, m, { shadowExists });
const build = await buildStateOf(dir, m);
- const rawDir = path.join(dir, "out", "clips-raw");
+ // ONE listing of out/clips-raw for the whole project. This used to be two
+ // readdirs PER CLIP -- the padded lookup and the full list -- so a forty-clip
+ // cut paid eighty directory reads to draw one page.
+ const raw = await rawCacheOf(dir);
const segDirs = segmentDirs(path.join(dir, "out"));
const segLists = await Promise.all(segDirs.map((d) => readdir(d).catch(() => [])));
const segWhich = segLists.findIndex((l) => l.length);
@@ -1010,8 +1045,8 @@ export async function readClipDetail(dir, { manifest = null } = {}) {
const pad = m.render?.fetchPad ?? 3.0;
const from = Math.max(0, e.start - pad);
const to = e.end + pad;
- const cached = await findContainingWindow(rawDir, e.video, from, to);
- const allWindows = await cachedWindowsFor(rawDir, e.video);
+ const cached = raw.containing(e.video, from, to);
+ const allWindows = raw.windows(e.video);
// The bench wants the WIDEST containing file (room to drag); the build wants
// the tightest (least to decode). They are different questions.
const widest = allWindows
@@ -1067,6 +1102,10 @@ export async function readClipDetail(dir, { manifest = null } = {}) {
noPunctuation,
proposed,
cached: cached ? { name: cached.name, from: cached.from, to: cached.to } : null,
+ // `cached` is the BUILD's question (is the padded window on disk); this is
+ // the PLAYER's (can this clip be watched end to end right now), and it is
+ // what the walk skips on. Same predicate, different window.
+ fetched: raw.isFetched(e),
widest: widest ? { name: widest.name, from: widest.from, to: widest.to } : null,
segment: segNames.has(`${e.id}.mp4`) ? path.posix.join(segRel, `${e.id}.mp4`) : null,
wantFrom: from,
diff --git a/umtool/lib/report/raw-cache.mjs b/umtool/lib/report/raw-cache.mjs
@@ -0,0 +1,48 @@
+// ONE read of out/clips-raw, answering for every clip in a project.
+//
+// Its own module, and deliberately a LEAF: lib/projects/report.mjs reads it, and
+// so does lib/report/serve.mjs, which report.mjs is itself imported by. Putting
+// it in serve.mjs closed that circle and broke module initialisation outright
+// (kinds.mjs reads a const of report.mjs at top level, and in the cycle that
+// const is still in its temporal dead zone). Nothing here imports anything of
+// ours but the predicate.
+import path from "node:path";
+import {
+ listRawNames,
+ tightestContaining,
+ windowsFromNames,
+} from "umtool-report-to-video/build-video";
+
+/**
+ * The cache. The directory is keyed by VIDEO and a page asks about it per clip
+ * -- the bench page did two readdirs per clip, and a forty-clip cut paid eighty
+ * of them for one screen. The listing is read once, the per-video parse is
+ * memoised, and every question below is then arithmetic.
+ *
+ * `isFetched(clip)` is the walk's question and the one a follow-up importer
+ * wants: is there a cached file holding this clip's OWN window, end to end.
+ * Not the PADDED window the build would fetch (that is `containing()` with the
+ * pad applied, and it calls three of four fixture clips unfetched), and not
+ * mere overlap -- a half-covered clip cannot be watched through, so it is not
+ * ready to judge.
+ */
+export async function rawCacheOf(projectDir) {
+ const rawDir = path.join(projectDir, "out", "clips-raw");
+ const names = await listRawNames(rawDir);
+ const byVideo = new Map();
+ const windows = (video) => {
+ if (!byVideo.has(video)) byVideo.set(video, windowsFromNames(names, rawDir, video));
+ return byVideo.get(video);
+ };
+ return {
+ rawDir,
+ windows,
+ containing: (video, from, to) => tightestContaining(windows(video), from, to),
+ isFetched: (clip) => {
+ const { start, end, video } = clip ?? {};
+ if (!Number.isFinite(start) || !Number.isFinite(end)) return false;
+ return !!tightestContaining(windows(video), start, end);
+ },
+ };
+}
+
diff --git a/umtool/lib/report/serve.mjs b/umtool/lib/report/serve.mjs
@@ -9,7 +9,8 @@ import path from "node:path";
import { stat } from "node:fs/promises";
import { REPORTS_ROOT, resolveInRoots } from "../paths.mjs";
import { walkProjects } from "../projects/walk.mjs";
-import { DEFAULT_VARIANT, cachedWindowsFor } from "umtool-report-to-video/build-video";
+import { DEFAULT_VARIANT, WIN_EPS } from "umtool-report-to-video/build-video";
+import { rawCacheOf } from "./raw-cache.mjs";
import { clipsOf, readManifest } from "../projects/report.mjs";
export async function resolveClip(projectId, clipId) {
@@ -23,8 +24,8 @@ export async function resolveClip(projectId, clipId) {
return { project, manifest, clip };
}
-/** build-video's own tolerance for "this file holds that window". */
-const WIN_EPS = 0.02;
+/** The clips-raw cache, re-exported so `serve.mjs` stays the bench's one door. */
+export { rawCacheOf } from "./raw-cache.mjs";
/**
* The cached source windows for a clip: the ones that hold ITS window, widest
@@ -42,10 +43,17 @@ const WIN_EPS = 0.02;
* into a 44-second file, i.e. 1146 s past its end, and plays nothing or the
* wrong seconds. A caller with no window (a ledger claim asking for the raw
* files of its video) still gets every file, widest first.
+ *
+ * @param {{dir: string}} project
+ * @param {{video: string, start?: number, end?: number}} clip
+ * @param {Awaited<ReturnType<typeof rawCacheOf>> | null} [cache] a clips-raw
+ * listing already read this request; one is built when none is passed.
*/
-export async function windowsFor(project, clip) {
- const rawDir = path.join(project.dir, "out", "clips-raw");
- const all = await cachedWindowsFor(rawDir, clip.video);
+export async function windowsFor(project, clip, cache = null) {
+ // The cache is optional and passed in by callers that already have one (the
+ // bench page asks for every clip), so the directory is read once per request.
+ const c = cache ?? (await rawCacheOf(project.dir));
+ const all = c.windows(clip.video);
const width = (w) => w.to - w.from;
const { start, end } = clip;
if (!Number.isFinite(start) || !Number.isFinite(end)) {
diff --git a/umtool/report-to-video/build-video.mjs b/umtool/report-to-video/build-video.mjs
@@ -288,15 +288,26 @@ const WINDOW_RE = /^(\d+(?:\.\d+)?)-(\d+(?:\.\d+)?)$/;
// A window read back from a 2 dp manifest can sit a hair outside the file that
// produced it; the same tolerance resolve-windows.mjs uses for the same reason.
-const WIN_EPS = 0.02;
+export const WIN_EPS = 0.02;
-export async function cachedWindowsFor(rawDir, video) {
- let names;
+/**
+ * The question is asked FOUR times per page -- the build, the bench, the
+ * project page's pill and the walk -- so it is one predicate in one place,
+ * split into the I/O and the arithmetic. A caller with a directory listing
+ * already in hand (the bench reads clips-raw ONCE per request and answers for
+ * every clip) uses the pure halves; the two original functions are those halves
+ * composed, and behave exactly as they did.
+ */
+export async function listRawNames(rawDir) {
try {
- names = await readdir(rawDir);
+ return await readdir(rawDir);
} catch {
return [];
}
+}
+
+/** The windows THIS video's files hold, parsed out of a directory listing. */
+export function windowsFromNames(names, rawDir, video) {
const prefix = `${video}_`;
const out = [];
for (const name of names) {
@@ -310,17 +321,30 @@ export async function cachedWindowsFor(rawDir, video) {
return out;
}
-/** The tightest cached file containing [from, to], or null. */
-export async function findContainingWindow(rawDir, video, from, to) {
- const windows = await cachedWindowsFor(rawDir, video);
+/** Does this cached file hold [from, to] whole, to the manifest's tolerance? */
+export function windowContains(w, from, to) {
+ return !(w.from > from + WIN_EPS || w.to < to - WIN_EPS);
+}
+
+/** The tightest of `windows` containing [from, to], or null. */
+export function tightestContaining(windows, from, to) {
let best = null;
for (const w of windows) {
- if (w.from > from + WIN_EPS || w.to < to - WIN_EPS) continue;
+ if (!windowContains(w, from, to)) continue;
if (!best || w.to - w.from < best.to - best.from) best = w;
}
return best;
}
+export async function cachedWindowsFor(rawDir, video) {
+ return windowsFromNames(await listRawNames(rawDir), rawDir, video);
+}
+
+/** The tightest cached file containing [from, to], or null. */
+export async function findContainingWindow(rawDir, video, from, to) {
+ return tightestContaining(await cachedWindowsFor(rawDir, video), from, to);
+}
+
async function fetchClip(entry, meta, render, rawDir, opts) {
// Deliberately over-fetch: the snapping pass below needs room on both sides to
// find a silence, and a clip that has no slack can only be cut where the cue