commit d6afa774ba81b366cc9f5cdb3e2b27f68789a121
parent 5aaeec9382a2492196141dbdbe2a4948a212f663
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 9 Oct 2026 12:01:32 -0400
ops: archival writes — settings, publish lane, platform holds, workers, one video, cleanup, relocate dryRun; archilyzer storage report
Release 19 slice A4. Each route is the existing action:
- POST /api/ops/settings {patch}: saveSettings, refusing an unknown key, a
key another writer owns (channelPriority, autoQueue, workers, storage) and a
value the schema would coerce (settings/patch.ts compares every leaf sent
with the schema's reading of the merged settings — the schema clamps rather
than throwing, so { ok: true } would otherwise answer an unsaved value).
- lane: the publish lane (savePublishSettingsAction with the stored fields,
the publish hold, start/stop/drainPublishLaneAction); the four ingest lanes'
drain now goes through drainAutoQueueAction.
- clear-platform-hold {platform}: clearPlatformHoldAction.
- POST /api/ops/workers {op, ids}: enable/disable/drainWorkerAction per id.
- transcribe-one {slug, id, file?}: transcribeOneAction / whisperVideoAction.
- delete-file {slug, id, file}: deleteVideoFileAction (removeMediaFile).
- do-not-clean {slug, id, keep?}: toggleDoNotCleanAction, set not toggled.
- POST /api/ops/cleanup {slug, sweep}: the channel page's four sweeps.
- relocate {dryRun}: previewRelocationAction per slug (a missing channel is
the bulk path's "channel not found").
`archilyzer storage report [<slug>...] [--json]` (common/bin/storage-report.ts):
each channel's tierable media, text and clip bytes off its report and where
its media is (corpus disk, a location, a custom path, legacy), with totals per
place; unknown is never 0; no drive is touched.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
17 files changed, 922 insertions(+), 37 deletions(-)
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -507,6 +507,18 @@ export const COMMANDS: Command[] = [
"--channel <slug> list duplicate and missing transcripts"),
script(["migrate", "channel-priority"], "migrate-channel-priority.ts",
"[--dry-run] the one-shot channel-priority migration (plans/channel-priority.md, S5)"),
+ {
+ path: ["storage", "report"],
+ usage:
+ "[<slug>…] [--json] each channel's media (tierable), text and clip bytes and where its media is — off the reports, no drive touched; totals per place (the corpus disk's is what a move would free)",
+ flags: { json: "boolean" },
+ maxPositionals: 1000,
+ run: async ({ positionals, flags }) =>
+ (await import("./storage-report")).main({
+ slugs: positionals,
+ json: flags.json === true,
+ }),
+ },
script(["storage", "migrate-tier"], "migrate-media-tier.ts",
"<slug>…|--all [--order smallest] [--include-large] [--dry-run] [--reclaim] bring a channel off the retired whole-directory layout onto the media tier, its text home to the corpus disk (editor stopped; --all stops before the three big-text channels)"),
{
diff --git a/common/bin/storage-report.test.ts b/common/bin/storage-report.test.ts
@@ -0,0 +1,93 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import type { ChannelBrief } from "../controller/channels";
+import type { ChannelSnapshot } from "../controller/channelSnapshot";
+import type { StorageLocation } from "../lib/storageLocations";
+import {
+ buildStorageReport,
+ renderStorageReport,
+ storageReportRow,
+} from "./storage-report";
+
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test bin/storage-report.test.ts
+//
+// `archilyzer storage report` is a pure fold over the channel briefs: where
+// each channel's media is (corpus, a location, a custom path, legacy), its
+// three tiers' bytes off the report — unknown when the report has no figure,
+// never 0 — and the totals per place.
+
+const LOCATIONS = [
+ { id: "platter", label: "Platter", root: "/mnt/platter" },
+ { id: "archive", label: "Archive", root: "/mnt/platter/archive" },
+] as unknown as StorageLocation[];
+
+function brief(
+ slug: string,
+ config: Record<string, unknown>,
+ snap: Partial<ChannelSnapshot> | null,
+): ChannelBrief {
+ return {
+ slug,
+ config: { handling: "youtube", name: slug, ...config },
+ snapshot: snap ? ({ generatedAt: "2026-10-01T00:00:00.000Z", ...snap } as ChannelSnapshot) : null,
+ } as ChannelBrief;
+}
+
+const briefs = [
+ brief("in-place", {}, { totalMediaBytes: 500, totalTextBytes: 20, totalClipsBytes: 5 }),
+ brief("moved", { mediaDir: "/mnt/platter/archive/moved/media" }, { totalMediaBytes: 900, totalTextBytes: 10 }),
+ brief("custom", { mediaDir: "/srv/elsewhere/custom/media" }, { totalMediaBytes: 100 }),
+ brief("legacy", { dataDir: "/mnt/platter/legacy/data" }, { totalTextBytes: 1 }),
+ brief("old-report", {}, null),
+ brief("posts", { sourceKind: "social", handling: "transcribe", platform: "twitter", postFetcher: "x-gallery-dl" }, {
+ totalMediaBytes: 0,
+ }),
+];
+
+test("a channel's place: corpus, the longest location, a custom path, legacy", () => {
+ const rows = Object.fromEntries(briefs.map((b) => [b.slug, storageReportRow(b, LOCATIONS)]));
+ assert.equal(rows["in-place"].place, "corpus");
+ assert.equal(rows["in-place"].location, null);
+ assert.equal(rows.moved.place, "location");
+ assert.equal(rows.moved.location, "archive");
+ assert.equal(rows.custom.place, "custom");
+ assert.equal(rows.custom.mediaDir, "/srv/elsewhere/custom/media");
+ assert.equal(rows.legacy.place, "legacy");
+ assert.equal(rows.legacy.location, "platter");
+});
+
+test("a figure the report does not have is unknown, never 0", () => {
+ const row = storageReportRow(briefs[4], LOCATIONS);
+ assert.deepEqual(
+ [row.mediaBytes, row.textBytes, row.clipsBytes, row.reportedAt],
+ [null, null, null, null],
+ );
+ assert.equal(storageReportRow(briefs[2], LOCATIONS).textBytes, null);
+});
+
+test("the report leaves out social channels, sorts by media, and totals per place", () => {
+ const report = buildStorageReport(briefs, LOCATIONS);
+ assert.deepEqual(
+ report.rows.map((r) => r.slug),
+ ["moved", "in-place", "custom", "legacy", "old-report"],
+ );
+ assert.deepEqual(report.totals, {
+ "location:archive": { channels: 1, mediaBytes: 900, unknown: 0 },
+ corpus: { channels: 2, mediaBytes: 500, unknown: 1 },
+ custom: { channels: 1, mediaBytes: 100, unknown: 0 },
+ legacy: { channels: 1, mediaBytes: 0, unknown: 1 },
+ });
+ // Narrowed to named channels.
+ assert.deepEqual(
+ buildStorageReport(briefs, LOCATIONS, ["custom"]).rows.map((r) => r.slug),
+ ["custom"],
+ );
+});
+
+test("the table says unknown as a dash and names what a move would free", () => {
+ const text = renderStorageReport(buildStorageReport(briefs, LOCATIONS));
+ assert.match(text, /^channel\s+media \(tierable\)\s+text\s+clips\s+where$/m);
+ assert.match(text, /^old-report\s+—\s+—\s+—\s+corpus disk$/m);
+ assert.match(text, /^moved\s.*location archive$/m);
+ assert.match(text, /on the corpus disk \(a move would free\): 2 channel\(s\), .* media, 1 unknown/);
+});
diff --git a/common/bin/storage-report.ts b/common/bin/storage-report.ts
@@ -0,0 +1,158 @@
+// `archilyzer storage report [<slug>…] [--json]` — WHERE EACH CHANNEL'S BYTES
+// ARE, by tier, off its report (release 19, A4).
+//
+// The tally an operator used to do by hand before freeing the corpus disk: per
+// channel, the MEDIA tier's bytes (what `lib/mediaTier.ts`'s `isTierable`
+// names — the audio and the raw live-chat replay; the bytes a media move
+// carries), the text, and the clip-window cache, and WHERE the media is: on the
+// corpus disk (in place: what a move would free), on a storage location
+// (`config.mediaDir` under its root — a channel is never tagged, it is on L iff
+// that is true), at a path no location names, or `legacy` (the retired
+// whole-`data/` move, held by every guard until `storage migrate-tier`).
+//
+// OFF THE REPORTS, NEVER A WALK — the /storage page's rule. One config and one
+// snapshot per channel; no drive is touched, so an unmounted platter cannot
+// stall it. A channel whose report predates a figure, or whose media drive was
+// not answering when the report was written, has that figure UNKNOWN, never 0
+// (`—` in the table, `null` in --json), and "Refresh report" is the fix.
+// Social channels hold no media and are left out.
+//
+// Needs no editor. Writes nothing.
+
+import { getPaths, type Paths } from "../lib/paths";
+import { getSettings } from "../lib/settings";
+import { formatBytes } from "../lib/format";
+import { isSocialChannel } from "../lib/channelConfig";
+import { locationOfDataDir, type StorageLocation } from "../lib/storageLocations";
+import { listChannelBriefs, type ChannelBrief } from "../controller/channels";
+
+export type StoragePlace = "corpus" | "location" | "custom" | "legacy";
+
+export type StorageReportRow = {
+ slug: string;
+ place: StoragePlace;
+ // The location's id when `place` is "location" (or a legacy channel's, when
+ // its retired dataDir sits under one).
+ location: string | null;
+ // config.mediaDir (or a legacy channel's dataDir): where its media is.
+ mediaDir: string | null;
+ mediaBytes: number | null;
+ textBytes: number | null;
+ clipsBytes: number | null;
+ reportedAt: string | null;
+};
+
+export type StorageReport = {
+ rows: StorageReportRow[];
+ // Per place ("corpus", "location:<id>", "custom", "legacy"): channels, the
+ // media bytes known, and how many channels' media bytes are unknown.
+ totals: Record<string, { channels: number; mediaBytes: number; unknown: number }>;
+};
+
+export function storageReportRow(
+ brief: ChannelBrief,
+ locations: StorageLocation[],
+): StorageReportRow {
+ const { config, snapshot } = brief;
+ const legacyDir = config.dataDir?.trim() || "";
+ const mediaDir = config.mediaDir?.trim() || "";
+ const dir = legacyDir || mediaDir;
+ const loc = dir ? locationOfDataDir(dir, locations) : null;
+ const place: StoragePlace = legacyDir
+ ? "legacy"
+ : !mediaDir
+ ? "corpus"
+ : loc
+ ? "location"
+ : "custom";
+ const n = (v: number | undefined) => (typeof v === "number" ? v : null);
+ return {
+ slug: brief.slug,
+ place,
+ location: loc?.id ?? null,
+ mediaDir: dir || null,
+ mediaBytes: n(snapshot?.totalMediaBytes),
+ textBytes: n(snapshot?.totalTextBytes),
+ clipsBytes: n(snapshot?.totalClipsBytes),
+ reportedAt: snapshot?.generatedAt ?? null,
+ };
+}
+
+export function placeKey(row: StorageReportRow): string {
+ return row.place === "location" ? `location:${row.location}` : row.place;
+}
+
+export function buildStorageReport(
+ briefs: ChannelBrief[],
+ locations: StorageLocation[],
+ only: readonly string[] = [],
+): StorageReport {
+ const wanted = new Set(only);
+ const rows = briefs
+ .filter((b) => !isSocialChannel(b.config))
+ .filter((b) => wanted.size === 0 || wanted.has(b.slug))
+ .map((b) => storageReportRow(b, locations))
+ // Biggest media first; unknowns last, by slug.
+ .sort(
+ (a, b) =>
+ (b.mediaBytes ?? -1) - (a.mediaBytes ?? -1) || a.slug.localeCompare(b.slug),
+ );
+ const totals: StorageReport["totals"] = {};
+ for (const r of rows) {
+ const t = (totals[placeKey(r)] ??= { channels: 0, mediaBytes: 0, unknown: 0 });
+ t.channels++;
+ if (r.mediaBytes === null) t.unknown++;
+ else t.mediaBytes += r.mediaBytes;
+ }
+ return { rows, totals };
+}
+
+const size = (v: number | null) => (v === null ? "—" : formatBytes(v));
+
+export function renderStorageReport(report: StorageReport): string {
+ const head = ["channel", "media (tierable)", "text", "clips", "where"];
+ const body = report.rows.map((r) => [
+ r.slug,
+ size(r.mediaBytes),
+ size(r.textBytes),
+ size(r.clipsBytes),
+ r.place === "location"
+ ? `location ${r.location}`
+ : r.place === "corpus"
+ ? "corpus disk"
+ : `${r.place} ${r.mediaDir ?? ""}`.trim(),
+ ]);
+ const widths = head.map((h, i) => Math.max(h.length, ...body.map((row) => row[i].length)));
+ const line = (cells: string[]) =>
+ cells
+ .map((c, i) => (i >= 1 && i <= 3 ? c.padStart(widths[i]) : c.padEnd(widths[i])))
+ .join(" ")
+ .trimEnd();
+ const totals = Object.entries(report.totals).map(
+ ([place, t]) =>
+ ` ${place === "corpus" ? "on the corpus disk (a move would free)" : place}: ` +
+ `${t.channels} channel(s), ${formatBytes(t.mediaBytes)} media` +
+ (t.unknown ? `, ${t.unknown} unknown (refresh their reports)` : ""),
+ );
+ return [line(head), ...body.map(line), "", "Media by place:", ...totals].join("\n");
+}
+
+export async function main(opts: {
+ slugs: string[];
+ json: boolean;
+ paths?: Paths;
+ out?: (s: string) => void;
+}): Promise<number> {
+ const paths = opts.paths ?? getPaths();
+ const out = opts.out ?? ((s: string) => console.log(s));
+ const briefs = await listChannelBriefs(paths);
+ const known = new Set(briefs.map((b) => b.slug));
+ const missing = opts.slugs.filter((s) => !known.has(s));
+ if (missing.length) {
+ console.error(`storage report: no channel ${missing.join(", ")}`);
+ return 2;
+ }
+ const report = buildStorageReport(briefs, getSettings().storage.locations, opts.slugs);
+ out(opts.json ? JSON.stringify(report, null, 2) : renderStorageReport(report));
+ return 0;
+}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -4,6 +4,8 @@
- **`pnpm ops` finds the editor's token itself.** When `WORKER_TOKEN` or `ARCHILYZER_EDITOR_URL` is not set, it reads them from `editor/.env.local` and `editor/.env` of the checkout the script is in — wherever it is run from — and, in a linked worktree, which has no `editor/.env`, from the main worktree's; nothing else is taken from those files, and a variable already set wins. A 401 or 503 now says where the token came from (never what it is). `--wait` asks `GET /api/ops/job/<id>` (behind the token) whether a job it can no longer follow is still there, instead of the UI's live-jobs view; a job the editor has forgotten since (100 later jobs, or a restart) is reported with the status it ended with — it used to print `archived` and exit 1 for a job that finished `done` — and while a job waits, `--wait` prints its place in the queue whenever it changes. `GET /api/ops/job/<id>[?tail=N]` answers one job: its kind, channel, status, times and exit code, from its record or its sidecar; where it waits (`queue: {key, position, queued, head}`); and with `tail` its log's last N lines. Needs a restart of the editor.
- **The /jobs list and its buttons over `pnpm ops`.** `pnpm ops get jobs` lists jobs newest first — `--active` every queued and running job in queue order with its place, `--failed` the failed ones, `--kind` and `--slug` to narrow either, `--limit` (50, at most 500); a filtered list looks through the newest 2000 jobs. `pnpm ops get job <id> --tail [N]` is one job with the last N lines of its log. `pnpm ops job cancel|drain|promote|force-release|retry <id>…` presses that row's button for each id (`POST /api/ops/job {"verb", "ids"}`) — the same server action, so the same refusals — and answers each id in `results`; an id that could not be acted on (unknown, or a promote of a job already at the front) makes the answer `ok: false`, named, without stopping the others. `job retry-failed` is the page's **Retry all**, and `--wait` follows the jobs a retry started; `job wait <id>…` follows jobs already running, printing each one's place in its queue while it waits. Retry all now returns the new jobs' ids beside its count. Needs a restart of the editor.
- **What the editor's pages show, over `pnpm ops`.** `get settings [<key>]` is settings.json as the editor reads it — migrated and defaulted — or one top-level key of it; `get storage` the /storage page (each location, mounted or not, its free space and tiers); `get sites` every site's id, title, public URL, Pages project, audience, whether it is listed, publish policy and channels; `get workers`, `get auto-queue` and `get scheduler` the payloads /workers, the four lanes and /operations/sync poll; `get cleanup <slug>` one channel's /cleanup row — what each sweep would reclaim, what holds the rest, the failed-transcriptions count. All behind the token, and no secret leaves through any of them: a remote worker's token reads `<redacted>`. Needs a restart of the editor.
+- **Settings, lanes, workers, single videos and cleanup over `pnpm ops`.** `pnpm ops settings {"patch": {…}}` writes settings.json through the editor's one writer (a block's keys merge one level): an unknown key is refused, as is a key another writer owns (`channelPriority`, `autoQueue`, `workers`, `storage` — the refusal names the command or page), and so is a value the schema would not keep as sent — a clamped number, a dropped choice — named with what would have been saved; the answer is the saved values. `lane` takes the publish lane too (its switch, its hold, and Start, Drain and Stop, as on /operations/publish). `clear-platform-hold {"platform"}` is the lane strip's Clear hold. `workers {"op": "enable" | "disable" | "drain", "ids"}` is /workers' switch and Drain. `transcribe-one {"slug", "id", "file"?}` transcribes one video — a named audio file, or the video page's Transcribe, which fetches the audio first when there is none; `delete-file {"slug", "id", "file"}` is the Files list's delete (a tiered file goes with its bytes on the media drive, and nothing is deleted while that drive is not answering); `do-not-clean {"slug", "id", "keep"?}` sets the marker. `cleanup {"slug", "sweep"}` runs one of the channel's cleanup sweeps as a job (`transcribed`, `extra-formats`, `wrong-format`, `failed-transcriptions`). `relocate` takes `"dryRun": true`: each channel's preview — bytes to copy, free space on both sides — and no move (a channel never tiered is tiered in place first, as the Storage panel's preview does). Needs a restart of the editor.
+- **`archilyzer storage report [<slug>…] [--json]`** lists each channel's media (the tierable files), text and clip-window bytes and where its media is — the corpus disk, a storage location, another path, or the retired layout — with totals per place; the corpus disk's total is what moving media would free. It reads the channels' reports and touches no drive, so an unmounted one cannot stall it; a figure a report does not have is shown as unknown, never 0. Needs no editor.
- **`/api/auto-queue/control` needs the ops token.** It started, stopped and drained lanes for anything that could reach the editor, a form posted from another page in the operator's own browser included. A script now sends `Authorization: Bearer $WORKER_TOKEN` (or uses `pnpm ops lane {"lane", "action"}`, the same three verbs); the Start, Drain and Stop buttons on /operations call server actions and are unchanged to use. Needs a restart of the editor.
- **A curated tag can exist on some sites only.** A tag's new **Sites** field on /tags (`sites` in `transcripts/tags.json`; `pnpm ops tags` takes it in a define) names the sites it exists on. Its rules then fire, and its pins apply, only to videos on those sites' channels, and every other site drops it from its records, its counts and its `/tags.json` — where **Hidden** only hid the chip. Empty is every site, as before. Setting it, or changing the channels of those sites, re-derives the corpus's tags once at the next index update. The Eva tags are what this is for: they belong on Anilyzer alone.
- **The publish lane.** Publishing can run itself: turn it on at **/operations/publish** (the runner's Start, Drain and Stop, the hold, and the lane's settings; or `publish.enabled` in settings) and the lane checks every `checkEveryMinutes` (10) whether the index is stale; when it is — and its last update is at least `refreshEveryMinutes` (360) old — it updates it, then builds every site whose channels changed or whose data the new index moved, one stage at a time on the `publish` queue. What it may do with a site is the site's own — the **Publish policy** on the site's settings form, `site.json` `publish.auto` —: `off` (the default: left alone), `build`, `preview` (built and deployed to the preview branch `publish.previewBranch`) or `production`; the hub and the homepage have `publish.hub` and `publish.homepage`. A private site is only ever built, and a site needs its Cloudflare Pages project before it may deploy. Hold the lane and the stage running finishes and no next one starts; quiet hours (`publish.quietHours`) do the same; Drain finishes the stage and ends the runner. The lane never forces a stage: a stage that finds its target current does nothing. On /jobs every stage of one run reads `run <id> · <target>`, and a stage still queued when the editor restarts is cancelled, never re-queued — the lane works out again what is stale from what is on disk. `archilyzer publish now` runs the same plan from the command line, one stage after another in its own process.
diff --git a/editor/app/api/ops/_writes.test.ts b/editor/app/api/ops/_writes.test.ts
@@ -0,0 +1,168 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, readFile, rm } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { callPost, setupOpsCorpus } from "./_testCorpus";
+import { coercedLeaves, settingsPatchProblem } from "./settings/patch";
+
+// Run with:
+// pnpm -C editor exec tsx --test "app/api/ops/_writes.test.ts"
+//
+// The archival writes (release 19, A4), against a temp corpus: the settings
+// patch end to end (its write needs no revalidation), and every other route's
+// refusals — the ones answered before an action revalidates a page, which only
+// a running server can do. Nothing here starts a job.
+
+const corpus = await setupOpsCorpus("one-youtube-channel-with-data");
+type Post = Parameters<typeof callPost>[0];
+const route = async (name: string) => (await import(`./${name}/route`)).POST as Post;
+const SLUG = "test-youtube";
+const VIDEO = "20240101_test1234567";
+const elsewhere = await mkdtemp(path.join(os.tmpdir(), "ops-relocate-root-"));
+test.after(async () => {
+ await corpus.cleanup();
+ await rm(elsewhere, { recursive: true, force: true });
+});
+
+const readSettingsFile = async () =>
+ JSON.parse(await readFile(corpus.settingsFile, "utf8")) as Record<string, unknown>;
+
+test("settings: an unknown key, an owned key and a coerced value are refused, and nothing is written", async () => {
+ const settings = await route("settings");
+ const before = await readSettingsFile();
+
+ const unknown = await callPost(settings, { patch: { minFreeDiskGb: 5 } });
+ assert.equal(unknown.status, 400);
+ assert.match(String(unknown.body.error), /unknown settings key\(s\): minFreeDiskGb/);
+
+ const owned = await callPost(settings, { patch: { channelPriority: {} } });
+ assert.equal(owned.status, 400);
+ assert.match(String(owned.body.error), /"channelPriority" is not patched here: .*pnpm ops channel-priority/);
+
+ // The schema clamps rather than throwing: a value it would not keep as sent
+ // is named with what it would have saved.
+ const clamped = await callPost(settings, { patch: { minFreeDiskGB: -5 } });
+ assert.equal(clamped.status, 400);
+ assert.match(String(clamped.body.error), /minFreeDiskGB: sent -5, would be saved as \d+/);
+
+ const empty = await callPost(settings, { patch: {} });
+ assert.equal(empty.status, 400);
+ const notObject = await callPost(settings, { patch: [] });
+ assert.equal(notObject.status, 400);
+
+ assert.deepEqual(await readSettingsFile(), before);
+});
+
+test("settings: a valid patch is written through saveSettings and answered with the saved value", async () => {
+ const settings = await route("settings");
+ const res = await callPost(settings, {
+ patch: { minFreeDiskGB: 7, syncScheduler: { fullSweepIntervalMinutes: 0 } },
+ });
+ assert.equal(res.status, 200, JSON.stringify(res.body));
+ assert.deepEqual(res.body.changed, ["minFreeDiskGB", "syncScheduler"]);
+ const value = res.body.value as Record<string, unknown>;
+ assert.equal(value.minFreeDiskGB, 7);
+ const file = await readSettingsFile();
+ assert.equal(file.minFreeDiskGB, 7);
+ // One level merged: the block's other keys are kept, not dropped.
+ assert.equal((file.syncScheduler as Record<string, unknown>).fullSweepIntervalMinutes, 0);
+ await corpus.writeSettings({ minFreeDiskGB: 0 });
+});
+
+test("the patch rule: leaves compared, arrays element by element, absent keys not compared", () => {
+ assert.deepEqual(coercedLeaves({ a: 1, b: { c: "x" } }, { a: 1, b: { c: "x", d: 2 } }, "k"), []);
+ assert.deepEqual(coercedLeaves({ b: { c: "x" } }, { b: { c: "y" } }, "k"), [
+ 'k.b.c: sent "x", would be saved as "y"',
+ ]);
+ assert.deepEqual(coercedLeaves([1, 2], [1], "k"), ["k: sent [1,2], would be saved as [1]"]);
+ assert.equal(settingsPatchProblem({ a: 1 }, ["a"], { a: 1 }), null);
+});
+
+test("lane: the publish lane is a lane; an unknown one is refused by the full list", async () => {
+ const lane = await route("lane");
+ const bogus = await callPost(lane, { lane: "transcode", held: true });
+ assert.equal(bogus.status, 400);
+ assert.match(String(bogus.body.error), /transcription, download, digest, backfill, publish/);
+ // Switched off: the publish page's Start refuses, and so does this.
+ const start = await callPost(lane, { lane: "publish", action: "start" });
+ assert.equal(start.status, 400, JSON.stringify(start.body));
+ assert.ok(String(start.body.error).length > 0);
+});
+
+test("clear-platform-hold refuses a name that is not a platform", async () => {
+ const res = await callPost(await route("clear-platform-hold"), { platform: "you tube/../x" });
+ assert.equal(res.status, 400);
+ assert.match(String(res.body.error), /^Not a platform: /);
+ const none = await callPost(await route("clear-platform-hold"), {});
+ assert.equal(none.status, 400);
+});
+
+test("workers: the op and the ids are checked at the door", async () => {
+ const workers = await route("workers");
+ const op = await callPost(workers, { op: "restart", ids: ["a"] });
+ assert.equal(op.status, 400);
+ assert.match(String(op.body.error), /"op" must be one of enable, disable, drain/);
+ const ids = await callPost(workers, { op: "enable", ids: [] });
+ assert.equal(ids.status, 400);
+});
+
+test("transcribe-one, delete-file and do-not-clean refuse before anything is touched", async () => {
+ const one = await route("transcribe-one");
+ const traversing = await callPost(one, { slug: SLUG, id: VIDEO, file: "../audio.mp3" });
+ assert.equal(traversing.status, 400);
+ assert.match(String(traversing.body.error), /is not a file name in the video's directory/);
+ const noChannel = await callPost(one, { slug: "no-such", id: VIDEO });
+ assert.equal(noChannel.status, 400);
+ assert.match(String(noChannel.body.error), /no-such/);
+
+ const del = await route("delete-file");
+ const suspicious = await callPost(del, { slug: SLUG, id: VIDEO, file: "../config.json" });
+ assert.equal(suspicious.status, 400);
+ assert.match(String(suspicious.body.error), /Refusing to delete suspicious filename/);
+ const noFile = await callPost(del, { slug: SLUG, id: VIDEO });
+ assert.equal(noFile.status, 400);
+ assert.match(String(noFile.body.error), /"file" is required/);
+
+ const keep = await route("do-not-clean");
+ const missing = await callPost(keep, { slug: SLUG, id: "not-a-video" });
+ assert.equal(missing.status, 400);
+ assert.equal(missing.body.error, "Video directory not found: not-a-video");
+ const badKeep = await callPost(keep, { slug: SLUG, id: VIDEO, keep: "yes" });
+ assert.equal(badKeep.status, 400);
+ assert.match(String(badKeep.body.error), /"keep" must be a boolean/);
+ assert.deepEqual(await corpus.listJobIds(), []);
+});
+
+test("cleanup refuses an unknown sweep and a traversing slug, before any job", async () => {
+ const cleanup = await route("cleanup");
+ const sweep = await callPost(cleanup, { slug: SLUG, sweep: "everything" });
+ assert.equal(sweep.status, 400);
+ assert.match(
+ String(sweep.body.error),
+ /"sweep" must be one of transcribed, extra-formats, wrong-format, failed-transcriptions/,
+ );
+ const slug = await callPost(cleanup, { slug: "../x", sweep: "transcribed" });
+ assert.equal(slug.status, 400);
+ assert.deepEqual(await corpus.listJobIds(), []);
+});
+
+test("relocate dryRun answers each channel's preview and moves nothing", async () => {
+ const relocate = await route("relocate");
+ const res = await callPost(relocate, { slugs: [SLUG, "no-such"], root: elsewhere, dryRun: true });
+ assert.equal(res.status, 200, JSON.stringify(res.body));
+ assert.equal(res.body.dryRun, true);
+ const [ok, missing] = res.body.previews as { slug: string; preview?: unknown; error?: string }[];
+ assert.equal(ok.slug, SLUG);
+ assert.ok(ok.preview, JSON.stringify(ok));
+ assert.equal(missing.slug, "no-such");
+ assert.ok(missing.error);
+ // No move: the config names no media dir, and no job exists.
+ const config = JSON.parse(
+ await readFile(path.join(corpus.transcripts, "channels", SLUG, "config.json"), "utf8"),
+ ) as { mediaDir?: string };
+ assert.equal(config.mediaDir, undefined);
+ assert.deepEqual(await corpus.listJobIds(), []);
+ const both = await callPost(relocate, { slugs: [SLUG], root: elsewhere, locationId: "x", dryRun: true });
+ assert.equal(both.status, 400);
+});
diff --git a/editor/app/api/ops/cleanup/route.ts b/editor/app/api/ops/cleanup/route.ts
@@ -0,0 +1,43 @@
+import {
+ cleanAudioAction,
+ cleanExtraAudioFormatsAction,
+ clearFailedTranscriptionsAction,
+ removeWrongFormatAudioAction,
+} from "../../../channels/[slug]/whisperActions";
+import { jobResponse, oneOf, ops, reqSlug } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, sweep: "transcribed" | "extra-formats" | "wrong-format" |
+// "failed-transcriptions" }
+//
+// One of a channel's cleanup sweeps — the channel page's Cleanup buttons, as a
+// job on the channel's queue ({ ok, jobId }; --wait follows it):
+// transcribed "Clean audio": delete the audio of every video that
+// has a transcript (cleanAudioAction) — the sweep the
+// /cleanup total counts;
+// extra-formats delete a video's audio in formats other than the
+// channel's target, where the target is on disk;
+// wrong-format delete every finalized audio file not in the target
+// format, including failed-extract orphans;
+// failed-transcriptions clear the channel's failed-transcriptions list, so
+// "Transcribe missing" retries them (deletes no media).
+// Each is the page's own action, so it keeps what the page's keeps: every
+// sweep skips a do-not-clean video; "transcribed" also keeps the channel's
+// keep-latest window and audio still awaiting diarization; every one deletes
+// through the media tier's remover (removeMediaFile), never a bare link.
+// `get cleanup <slug>` says what each would reclaim.
+const SWEEPS = {
+ transcribed: cleanAudioAction,
+ "extra-formats": cleanExtraAudioFormatsAction,
+ "wrong-format": removeWrongFormatAudioAction,
+ "failed-transcriptions": clearFailedTranscriptionsAction,
+} as const;
+
+export async function POST(request: Request) {
+ return ops(request, ["slug", "sweep"], async (body) => {
+ const slug = reqSlug(body, "slug");
+ const sweep = oneOf(body, "sweep", Object.keys(SWEEPS) as (keyof typeof SWEEPS)[]);
+ return jobResponse(await SWEEPS[sweep](slug));
+ });
+}
diff --git a/editor/app/api/ops/clear-platform-hold/route.ts b/editor/app/api/ops/clear-platform-hold/route.ts
@@ -0,0 +1,21 @@
+import { NextResponse } from "next/server";
+import { clearPlatformHoldAction } from "../../../operations/pacingActions";
+import { ops, opsFail, reqString } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { platform: "youtube" | "rumble" | … }
+//
+// The lane strip's "Clear hold": a held platform's hold, its backoff and its
+// raised pace all go, and the next failure starts the escalation from the
+// bottom. The action runs it as a one-step job (kind `clear-platform-hold`, on
+// its own queue) and waits for it, so the answer carries the job's sentence:
+// { ok, message } — "youtube had no hold, backoff or raised pace to clear." when
+// there was nothing to clear.
+export async function POST(request: Request) {
+ return ops(request, ["platform"], async (body) => {
+ const res = await clearPlatformHoldAction(reqString(body, "platform"));
+ if (!res.ok) return opsFail(res.error);
+ return NextResponse.json({ ok: true, message: res.message });
+ });
+}
diff --git a/editor/app/api/ops/delete-file/route.ts b/editor/app/api/ops/delete-file/route.ts
@@ -0,0 +1,26 @@
+import { NextResponse } from "next/server";
+import { deleteVideoFileAction } from "../../../channels/[slug]/videos/[id]/videoActions";
+import { ops, opsFail, reqSlug, reqString, reqVideoId } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, id, file }
+//
+// The video page's Files list delete: one file of one video's directory. It is
+// deleteVideoFileAction, so it is `removeMediaFile` — a tiered media file goes
+// with its bytes on the media drive, never as a bare link — and it refuses what
+// the button refuses: a name that is not one entry of the directory, a file on
+// a media drive that is not mounted or not answering (nothing is deleted), a
+// channel whose drive is not answering, something that is not a regular file.
+// The channel's report is refreshed after. No undo.
+export async function POST(request: Request) {
+ return ops(request, ["slug", "id", "file"], async (body) => {
+ const res = await deleteVideoFileAction(
+ reqSlug(body, "slug"),
+ reqVideoId(body, "id"),
+ reqString(body, "file"),
+ );
+ if (!res.ok) return opsFail(res.error);
+ return NextResponse.json({ ok: true });
+ });
+}
diff --git a/editor/app/api/ops/do-not-clean/route.ts b/editor/app/api/ops/do-not-clean/route.ts
@@ -0,0 +1,23 @@
+import { NextResponse } from "next/server";
+import { toggleDoNotCleanAction } from "../../../channels/[slug]/videos/[id]/videoActions";
+import { ops, opsFail, optBool, reqSlug, reqVideoId } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, id, keep?: boolean }
+//
+// One video's "Do not clean" marker (do-not-clean.json), set to the value
+// given — `keep` defaults to true; false removes it. Set, not toggled: a retry
+// of the same body is the same answer. The cleanup sweeps skip a marked
+// video's audio. `keep-videos` is the bulk form, by title/description match.
+export async function POST(request: Request) {
+ return ops(request, ["slug", "id", "keep"], async (body) => {
+ const res = await toggleDoNotCleanAction(
+ reqSlug(body, "slug"),
+ reqVideoId(body, "id"),
+ optBool(body, "keep") ?? true,
+ );
+ if (!res.ok) return opsFail(res.error);
+ return NextResponse.json({ ok: true });
+ });
+}
diff --git a/editor/app/api/ops/lane/route.ts b/editor/app/api/ops/lane/route.ts
@@ -1,19 +1,29 @@
import { NextResponse } from "next/server";
-import { LANES, type AutoQueueKind } from "yt-dlp-transcript-common/lib/autoQueueTypes";
+import {
+ LANES,
+ PIPELINE_LANES,
+ type AutoQueueKind,
+} from "yt-dlp-transcript-common/lib/autoQueueTypes";
import { getSettings } from "yt-dlp-transcript-common/lib/settings";
-import { drainAutoRunner } from "yt-dlp-transcript-common/controller/autoRunner";
import {
+ drainAutoQueueAction,
pauseLaneAction,
resumeLaneAction,
saveAutoQueueAction,
startAutoQueueAction,
stopAutoQueueAction,
} from "../../../operations/actions";
+import { savePublishSettingsAction } from "../../../operations/settingsActions";
+import {
+ drainPublishLaneAction,
+ startPublishLaneAction,
+ stopPublishLaneAction,
+} from "../../../sites/lib/publishActions";
import { OpsInputError, okResponse, ops, oneOf, optBool } from "../_lib";
export const dynamic = "force-dynamic";
-// POST { lane: transcription|download|digest|backfill,
+// POST { lane: transcription|download|digest|backfill|publish,
// held?: boolean, enabled?: boolean, action?: "start"|"stop"|"drain" }
//
// One lane, up to three independent changes, applied in the order the operator
@@ -26,9 +36,35 @@ export const dynamic = "force-dynamic";
//
// `action` mirrors /api/auto-queue/control's body — same three verbs, same
// meanings — so a caller that already drives that route needs nothing new.
+//
+// THE PUBLISH LANE (release 19, A4) is a pipeline lane, not a LANES entry: its
+// switch is `settings.publish.enabled` (the /operations/publish form's, saved
+// through that form's own action with every other field as stored), its gate
+// `publish.held`, and Start, Drain and Stop are that page's three buttons.
+// A Start the lane refuses (switched off, or quiet hours) is a 400 with the
+// page's sentence.
+const ALL_LANES = [...LANES, ...PIPELINE_LANES] as const;
+
+// The /operations/publish form, filled from the stored block with only the
+// switch changed — so the form's own validation runs on what is stored.
+function publishForm(enabled: boolean): FormData {
+ const p = getSettings().publish;
+ const fd = new FormData();
+ if (enabled) fd.set("publishEnabled", "on");
+ fd.set("publishCheckEveryMinutes", String(p.checkEveryMinutes));
+ fd.set("publishRefreshEveryMinutes", String(p.refreshEveryMinutes));
+ fd.set("publishQuietStart", p.quietHours ? String(p.quietHours.start) : "");
+ fd.set("publishQuietEnd", p.quietHours ? String(p.quietHours.end) : "");
+ fd.set("publishRunner", p.runner);
+ fd.set("publishPreviewBranch", p.previewBranch);
+ fd.set("publishHub", p.hub);
+ fd.set("publishHomepage", p.homepage);
+ return fd;
+}
+
export async function POST(request: Request) {
return ops(request, ["lane", "held", "enabled", "action"], async (body) => {
- const lane = oneOf(body, "lane", LANES) as AutoQueueKind;
+ const lane = oneOf(body, "lane", ALL_LANES);
const enabled = optBool(body, "enabled");
const held = optBool(body, "held");
const action = body.action;
@@ -46,19 +82,24 @@ export async function POST(request: Request) {
);
}
if (enabled !== undefined) {
- // EVERY OTHER POLICY FIELD IS CARRIED FROM THE STORED POLICY, including
- // `root`: saveAutoQueueAction refuses a tree edit while the priority
- // model compiles the roots, and handing it back the stored tree is what
- // makes this a pure enable/disable rather than a tree write.
- const policy = getSettings().autoQueue[lane];
- const saved = await saveAutoQueueAction(lane, {
- enabled,
- maxWorkers: policy.maxWorkers,
- replaceAutoSubs: policy.replaceAutoSubs === true,
- order: policy.order ?? "listed",
- root: policy.root,
- });
- if (!saved.ok) return okResponse(saved);
+ if (lane === "publish") {
+ const saved = await savePublishSettingsAction(undefined, publishForm(enabled));
+ if (!saved.ok) return okResponse(saved);
+ } else {
+ // EVERY OTHER POLICY FIELD IS CARRIED FROM THE STORED POLICY, including
+ // `root`: saveAutoQueueAction refuses a tree edit while the priority
+ // model compiles the roots, and handing it back the stored tree is what
+ // makes this a pure enable/disable rather than a tree write.
+ const policy = getSettings().autoQueue[lane];
+ const saved = await saveAutoQueueAction(lane, {
+ enabled,
+ maxWorkers: policy.maxWorkers,
+ replaceAutoSubs: policy.replaceAutoSubs === true,
+ order: policy.order ?? "listed",
+ root: policy.root,
+ });
+ if (!saved.ok) return okResponse(saved);
+ }
}
if (held !== undefined) {
const gated = held
@@ -66,14 +107,17 @@ export async function POST(request: Request) {
: await resumeLaneAction(lane);
if (!gated.ok) return okResponse(gated);
}
- if (action === "start") {
- const started = await startAutoQueueAction(lane);
- if (!started.ok) return okResponse(started);
- } else if (action === "stop") {
- const stopped = await stopAutoQueueAction(lane);
- if (!stopped.ok) return okResponse(stopped);
- } else if (action === "drain") {
- drainAutoRunner(lane);
+ if (action !== undefined) {
+ const run =
+ lane === "publish"
+ ? { start: startPublishLaneAction, stop: stopPublishLaneAction, drain: drainPublishLaneAction }[action]
+ : {
+ start: () => startAutoQueueAction(lane as AutoQueueKind),
+ stop: () => stopAutoQueueAction(lane as AutoQueueKind),
+ drain: () => drainAutoQueueAction(lane as AutoQueueKind),
+ }[action];
+ const done = await run();
+ if (!done.ok) return okResponse(done);
}
return NextResponse.json({ ok: true, lane });
});
diff --git a/editor/app/api/ops/relocate/route.ts b/editor/app/api/ops/relocate/route.ts
@@ -1,7 +1,12 @@
+import { NextResponse } from "next/server";
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels";
import { bulkRelocateChannelMediaAction } from "../../../channels/bulkStorageActions";
+import { previewRelocationAction } from "../../../channels/[slug]/storageActions";
import {
OpsInputError,
ops,
+ optBool,
optString,
queueResponse,
reqSlugs,
@@ -9,7 +14,7 @@ import {
export const dynamic = "force-dynamic";
-// POST { slugs: string[], locationId?: string, root?: string }
+// POST { slugs: string[], locationId?: string, root?: string, dryRun?: boolean }
// -> { ok: true, queued, skipped }
//
// THE DESTINATION IS A LOCATION ID WHEREVER POSSIBLE — the root is resolved on
@@ -17,23 +22,43 @@ export const dynamic = "force-dynamic";
// cannot aim a batch somewhere a re-point has moved. `root` is the one-off
// escape hatch the panel also offers. A skip is not a failure: every slug that
// did not queue comes back with the same sentence the bulk bar shows.
+//
+// `dryRun: true` (release 19, A4) is the Storage panel's preview per channel
+// (previewRelocationAction) and starts nothing: { ok, dryRun, previews:
+// [{slug, preview} | {slug, error}] } — the bytes to copy and both sides'
+// free space. As on the panel, a preview of a CLASSIC channel (its big files
+// still real files in data/) first tiers it in place — same-disk renames into
+// its own media/, the step every move's preflight takes (release 17's ruling);
+// no byte leaves the corpus disk.
export async function POST(request: Request) {
- return ops(request, ["slugs", "locationId", "root"], async (body) => {
+ return ops(request, ["slugs", "locationId", "root", "dryRun"], async (body) => {
const slugs = reqSlugs(body, "slugs");
const locationId = optString(body, "locationId");
const root = optString(body, "root");
+ const dryRun = optBool(body, "dryRun") === true;
if ((locationId ? 1 : 0) + (root ? 1 : 0) !== 1) {
throw new OpsInputError(
'send exactly one of "locationId" (a location configured on /storage) or "root" (an absolute path)',
);
}
- return queueResponse(
- await bulkRelocateChannelMediaAction(
- slugs,
- locationId
- ? { kind: "location", locationId }
- : { kind: "custom", root: root as string },
- ),
- );
+ const dest = locationId
+ ? ({ kind: "location", locationId } as const)
+ : ({ kind: "custom", root: root as string } as const);
+ if (dryRun) {
+ const previews = [];
+ for (const slug of slugs) {
+ // The bulk move's own skip for a slug with no channel: the panel's
+ // preview is only ever asked about a channel that exists, and a
+ // zero-byte preview of a typo is the silent success this layer refuses.
+ if (!(await readChannelConfig(getPaths(), slug))) {
+ previews.push({ slug, error: "channel not found" });
+ continue;
+ }
+ const res = await previewRelocationAction(slug, dest);
+ previews.push(res.ok ? { slug, preview: res.preview } : { slug, error: res.error });
+ }
+ return NextResponse.json({ ok: true, dryRun: true, previews });
+ }
+ return queueResponse(await bulkRelocateChannelMediaAction(slugs, dest));
});
}
diff --git a/editor/app/api/ops/settings/patch.ts b/editor/app/api/ops/settings/patch.ts
@@ -0,0 +1,66 @@
+// WHAT `POST /api/ops/settings` REFUSES, as one pure function — testable
+// without a settings file, a schema or a server.
+
+// Keys another writer owns. Writing one raw would skip what that writer keeps
+// true, so the refusal names the writer:
+// - channelPriority: the priority write compiles the four lanes' trees in the
+// SAME write (saveSettings.ts' header); a raw one leaves them disagreeing.
+// - autoQueue: a lane's tree is its policy editor's, or compiled from the
+// channel priorities — and the editor refuses a tree edit while they rule.
+// - workers: /workers validates each worker (templates, URLs) on save.
+// - storage: a location is probed and its volume recorded when added or
+// re-pointed (/storage); a raw root would skip both.
+export const OWNED_KEYS: Readonly<Record<string, string>> = {
+ channelPriority: "it is written with the lanes' trees — use `pnpm ops channel-priority`",
+ autoQueue:
+ "a lane's switch, hold and runner are `pnpm ops lane`; its tree is the policy editor's on /operations, or compiled from channel priorities",
+ workers: "workers are saved on /workers; `pnpm ops workers` enables and disables them",
+ storage: "locations are added and re-pointed on /storage, which probes them; a channel's media moves with `pnpm ops relocate`",
+};
+
+function isObject(v: unknown): v is Record<string, unknown> {
+ return typeof v === "object" && v !== null && !Array.isArray(v);
+}
+
+// Every leaf of `sent` whose value in `saved` is not the same, as
+// "path: sent X, would be saved as Y". An object in `sent` is compared key by
+// key (a key it leaves out is not compared); an array must match in length and
+// element by element.
+export function coercedLeaves(sent: unknown, saved: unknown, at: string): string[] {
+ if (isObject(sent)) {
+ if (!isObject(saved)) return [`${at}: sent an object, would be saved as ${JSON.stringify(saved)}`];
+ return Object.keys(sent).flatMap((k) => coercedLeaves(sent[k], saved[k], `${at}.${k}`));
+ }
+ if (Array.isArray(sent)) {
+ if (!Array.isArray(saved) || saved.length !== sent.length) {
+ return [`${at}: sent ${JSON.stringify(sent)}, would be saved as ${JSON.stringify(saved)}`];
+ }
+ return sent.flatMap((v, i) => coercedLeaves(v, saved[i], `${at}[${i}]`));
+ }
+ return Object.is(sent, saved)
+ ? []
+ : [`${at}: sent ${JSON.stringify(sent)}, would be saved as ${JSON.stringify(saved)}`];
+}
+
+// The sentence to refuse `patch` with, or null. `known` is the schema's
+// top-level keys; `parsed` is the schema's reading of the merged settings.
+export function settingsPatchProblem(
+ patch: Record<string, unknown>,
+ known: readonly string[],
+ parsed: Record<string, unknown>,
+): string | null {
+ const keys = Object.keys(patch);
+ const unknown = keys.filter((k) => !known.includes(k));
+ if (unknown.length) {
+ return `unknown settings key(s): ${unknown.join(", ")} — SETTINGS.md lists them`;
+ }
+ const owned = keys.filter((k) => Object.hasOwn(OWNED_KEYS, k));
+ if (owned.length) {
+ return owned.map((k) => `"${k}" is not patched here: ${OWNED_KEYS[k]}`).join("; ");
+ }
+ const coerced = keys.flatMap((k) => coercedLeaves(patch[k], parsed[k], k));
+ if (coerced.length) {
+ return `not saved — the schema would not keep ${coerced.length === 1 ? "this value" : "these values"} as sent: ${coerced.join("; ")}`;
+ }
+ return null;
+}
diff --git a/editor/app/api/ops/settings/route.ts b/editor/app/api/ops/settings/route.ts
@@ -1,6 +1,9 @@
-import { getSettings } from "yt-dlp-transcript-common/lib/settings";
-import { opsFail } from "../_lib";
+import { NextResponse } from "next/server";
+import { getSettings, siteSettingsSchema } from "yt-dlp-transcript-common/lib/settings";
+import { mergeSettingsPatch, saveSettings } from "../../../settings/saveSettings";
+import { OpsInputError, ops, opsFail } from "../_lib";
import { readRoute, redactSecrets } from "../_read";
+import { settingsPatchProblem } from "./patch";
export const dynamic = "force-dynamic";
@@ -21,3 +24,52 @@ export async function GET(request: Request) {
return { key, value: settings[key] };
});
}
+
+// POST { patch: { <top-level key>: <value>, … } }
+//
+// THE EDITOR'S ONE SETTINGS WRITER, over HTTP: `saveSettings(patch)` — the
+// function every settings form calls — with its merge rule (a block's keys
+// merge one level; an array, a scalar, or an object nested in a block
+// replaces). SETTINGS.md is the key table.
+//
+// REFUSED, before anything is written (`settings/patch.ts`):
+// - an unknown top-level key, named;
+// - a key another writer owns, because writing it raw breaks what that
+// writer keeps true (channelPriority, autoQueue, workers, storage — each
+// refusal names the command or page that writes it);
+// - a value the schema would not keep as sent. The schema never throws: it
+// coerces (clamps a number, drops an unknown enum, fills a default), so a
+// `{ ok: true }` would otherwise be the answer to a value that was never
+// saved. Every leaf the patch names is compared with what the schema makes
+// of it, and a difference is a 400 naming the path, what was sent and what
+// would be saved. A leaf the patch leaves out may still be filled with its
+// default — which is the merge rule, and is what the answer's `value` shows.
+//
+// Answers { ok, changed: [keys], value: { <key>: <saved value> } } — read back
+// after the write, secrets redacted.
+export async function POST(request: Request) {
+ return ops(request, ["patch"], async (body) => {
+ const patch = body.patch;
+ if (typeof patch !== "object" || patch === null || Array.isArray(patch)) {
+ throw new OpsInputError('"patch" is required and must be an object of top-level settings keys');
+ }
+ const keys = Object.keys(patch);
+ if (keys.length === 0) throw new OpsInputError('"patch" is empty — nothing to change');
+ const current = getSettings();
+ const problem = settingsPatchProblem(
+ patch as Record<string, unknown>,
+ Object.keys(siteSettingsSchema.shape),
+ siteSettingsSchema.parse(
+ mergeSettingsPatch(current, patch as Partial<typeof current>),
+ ) as unknown as Record<string, unknown>,
+ );
+ if (problem) throw new OpsInputError(problem);
+ await saveSettings(patch as Partial<typeof current>);
+ const saved = redactSecrets(getSettings()) as unknown as Record<string, unknown>;
+ return NextResponse.json({
+ ok: true,
+ changed: keys,
+ value: Object.fromEntries(keys.map((k) => [k, saved[k]])),
+ });
+ });
+}
diff --git a/editor/app/api/ops/transcribe-one/route.ts b/editor/app/api/ops/transcribe-one/route.ts
@@ -0,0 +1,36 @@
+import {
+ transcribeOneAction,
+ whisperVideoAction,
+} from "../../../channels/[slug]/videos/[id]/videoActions";
+import { jobResponse, ops, opsFail, optString, reqSlug, reqVideoId } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, id, file? }
+//
+// One video, transcribed — the video page's two buttons:
+// - with "file" (an audio file in the video's dir, e.g. "audio.mp3"): that
+// file, through the worker pool (transcribeOneAction — the per-file
+// Transcribe on the Files list);
+// - without: the page's Transcribe (whisperVideoAction), which uses the audio
+// on disk, or first downloads it through the channel's managed path when
+// there is none (the disk floor applies to that download only).
+// A job on the video's platform queue: { ok, jobId }, so --wait follows it.
+// The file name is one path segment; a refused one is the action's sentence.
+export async function POST(request: Request) {
+ return ops(request, ["slug", "id", "file"], async (body) => {
+ const slug = reqSlug(body, "slug");
+ const id = reqVideoId(body, "id");
+ const file = optString(body, "file");
+ // One path segment, checked at the door as reqVideoId checks an id: the
+ // name is joined under the video's directory.
+ if (file !== undefined && (file === "" || /[/\\\0]/.test(file) || file.startsWith("."))) {
+ return opsFail(`"${file}" is not a file name in the video's directory`);
+ }
+ return jobResponse(
+ file !== undefined
+ ? await transcribeOneAction(slug, id, file)
+ : await whisperVideoAction(slug, id),
+ );
+ });
+}
diff --git a/editor/app/api/ops/workers/route.ts b/editor/app/api/ops/workers/route.ts
@@ -1,4 +1,11 @@
+import { NextResponse } from "next/server";
import { buildWorkersPayload } from "../../../workers/buildWorkers";
+import {
+ disableWorkerAction,
+ drainWorkerAction,
+ enableWorkerAction,
+} from "../../../workers/actions";
+import { oneOf, ops, opsFail, reqStringArray } from "../_lib";
import { readRoute, redactSecrets } from "../_read";
export const dynamic = "force-dynamic";
@@ -14,3 +21,28 @@ export async function GET(request: Request) {
workers: redactSecrets(buildWorkersPayload()),
}));
}
+
+// POST { op: "enable" | "disable" | "drain", ids: [<worker id>, …] }
+//
+// The /workers row's switch and its Drain, per id (release 19, A4): enable
+// and disable are the switch — the pool's LIVE state, as the page sets it: a
+// restart brings back the launch default (/workers' "Set as default"); drain
+// stops giving the worker new transcriptions and lets the one in flight
+// finish. An unknown id is named and does not stop the others; the answer is
+// `ok: false` (400) when any id was refused, with every id's result.
+const OPS = ["enable", "disable", "drain"] as const;
+
+export async function POST(request: Request) {
+ return ops(request, ["op", "ids"], async (body) => {
+ const op = oneOf(body, "op", OPS);
+ const ids = [...new Set(reqStringArray(body, "ids"))];
+ const run = { enable: enableWorkerAction, disable: disableWorkerAction, drain: drainWorkerAction }[op];
+ const results: { id: string; ok: boolean; error?: string }[] = [];
+ for (const id of ids) results.push({ id, ...(await run(id)) });
+ const failed = results.filter((r) => !r.ok);
+ if (failed.length) {
+ return opsFail(failed.map((r) => r.error ?? r.id).join("; "), 400, { op, results });
+ }
+ return NextResponse.json({ ok: true, op, results });
+ });
+}
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -51,6 +51,13 @@
// pnpm ops lane --json '{"lane":"download","held":true}'
// pnpm ops refresh-report --json '{"all":true}'
// pnpm ops relocate --json '{"slugs":["x"],"locationId":"platter"}'
+// pnpm ops relocate --json '{"slugs":["x"],"locationId":"platter","dryRun":true}'
+// pnpm ops settings --json '{"patch":{"minFreeDiskGB":20}}'
+// pnpm ops lane --json '{"lane":"publish","action":"drain"}'
+// pnpm ops cleanup --json '{"slug":"x","sweep":"transcribed"}' --wait
+// pnpm ops workers --json '{"op":"disable","ids":["parakeet-cpu"]}'
+// pnpm ops get job <jobId> --tail 20
+// pnpm ops get jobs --failed --slug x
// pnpm ops build-site --json '{"siteId":"anilyzer"}' --wait
// pnpm ops build-deploy --json '{"siteIds":["anilyzer","jeralyzer"]}' --wait
// pnpm ops deploy-site --json '{"siteId":"anilyzer","preview":"tags-exclude"}' --wait
@@ -273,6 +280,19 @@ const ACTIONS = [
// ({path, start?, end?, workerId?, out?}) — the corpus's own engine and
// model, as a job. With --wait the result JSON is what stdout carries.
"transcribe",
+ // ARCHIVAL WRITES (release 19, A4). settings.json through the editor's one
+ // writer ({patch}); a platform's rate-limit hold cleared ({platform});
+ // workers switched on, off or drained ({op, ids}); one video transcribed
+ // ({slug, id, file?}), one of its files deleted ({slug, id, file}), its
+ // do-not-clean marker set ({slug, id, keep?}); a channel's cleanup sweep
+ // ({slug, sweep}).
+ "settings",
+ "clear-platform-hold",
+ "workers",
+ "transcribe-one",
+ "delete-file",
+ "do-not-clean",
+ "cleanup",
];
// The log line a transcribe job ends with: this marker, then the result as
@@ -632,6 +652,48 @@ export function usage() {
' report.md and evidence-pack.zip for the site\'s build to publish:',
' {"siteId", "reportId"?, "formats"?: ["html","pdf","md","zip"]}.',
"",
+ 'settings patches settings.json through the editor\'s one writer:',
+ ' {"patch": {"<top-level key>": value, …}} (SETTINGS.md). A block\'s keys',
+ ' merge one level; an array, a scalar or an object nested in a block',
+ ' replaces. Refused before anything is written: an unknown key; a key',
+ ' another writer owns (channelPriority, autoQueue, workers, storage — the',
+ ' refusal names it); a value the schema would not keep as sent (clamped,',
+ ' dropped), named with what would be saved. Answers the saved values.',
+ "",
+ 'lane takes {"lane": "transcription"|"download"|"digest"|"backfill"|',
+ ' "publish", "enabled"?, "held"?, "action"?: "start"|"stop"|"drain"}.',
+ "",
+ 'clear-platform-hold clears a platform\'s rate-limit hold, backoff and raised',
+ ' pace, as the lane strip\'s Clear hold: {"platform"}. Answers the sentence.',
+ "",
+ 'workers switches transcription workers as /workers does: {"op": "enable" |',
+ ' "disable" | "drain", "ids": [...]} — the live pool; a restart restores',
+ ' the launch default. An unknown id is named; the others still apply.',
+ "",
+ 'transcribe-one transcribes one video: {"slug", "id", "file"?}. With "file"',
+ ' (an audio file in its dir) that file; without, the video page\'s',
+ ' Transcribe — the audio on disk, or downloaded first through the channel\'s',
+ ' managed path. A job; --wait follows it.',
+ "",
+ 'delete-file deletes one file of one video, as its Files list does:',
+ ' {"slug", "id", "file"}. A tiered media file goes with its bytes on the',
+ ' media drive (never a bare link); refused while that drive is not',
+ ' answering. No undo.',
+ "",
+ 'do-not-clean sets one video\'s "Do not clean" marker: {"slug", "id",',
+ ' "keep"?: true}; "keep": false removes it. (keep-videos is the bulk form.)',
+ "",
+ 'cleanup runs one of a channel\'s cleanup sweeps as a job: {"slug", "sweep":',
+ ' "transcribed" (audio of transcribed videos — the /cleanup total) |',
+ ' "extra-formats" | "wrong-format" | "failed-transcriptions" (clears the',
+ ' list; deletes nothing)}. Do-not-clean videos are skipped; `get cleanup',
+ ' <slug>` says what each would reclaim.',
+ "",
+ 'relocate moves channels\' media to a location: {"slugs", "locationId" |',
+ ' "root"}. "dryRun": true answers each channel\'s preview (bytes to copy,',
+ ' free space both sides) and moves nothing — though, as on the Storage',
+ ' panel, a channel never tiered is first tiered in place on the corpus disk.',
+ "",
'channel-config changes a channel as its Configure form does: {"slug"} and',
' any of "patch" (form field names; "" clears one), "sites" (the WHOLE',
' membership set: [{"siteId", "groupId"? | "newGroupName"?}], [] = on no',
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -761,3 +761,25 @@ test("the read-side nouns are GETs on their routes, named in the usage", () => {
assert.match(usage(), /pnpm ops get settings \[<key>\]/);
assert.match(usage(), /get cleanup <slug> is one channel's \/cleanup row/);
});
+
+// ARCHIVAL WRITES (A4).
+test("the archival writes are POSTs to their routes, each named in the usage", () => {
+ for (const action of [
+ "settings",
+ "clear-platform-hold",
+ "workers",
+ "transcribe-one",
+ "delete-file",
+ "do-not-clean",
+ "cleanup",
+ ]) {
+ const p = parseArgs([action, "--json", "{}"]);
+ assert.equal(p.method, "POST", action);
+ assert.equal(p.path, `/api/ops/${action}`, action);
+ assert.match(usage(), new RegExp(`^${action} `, "m"), action);
+ }
+ // `get settings` and `settings` are two different requests on one route.
+ assert.equal(parseArgs(["get", "settings"]).method, "GET");
+ assert.match(usage(), /"dryRun": true answers each channel's preview/);
+ assert.match(usage(), /"publish", "enabled"\?, "held"\?, "action"\?/);
+});