commit 46f8f4ef3cbc46e5a20d02342f065ad3d892fd2d
parent e082d44f08184d7b1d5c3199c273033a94b6ed3a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 25 Sep 2026 12:38:52 -0400
ops: keep-videos — POST {slug, match, fields?, note?, dryRun?}
An adapter over keepVideosAction; pnpm ops keep-videos; ops-api.spec +1
(dry run writes nothing, only the matching dir gets the marker, a scan-only
match gets no dir, unknown key and bad regex are 400s).
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
4 files changed, 145 insertions(+), 0 deletions(-)
diff --git a/editor/app/api/ops/keep-videos/route.ts b/editor/app/api/ops/keep-videos/route.ts
@@ -0,0 +1,57 @@
+import { NextResponse } from "next/server";
+import { keepVideosAction } from "../../../channels/[slug]/videos/[id]/videoActions";
+import { OpsInputError, ops, optBool, optString, opsFail, reqSlug, reqString } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, match, fields?: ("title"|"description")[], note?, dryRun? }
+// -> { ok: true, pattern, fields, considered, matched: [{ id, title,
+// downloaded, alreadyKept, marked }], marked, alreadyKept,
+// notDownloaded, unscanned, noMetadata, dryRun }
+//
+// AN ADAPTER: one call to keepVideosAction, the bulk form of the video page's
+// "Do not clean" toggle. It writes the same per-video do-not-clean.json marker
+// every cleaner already honours; it adds no protection of its own.
+//
+// `match` IS A DOWNLOAD-FILTER PATTERN, matched the way the filter matches it —
+// a case-insensitive regex SOURCE over title + "\n" + description, compiled by
+// the same function. A pattern that keeps the right videos here would download
+// the right ones as a channel's `downloadFilterInclude`. `fields` narrows the
+// subject to one half.
+//
+// A MATCH WITH NO data/<id>/ IS REPORTED, NOT CREATED. The marker lives in the
+// video's dir, and a dir is what every enumerator reads as "fetched". Those ids
+// come back in `notDownloaded`: send them to `download-missing`, then run this
+// again. A fresh channel wants `metadata-scan` FIRST — an id with no dir and no
+// scan entry has no text to match, and is only counted (`unscanned`).
+//
+// `dryRun: true` answers the same question and writes nothing.
+export async function POST(request: Request) {
+ return ops(request, ["slug", "match", "fields", "note", "dryRun"], async (body) => {
+ const slug = reqSlug(body, "slug");
+ const pattern = reqString(body, "match");
+ const raw = body.fields;
+ let fields: ("title" | "description")[] | undefined;
+ if (raw !== undefined) {
+ if (
+ !Array.isArray(raw) ||
+ raw.length === 0 ||
+ raw.some((f) => f !== "title" && f !== "description")
+ ) {
+ throw new OpsInputError(
+ '"fields" must be a non-empty array of "title" and/or "description"',
+ );
+ }
+ fields = raw as ("title" | "description")[];
+ }
+ const result = await keepVideosAction({
+ slug,
+ pattern,
+ fields,
+ note: optString(body, "note"),
+ dryRun: optBool(body, "dryRun"),
+ });
+ if (!result.ok) return opsFail(result.error);
+ return NextResponse.json({ ok: true, ...result.result });
+ });
+}
diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts
@@ -718,6 +718,79 @@ test("tag-videos refuses a traversing slug, a bad op and an unknown key", async
expect(await pathExists("test-transcripts/tags.json")).toBe(false);
});
+// --- keep-videos: the bulk do-not-clean toggle --------------------------------
+//
+// The marker is the video page's own do-not-clean.json; what this pins is the
+// loop around it — the download filter's matcher over title + description, a
+// dry run that writes nothing, and a scan-only match that is REPORTED rather
+// than given a directory it was never downloaded into.
+test("keep-videos marks the matching video only, after a dry run that writes nothing", async ({
+ request,
+}) => {
+ await resetData("one-youtube-channel-with-data");
+ const slug = "test-youtube";
+ const kept = "20240102_keepme12345";
+ const other = "20240101_test1234567"; // "Synthetic Test Video" — no needle
+ const channel = `test-transcripts/channels/${slug}`;
+ await mkdir(resolvePath(`${channel}/data/${kept}`), { recursive: true });
+ await writeFile(
+ resolvePath(`${channel}/data/${kept}/metadata.info.json`),
+ JSON.stringify({ id: kept, title: "Reacting to THEQUARTERING", description: "" }),
+ );
+ await writeFile(
+ resolvePath(`${channel}/metadata-scan.json`),
+ JSON.stringify({
+ version: 1,
+ updatedAt: "",
+ lastRun: null,
+ errors: {},
+ entries: {
+ scanonly123: {
+ title: "Plain",
+ description: "with TheQuartering",
+ uploadDate: "20240103",
+ scannedAt: "2026-09-25T00:00:00.000Z",
+ },
+ },
+ }),
+ );
+ const marker = (id: string) => `${channel}/data/${id}/do-not-clean.json`;
+
+ type KeepBody = OpsResponse & {
+ matched?: { id: string; marked: boolean }[];
+ marked?: number;
+ notDownloaded?: string[];
+ };
+ const dry = await ops(request, "keep-videos", {
+ slug,
+ match: "thequartering",
+ dryRun: true,
+ });
+ expect(dry.status, JSON.stringify(dry.body)).toBe(200);
+ const dryBody = dry.body as KeepBody;
+ expect(dryBody.matched?.filter((m) => m.id === kept)).toHaveLength(1);
+ expect(dryBody.marked).toBe(0);
+ expect(await pathExists(marker(kept))).toBe(false);
+
+ const real = await ops(request, "keep-videos", { slug, match: "thequartering" });
+ expect(real.status, JSON.stringify(real.body)).toBe(200);
+ const body = real.body as KeepBody;
+ expect(body.marked).toBe(1);
+ expect(body.notDownloaded).toEqual(["scanonly123"]);
+ expect(await pathExists(marker(kept))).toBe(true);
+ expect(await pathExists(marker(other))).toBe(false);
+ // A scan-only match gets no directory — that would read as "downloaded".
+ expect(await pathExists(`${channel}/data/scanonly123`)).toBe(false);
+
+ const unknown = await ops(request, "keep-videos", { slug, match: "x", pattern: "x" });
+ expect(unknown.status).toBe(400);
+ expect(unknown.body.error).toContain("unknown key(s): pattern");
+
+ const badRegex = await ops(request, "keep-videos", { slug, match: "(" });
+ expect(badRegex.status).toBe(400);
+ expect(badRegex.body.error).toContain("invalid pattern");
+});
+
// --- the two build routes speak the same body ---------------------------------
//
// They did not. build-site took `siteIds` (a list) and build-deploy took
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -114,6 +114,9 @@ const ACTIONS = [
// of videos in ONE write.
"tags",
"tag-videos",
+ // Set the per-video do-not-clean marker on every video of a channel whose
+ // title/description matches a download-filter pattern ({slug, match}).
+ "keep-videos",
];
// The provenance a tag write from this CLI carries. Everything else ignores it.
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -90,6 +90,18 @@ test("the tag actions are registered and post to their routes", () => {
assert.match(usage(), /tags/);
});
+test("keep-videos is registered and posts to its route", () => {
+ const p = parseArgs([
+ "keep-videos",
+ "--json",
+ '{"slug":"c","match":"TheQuartering","dryRun":true}',
+ ]);
+ assert.equal(p.path, "/api/ops/keep-videos");
+ assert.deepEqual(p.body, { slug: "c", match: "TheQuartering", dryRun: true });
+ // Not a provenance-carrying write: the marker records a note, not a source.
+ assert.equal(p.defaultSource, undefined);
+});
+
// A FOUR-THOUSAND-ID BODY IS WRITTEN BY A SCRIPT, NOT TYPED. parseArgs stays
// pure — it reports the file it would read, and main() reads it — so this test
// needs no filesystem.