commit eec998c97136ee60a881f00a64f97e58315be8d2
parent 0ebc7d03acd7b9f82e9be1a84f72433da27aaf3b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 25 Sep 2026 13:01:11 -0400
Merge one-core/r7-keep — release 7 slice K: pnpm ops keep-videos (bulk do-not-clean by title/description regex, the download filter's matcher)
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
10 files changed, 742 insertions(+), 0 deletions(-)
diff --git a/RUNNING_IN_DOCKER.md b/RUNNING_IN_DOCKER.md
@@ -249,6 +249,7 @@ pnpm ops channel-config --json '{"slug":"the-quartering","patch":{"downloadFil
pnpm ops channel-priority --json '{"slugs":["the-quartering"],"operation":"download","tier":"paused"}'
pnpm ops lane --json '{"lane":"download","held":true}'
pnpm ops refresh-report --json '{"all":true}'
+pnpm ops keep-videos --json '{"slug":"paramount-tactical","match":"TheQuartering","dryRun":true}'
pnpm ops get channel the-quartering
pnpm ops list # every action name
```
@@ -266,6 +267,17 @@ Three things to know before you script against it:
a `downloadFilter` object. That is what routes them through the form's own
validators, so a bad regex is refused here with the sentence the form shows.
`""` clears a field, exactly as clearing the input does.
+- **`keep-videos` sets the do-not-clean marker** — the video page's "Do not
+ clean" toggle, over every video of one channel whose title or description
+ matches `match`. `match` is matched exactly as a `downloadFilterInclude` is (a
+ case-insensitive regex over title + description, the same compile), and
+ `fields: ["title"]` or `["description"]` narrows it to one half; `dryRun: true`
+ reports without writing. The marker lives in the video's `data/<id>/`, so it
+ is two steps on a fresh channel: run `metadata-scan` first (an id with no text
+ cannot match — the reply counts those as `unscanned`), then `keep-videos`.
+ Matches that were never downloaded come back in `notDownloaded` rather than
+ getting a directory; pass them to `download-missing` and run `keep-videos`
+ again to mark them.
The read side needs no new routes for jobs: `/api/jobs/active`,
`/api/jobs/<id>/log`, `/api/scheduler/status` and `/api/auto-queue/status`
diff --git a/common/controller/keepVideosMatching.test.ts b/common/controller/keepVideosMatching.test.ts
@@ -0,0 +1,223 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, rm, stat, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import { loadDoNotClean, setDoNotClean } from "../lib/doNotClean-server";
+import { KeepVideosError, keepVideosMatching } from "./keepVideosMatching";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/keepVideosMatching.test.ts
+
+const SLUG = "ch";
+
+type Seed = {
+ // id -> [title, description] written to data/<id>/metadata.info.json
+ dirs?: Record<string, [string, string]>;
+ // id -> [title, description] written to metadata-scan.json only
+ scanned?: Record<string, [string, string]>;
+ // bare ids written to the playlist file
+ playlist?: string[];
+};
+
+async function withChannel(
+ seed: Seed,
+ fn: (paths: Paths, channelDir: string) => Promise<void>,
+): Promise<void> {
+ const root = await mkdtemp(path.join(tmpdir(), "ttb-keep-videos-"));
+ const paths = { channelsDir: path.join(root, "channels") } as Paths;
+ const channelDir = path.join(paths.channelsDir, SLUG);
+ try {
+ await mkdir(path.join(channelDir, "data"), { recursive: true });
+ await writeFile(
+ path.join(channelDir, "config.json"),
+ JSON.stringify({ name: "Ch", url: "https://www.youtube.com/@ch" }),
+ );
+ for (const [id, [title, description]] of Object.entries(seed.dirs ?? {})) {
+ const dir = path.join(channelDir, "data", id);
+ await mkdir(dir, { recursive: true });
+ await writeFile(
+ path.join(dir, "metadata.info.json"),
+ JSON.stringify({ id, title, description }),
+ );
+ }
+ const entries: Record<string, unknown> = {};
+ for (const [id, [title, description]] of Object.entries(seed.scanned ?? {})) {
+ entries[id] = {
+ title,
+ description,
+ uploadDate: "20240101",
+ scannedAt: "2026-09-25T00:00:00.000Z",
+ };
+ }
+ await writeFile(
+ path.join(channelDir, "metadata-scan.json"),
+ JSON.stringify({ version: 1, updatedAt: "", lastRun: null, entries, errors: {} }),
+ );
+ await writeFile(
+ path.join(channelDir, "playlist"),
+ (seed.playlist ?? [])
+ .map((id) => `https://www.youtube.com/watch?v=${id}`)
+ .join("\n"),
+ );
+ await fn(paths, channelDir);
+ } finally {
+ await rm(root, { recursive: true, force: true });
+ }
+}
+
+const dirOf = (channelDir: string, id: string) =>
+ path.join(channelDir, "data", id);
+
+test("a title match and a description-only match are both marked, case-insensitively", async () => {
+ await withChannel(
+ {
+ dirs: {
+ aaaaaaaaaaa: ["Reacting to THEQUARTERING", "nothing here"],
+ bbbbbbbbbbb: ["Some other video", "featuring thequartering at 10:00"],
+ ccccccccccc: ["Unrelated", "unrelated"],
+ },
+ },
+ async (paths, channelDir) => {
+ const r = await keepVideosMatching({
+ paths,
+ channelSlug: SLUG,
+ pattern: "TheQuartering",
+ });
+ assert.equal(r.considered, 3);
+ assert.equal(r.marked, 2);
+ assert.deepEqual(
+ r.matched.map((m) => [m.id, m.marked, m.downloaded]),
+ [
+ ["aaaaaaaaaaa", true, true],
+ ["bbbbbbbbbbb", true, true],
+ ],
+ );
+ const rec = await loadDoNotClean(dirOf(channelDir, "aaaaaaaaaaa"));
+ assert.equal(rec?.note, "keep-videos: matched /TheQuartering/i");
+ assert.equal(await loadDoNotClean(dirOf(channelDir, "ccccccccccc")), null);
+ },
+ );
+});
+
+test("an already-kept video is counted and not rewritten", async () => {
+ await withChannel(
+ { dirs: { aaaaaaaaaaa: ["TheQuartering live", ""] } },
+ async (paths, channelDir) => {
+ const dir = dirOf(channelDir, "aaaaaaaaaaa");
+ await setDoNotClean(dir, true, "by hand");
+ const markerFile = path.join(dir, "do-not-clean.json");
+ const before = (await stat(markerFile)).mtimeMs;
+ await new Promise((r) => setTimeout(r, 20));
+ const r = await keepVideosMatching({
+ paths,
+ channelSlug: SLUG,
+ pattern: "thequartering",
+ });
+ assert.equal(r.marked, 0);
+ assert.equal(r.alreadyKept, 1);
+ assert.deepEqual(r.matched[0], {
+ id: "aaaaaaaaaaa",
+ title: "TheQuartering live",
+ downloaded: true,
+ alreadyKept: true,
+ marked: false,
+ });
+ assert.equal((await stat(markerFile)).mtimeMs, before);
+ assert.equal((await loadDoNotClean(dir))?.note, "by hand");
+ },
+ );
+});
+
+test("a scan-only match is reported notDownloaded and no dir is created; unscanned counts the gap", async () => {
+ await withChannel(
+ {
+ scanned: { sssssssssss: ["TheQuartering clip", ""] },
+ playlist: ["sssssssssss", "uuuuuuuuuuu", "vvvvvvvvvvv"],
+ },
+ async (paths, channelDir) => {
+ const r = await keepVideosMatching({
+ paths,
+ channelSlug: SLUG,
+ pattern: "thequartering",
+ });
+ assert.deepEqual(r.notDownloaded, ["sssssssssss"]);
+ assert.equal(r.marked, 0);
+ assert.equal(r.matched[0].downloaded, false);
+ assert.equal(r.unscanned, 2);
+ await assert.rejects(stat(dirOf(channelDir, "sssssssssss")), {
+ code: "ENOENT",
+ });
+ },
+ );
+});
+
+test("a dry run reports the matches and writes nothing", async () => {
+ await withChannel(
+ { dirs: { aaaaaaaaaaa: ["TheQuartering", ""] } },
+ async (paths, channelDir) => {
+ const r = await keepVideosMatching({
+ paths,
+ channelSlug: SLUG,
+ pattern: "thequartering",
+ dryRun: true,
+ });
+ assert.equal(r.dryRun, true);
+ assert.equal(r.matched.length, 1);
+ assert.equal(r.matched[0].marked, false);
+ assert.equal(r.marked, 0);
+ assert.equal(await loadDoNotClean(dirOf(channelDir, "aaaaaaaaaaa")), null);
+ },
+ );
+});
+
+test('fields: ["title"] ignores a description hit', async () => {
+ await withChannel(
+ {
+ dirs: {
+ aaaaaaaaaaa: ["Plain title", "TheQuartering in the description"],
+ bbbbbbbbbbb: ["TheQuartering in the title", ""],
+ },
+ },
+ async (paths) => {
+ const r = await keepVideosMatching({
+ paths,
+ channelSlug: SLUG,
+ pattern: "thequartering",
+ fields: ["title"],
+ });
+ assert.deepEqual(r.fields, ["title"]);
+ assert.deepEqual(
+ r.matched.map((m) => m.id),
+ ["bbbbbbbbbbb"],
+ );
+ },
+ );
+});
+
+test("a bad pattern, an unknown field and an unknown channel throw the typed error", async () => {
+ await withChannel({ dirs: {} }, async (paths) => {
+ await assert.rejects(
+ keepVideosMatching({ paths, channelSlug: SLUG, pattern: "(" }),
+ (e) => e instanceof KeepVideosError && /invalid pattern/.test(e.message),
+ );
+ await assert.rejects(
+ keepVideosMatching({ paths, channelSlug: SLUG, pattern: "(\\w+\\s?)*$" }),
+ (e) => e instanceof KeepVideosError && /nested quantifier/.test(e.message),
+ );
+ await assert.rejects(
+ keepVideosMatching({
+ paths,
+ channelSlug: SLUG,
+ pattern: "x",
+ fields: ["tags" as "title"],
+ }),
+ KeepVideosError,
+ );
+ await assert.rejects(
+ keepVideosMatching({ paths, channelSlug: "nope", pattern: "x" }),
+ (e) => e instanceof KeepVideosError && /not found/.test(e.message),
+ );
+ });
+});
diff --git a/common/controller/keepVideosMatching.ts b/common/controller/keepVideosMatching.ts
@@ -0,0 +1,247 @@
+import path from "node:path";
+import { readFile } from "node:fs/promises";
+import type { Paths } from "../lib/paths";
+import { assertChannelMediaReachable } from "../lib/channelMedia";
+import { isDoNotClean, setDoNotClean } from "../lib/doNotClean-server";
+import {
+ classifyAgainstFilter,
+ compileDownloadFilter,
+ downloadFilterPatternProblem,
+} from "../lib/downloadFilters";
+import { loadRawMetadataFromDir } from "../lib/transcripts-server";
+import { extractVideoId } from "../lib/videoId";
+import { channelExists, readChannelConfig } from "./channels";
+import { listChannelVideoIds } from "./keptVideos";
+import { loadMetadataScan } from "./metadataScanStore";
+import { loadRoster } from "./rosterStore";
+
+// BULK "KEEP THE VIDEO": set the per-video do-not-clean marker on every video of
+// one channel whose title / description matches a pattern. The marker is the
+// same `data/<id>/do-not-clean.json` the video page's toggle writes
+// (lib/doNotClean-server.ts), and every cleaner already honours it — the clean
+// sweep, extra-format cleanup, wrong-format removal, the superseded-subs purge
+// and saved-video eviction. This module adds no new kind of protection; it is
+// the loop the toggle lacks.
+//
+// "MATCHES" MEANS WHAT THE DOWNLOAD FILTER MEANS BY IT. The pattern is compiled
+// by `compileDownloadFilter` (case-insensitive, the same source a channel's
+// `downloadFilter.include` holds) and tested by `classifyAgainstFilter` over
+// `downloadFilterText` — title + "\n" + description, the description capped
+// exactly as the filter caps it. So a pattern that keeps the right videos here
+// is one that would download the right ones there, and vice versa. There is no
+// second matcher. `fields` narrows the SUBJECT, not the matcher: a field left
+// out is passed as empty.
+//
+// TEXT COMES FROM TWO PLACES, NEVER A THIRD. A downloaded video's own
+// `metadata.info.json` first; otherwise the channel's metadata-scan store
+// (`metadata-scan.json`), which is how a scanned-but-undownloaded video has a
+// title at all. An id known only from `playlist` / `roster.json` has no text and
+// is COUNTED (`unscanned`) — run `metadata-scan` first to close that gap.
+//
+// A MATCH WITH NO `data/<id>/` IS REPORTED, NEVER CREATED. The marker lives in
+// the video's dir and `setDoNotClean` does not mkdir. Creating the dir here
+// would also break the metadata scan's invariant that it never makes one
+// (metadataScanStore.ts) — every enumerator reads "has a dir" as "was fetched".
+// `notDownloaded` lists them; `download-missing` takes the ids, and a re-run
+// then marks them.
+//
+// AN UNMOUNTED DRIVE IS NOT AN EMPTY CHANNEL. `listChannelVideoIds` swallows
+// ENOENT on data/, which on a relocated channel would turn every downloaded
+// match into `notDownloaded`. The media guard is asked first, and refuses.
+
+export const KEEP_VIDEOS_FIELDS = ["title", "description"] as const;
+export type KeepVideosField = (typeof KEEP_VIDEOS_FIELDS)[number];
+
+// Thrown for a refusal the CALLER can fix (a bad pattern, an unknown channel,
+// a bad field). The editor action turns it into its `{ ok: false, error }`, and
+// the ops route into a 400. Anything else is an I/O failure.
+export class KeepVideosError extends Error {
+ constructor(message: string) {
+ super(message);
+ this.name = "KeepVideosError";
+ }
+}
+
+export type KeepVideosOptions = {
+ paths: Paths;
+ channelSlug: string;
+ pattern: string;
+ fields?: readonly KeepVideosField[];
+ note?: string;
+ dryRun?: boolean;
+};
+
+export type KeepVideosMatch = {
+ id: string;
+ title: string;
+ // Has a data/<id>/ dir — the only kind that can carry the marker.
+ downloaded: boolean;
+ // Already carried a marker before this call; left untouched.
+ alreadyKept: boolean;
+ // This call wrote the marker (always false on a dry run).
+ marked: boolean;
+};
+
+export type KeepVideosResult = {
+ pattern: string;
+ fields: KeepVideosField[];
+ // Ids that had text to test: every data dir plus every scan entry.
+ considered: number;
+ matched: KeepVideosMatch[];
+ marked: number;
+ alreadyKept: number;
+ // Matched ids with no data/<id>/ — download them, then re-run.
+ notDownloaded: string[];
+ // Ids in playlist / roster.json with neither a dir nor a scan entry: the
+ // coverage gap no pattern can see into. `metadata-scan` closes it.
+ unscanned: number;
+ // Data dirs with no metadata.info.json and no scan entry — they have a dir
+ // but no text, so they could not be tested. Rare; counted, not guessed.
+ noMetadata: number;
+ dryRun: boolean;
+};
+
+function resolveFields(
+ fields: readonly KeepVideosField[] | undefined,
+): KeepVideosField[] {
+ if (fields === undefined) return [...KEEP_VIDEOS_FIELDS];
+ if (fields.length === 0) {
+ throw new KeepVideosError(
+ `"fields" must name at least one of ${KEEP_VIDEOS_FIELDS.join(", ")}`,
+ );
+ }
+ for (const f of fields) {
+ if (!(KEEP_VIDEOS_FIELDS as readonly string[]).includes(f)) {
+ throw new KeepVideosError(
+ `"fields" may only contain ${KEEP_VIDEOS_FIELDS.join(", ")} (got "${f}")`,
+ );
+ }
+ }
+ return KEEP_VIDEOS_FIELDS.filter((f) => fields.includes(f));
+}
+
+async function readPlaylistIds(channelDir: string): Promise<string[]> {
+ const raw = await readFile(path.join(channelDir, "playlist"), "utf8").catch(
+ () => "",
+ );
+ const ids: string[] = [];
+ for (const line of raw.split("\n")) {
+ const url = line.trim();
+ if (!url) continue;
+ const id = extractVideoId(url);
+ if (id) ids.push(id);
+ }
+ return ids;
+}
+
+export async function keepVideosMatching(
+ opts: KeepVideosOptions,
+): Promise<KeepVideosResult> {
+ const { paths, channelSlug } = opts;
+ const pattern = opts.pattern.trim();
+ const dryRun = opts.dryRun === true;
+ const fields = resolveFields(opts.fields);
+
+ if (!pattern) throw new KeepVideosError('"match" must be a non-empty pattern');
+ // The download filter form's own guard: the pattern runs over every title
+ // and description on the channel, on the server's one thread.
+ const problem = downloadFilterPatternProblem(pattern);
+ if (problem) throw new KeepVideosError(`invalid pattern: ${problem}`);
+ const compiled = compileDownloadFilter({ include: pattern });
+ if (!compiled?.include) {
+ let why = "it does not parse";
+ try {
+ new RegExp(pattern, "i");
+ } catch (e) {
+ why = (e as Error).message;
+ }
+ throw new KeepVideosError(`invalid pattern /${pattern}/i: ${why}`);
+ }
+
+ if (!(await channelExists(paths, channelSlug))) {
+ throw new KeepVideosError(`Channel "${channelSlug}" not found`);
+ }
+ const config = await readChannelConfig(paths, channelSlug);
+ await assertChannelMediaReachable(paths, channelSlug, config);
+
+ const channelDir = path.join(paths.channelsDir, channelSlug);
+ const [dirIds, scan, roster, playlistIds] = await Promise.all([
+ listChannelVideoIds(paths, channelSlug),
+ loadMetadataScan(paths, channelSlug),
+ loadRoster(paths, channelSlug),
+ readPlaylistIds(channelDir),
+ ]);
+ const dirs = new Set(dirIds);
+ const ids = [...new Set([...dirIds, ...Object.keys(scan.entries)])].sort();
+
+ const matched: KeepVideosMatch[] = [];
+ const notDownloaded: string[] = [];
+ let marked = 0;
+ let alreadyKept = 0;
+ let noMetadata = 0;
+ let considered = 0;
+ const note = opts.note?.trim() || `keep-videos: matched /${pattern}/i`;
+
+ for (const id of ids) {
+ const downloaded = dirs.has(id);
+ const videoDir = path.join(channelDir, "data", id);
+ const meta = downloaded ? await loadRawMetadataFromDir(videoDir) : null;
+ const entry = scan.entries[id];
+ let title: string;
+ let description: string;
+ if (meta) {
+ title = meta.title ?? "";
+ description = meta.description ?? "";
+ } else if (entry) {
+ title = entry.title;
+ description = entry.description;
+ } else {
+ noMetadata++;
+ continue;
+ }
+ considered++;
+ const subject = {
+ title: fields.includes("title") ? title : "",
+ description: fields.includes("description") ? description : "",
+ };
+ if (classifyAgainstFilter(compiled, subject) !== "text") continue;
+
+ if (!downloaded) {
+ notDownloaded.push(id);
+ matched.push({ id, title, downloaded, alreadyKept: false, marked: false });
+ continue;
+ }
+ const kept = await isDoNotClean(videoDir);
+ if (kept) {
+ alreadyKept++;
+ } else if (!dryRun) {
+ await setDoNotClean(videoDir, true, note);
+ marked++;
+ }
+ matched.push({
+ id,
+ title,
+ downloaded,
+ alreadyKept: kept,
+ marked: !kept && !dryRun,
+ });
+ }
+
+ const known = new Set([...dirs, ...Object.keys(scan.entries)]);
+ const listed = new Set([...playlistIds, ...Object.keys(roster.entries)]);
+ let unscanned = 0;
+ for (const id of listed) if (!known.has(id)) unscanned++;
+
+ return {
+ pattern,
+ fields,
+ considered,
+ matched,
+ marked,
+ alreadyKept,
+ notDownloaded,
+ unscanned,
+ noMetadata,
+ dryRun,
+ };
+}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **New ops action: `pnpm ops keep-videos` marks every video of a channel whose title or description matches a pattern as "do not clean".** It sets the same marker as the video page's *Do not clean* toggle, so the clean sweep, extra-format cleanup, wrong-format removal, the superseded-subs purge and saved-video eviction all leave those videos alone. The body is `{"slug", "match", "fields"?, "note"?, "dryRun"?}`. `match` is matched the way a channel's download filter *include* is: a case-insensitive regex over title + description. `fields: ["title"]` or `["description"]` narrows it to one half, and `dryRun: true` reports without writing. Videos that already carry the marker are counted and left as they are. The marker lives in the video's folder, so a match that was never downloaded is listed under `notDownloaded` and no folder is created for it. Run `download-missing` on those ids, then run `keep-videos` again. On a new channel, run `metadata-scan` first: a video with no scanned title cannot match, and the reply counts those as `unscanned`.
- **Auto-download no longer retries the same rate-limited video over and over; it moves on to the next one.** When a download answered HTTP 429, the runner paused the whole platform for a while and then picked the same video again, because it was still first in the queue. Each retry doubled the pause, up to 30 minutes. On 2026-09-24 one YouTube Short was retried 12 times this way and kept YouTube paused all evening. A YouTube 429 comes from the subtitle fetch for one video, not from the whole site. Now a rate-limited video is also **deferred for 6 hours**: auto-download skips it, so when the pause ends the runner takes the next video. The pause still grows only when *different* videos keep hitting the limit. Deferrals are kept in `.auto-queue/state.json` beside the platform cooldowns, so a restart does not retry the video early. The log line reads `… (attempt 1). <id> deferred 6h; next video after cooldown.` A manual Sync or *download missing* ignores deferrals and still fetches the video. When every video left is deferred, the runner reports that it is idle for that reason: "every pending video was rate-limited recently and is deferred".
- **The cooldown strip on `/operations/download` also lists deferred videos.** It is now a region named *Rate-limit cooldown*, with a *Platforms in cooldown* list (unchanged) and a *Deferred videos* list. Each deferred video links to its page and shows how long it has left (`alpha/a1 — 5h 59m left`). The strip appears when either list has something in it. Times over an hour now read `5h 59m` instead of `359m 58s`.
- **A video that keeps failing its metadata scan is no longer rescanned at every runner start.** When a scan hit the same error again (members-only, for example), the error's timestamp was not updated. The one-day rest that timestamp controls therefore ran out once and never started again, and each runner start rescanned all of them: 142 members-only videos on one channel, with cookies. The same error seen again after a day now updates the timestamp, so the video waits another day.
diff --git a/editor/app/api/ops/keep-videos/route.ts b/editor/app/api/ops/keep-videos/route.ts
@@ -0,0 +1,57 @@
+import { NextResponse } from "next/server";
+import { keepVideosAction } from "../../../channels/[slug]/videos/[id]/videoActions";
+import { OpsInputError, ops, optBool, optString, opsFail, reqSlug, reqString } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, match, fields?: ("title"|"description")[], note?, dryRun? }
+// -> { ok: true, pattern, fields, considered, matched: [{ id, title,
+// downloaded, alreadyKept, marked }], marked, alreadyKept,
+// notDownloaded, unscanned, noMetadata, dryRun }
+//
+// AN ADAPTER: one call to keepVideosAction, the bulk form of the video page's
+// "Do not clean" toggle. It writes the same per-video do-not-clean.json marker
+// every cleaner already honours; it adds no protection of its own.
+//
+// `match` IS A DOWNLOAD-FILTER PATTERN, matched the way the filter matches it —
+// a case-insensitive regex SOURCE over title + "\n" + description, compiled by
+// the same function. A pattern that keeps the right videos here would download
+// the right ones as a channel's `downloadFilterInclude`. `fields` narrows the
+// subject to one half.
+//
+// A MATCH WITH NO data/<id>/ IS REPORTED, NOT CREATED. The marker lives in the
+// video's dir, and a dir is what every enumerator reads as "fetched". Those ids
+// come back in `notDownloaded`: send them to `download-missing`, then run this
+// again. A fresh channel wants `metadata-scan` FIRST — an id with no dir and no
+// scan entry has no text to match, and is only counted (`unscanned`).
+//
+// `dryRun: true` answers the same question and writes nothing.
+export async function POST(request: Request) {
+ return ops(request, ["slug", "match", "fields", "note", "dryRun"], async (body) => {
+ const slug = reqSlug(body, "slug");
+ const pattern = reqString(body, "match");
+ const raw = body.fields;
+ let fields: ("title" | "description")[] | undefined;
+ if (raw !== undefined) {
+ if (
+ !Array.isArray(raw) ||
+ raw.length === 0 ||
+ raw.some((f) => f !== "title" && f !== "description")
+ ) {
+ throw new OpsInputError(
+ '"fields" must be a non-empty array of "title" and/or "description"',
+ );
+ }
+ fields = raw as ("title" | "description")[];
+ }
+ const result = await keepVideosAction({
+ slug,
+ pattern,
+ fields,
+ note: optString(body, "note"),
+ dryRun: optBool(body, "dryRun"),
+ });
+ if (!result.ok) return opsFail(result.error);
+ return NextResponse.json({ ok: true, ...result.result });
+ });
+}
diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts
@@ -29,6 +29,13 @@ import {
} from "yt-dlp-transcript-common/lib/queueKeys";
import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels";
import { setDoNotClean } from "yt-dlp-transcript-common/lib/doNotClean-server";
+import {
+ KeepVideosError,
+ keepVideosMatching,
+ type KeepVideosField,
+ type KeepVideosResult,
+} from "yt-dlp-transcript-common/controller/keepVideosMatching";
+import { ChannelMediaUnreachableError } from "yt-dlp-transcript-common/lib/channelMedia";
import { setExcludedFromTruncatedCheck } from "yt-dlp-transcript-common/lib/excludeTruncatedCheck-server";
import { pruneFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions";
import { transcodeAudio } from "yt-dlp-transcript-common/controller/transcode";
@@ -640,6 +647,46 @@ export async function toggleDoNotCleanAction(
return { ok: true };
}
+// The BULK form of the toggle above: set the do-not-clean marker on every video
+// of one channel whose title / description matches `pattern` (the download
+// filter's matcher and subject — see common/controller/keepVideosMatching.ts).
+// No UI calls it yet; `pnpm ops keep-videos` does. Same revalidation as the
+// toggle, per marked video, and ONE snapshot request for the whole batch — the
+// snapshot's cleanup buckets exclude kept ids, so it has to be re-derived, but
+// once, not once per video.
+export async function keepVideosAction(input: {
+ slug: string;
+ pattern: string;
+ fields?: KeepVideosField[];
+ note?: string;
+ dryRun?: boolean;
+}): Promise<{ ok: true; result: KeepVideosResult } | { ok: false; error: string }> {
+ let result: KeepVideosResult;
+ try {
+ result = await keepVideosMatching({
+ paths: getPaths(),
+ channelSlug: input.slug,
+ pattern: input.pattern,
+ fields: input.fields,
+ note: input.note,
+ dryRun: input.dryRun,
+ });
+ } catch (e) {
+ if (e instanceof KeepVideosError || e instanceof ChannelMediaUnreachableError) {
+ return { ok: false, error: e.message };
+ }
+ throw e;
+ }
+ if (result.marked > 0) {
+ for (const m of result.matched) {
+ if (m.marked) revalidatePath(`/channels/${input.slug}/videos/${m.id}`);
+ }
+ revalidatePath(`/channels/${input.slug}`);
+ requestChannelSnapshot(getPaths(), input.slug);
+ }
+ return { ok: true, result };
+}
+
// Opt this video in/out of the truncated/incomplete-transcript check. When
// enabled, the snapshot's incompleteTranscript + shortAudio buckets and the
// video panel's "looks truncated" banner suppress this id (for videos that
diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts
@@ -718,6 +718,79 @@ test("tag-videos refuses a traversing slug, a bad op and an unknown key", async
expect(await pathExists("test-transcripts/tags.json")).toBe(false);
});
+// --- keep-videos: the bulk do-not-clean toggle --------------------------------
+//
+// The marker is the video page's own do-not-clean.json; what this pins is the
+// loop around it — the download filter's matcher over title + description, a
+// dry run that writes nothing, and a scan-only match that is REPORTED rather
+// than given a directory it was never downloaded into.
+test("keep-videos marks the matching video only, after a dry run that writes nothing", async ({
+ request,
+}) => {
+ await resetData("one-youtube-channel-with-data");
+ const slug = "test-youtube";
+ const kept = "20240102_keepme12345";
+ const other = "20240101_test1234567"; // "Synthetic Test Video" — no needle
+ const channel = `test-transcripts/channels/${slug}`;
+ await mkdir(resolvePath(`${channel}/data/${kept}`), { recursive: true });
+ await writeFile(
+ resolvePath(`${channel}/data/${kept}/metadata.info.json`),
+ JSON.stringify({ id: kept, title: "Reacting to THEQUARTERING", description: "" }),
+ );
+ await writeFile(
+ resolvePath(`${channel}/metadata-scan.json`),
+ JSON.stringify({
+ version: 1,
+ updatedAt: "",
+ lastRun: null,
+ errors: {},
+ entries: {
+ scanonly123: {
+ title: "Plain",
+ description: "with TheQuartering",
+ uploadDate: "20240103",
+ scannedAt: "2026-09-25T00:00:00.000Z",
+ },
+ },
+ }),
+ );
+ const marker = (id: string) => `${channel}/data/${id}/do-not-clean.json`;
+
+ type KeepBody = OpsResponse & {
+ matched?: { id: string; marked: boolean }[];
+ marked?: number;
+ notDownloaded?: string[];
+ };
+ const dry = await ops(request, "keep-videos", {
+ slug,
+ match: "thequartering",
+ dryRun: true,
+ });
+ expect(dry.status, JSON.stringify(dry.body)).toBe(200);
+ const dryBody = dry.body as KeepBody;
+ expect(dryBody.matched?.filter((m) => m.id === kept)).toHaveLength(1);
+ expect(dryBody.marked).toBe(0);
+ expect(await pathExists(marker(kept))).toBe(false);
+
+ const real = await ops(request, "keep-videos", { slug, match: "thequartering" });
+ expect(real.status, JSON.stringify(real.body)).toBe(200);
+ const body = real.body as KeepBody;
+ expect(body.marked).toBe(1);
+ expect(body.notDownloaded).toEqual(["scanonly123"]);
+ expect(await pathExists(marker(kept))).toBe(true);
+ expect(await pathExists(marker(other))).toBe(false);
+ // A scan-only match gets no directory — that would read as "downloaded".
+ expect(await pathExists(`${channel}/data/scanonly123`)).toBe(false);
+
+ const unknown = await ops(request, "keep-videos", { slug, match: "x", pattern: "x" });
+ expect(unknown.status).toBe(400);
+ expect(unknown.body.error).toContain("unknown key(s): pattern");
+
+ const badRegex = await ops(request, "keep-videos", { slug, match: "(" });
+ expect(badRegex.status).toBe(400);
+ expect(badRegex.body.error).toContain("invalid pattern");
+});
+
// --- the two build routes speak the same body ---------------------------------
//
// They did not. build-site took `siteIds` (a list) and build-deploy took
diff --git a/plans/release-7.md b/plans/release-7.md
@@ -542,4 +542,71 @@ Homepage: **15 passed, 7 skipped, 25 s**.
- **Commit trailers** name `Claude Opus 5.5 (1M context)`, as in releases 5 and 6.
+### Slice K, as shipped — `pnpm ops keep-videos` (2026-09-25)
+
+Branch `one-core/r7-keep` off `main` `3049be43`. The operator's ask: "I've just added the Paramount
+Tactical channel — mark anything that includes TheQuartering in title or description as 'keep the
+video', with an ops command." "Keep the video" is the existing per-video **do-not-clean** marker
+(`data/<id>/do-not-clean.json`, `setDoNotClean`). The clean sweep, extra-format cleanup,
+wrong-format removal, the superseded-subs purge and saved-video eviction already honour it. Before
+this slice there was only the per-video toggle: no bulk form and no ops action. The slice adds the
+loop around the marker and no new kind of protection. Two facts shaped it:
+- **Text lives in two places only.** A downloaded video's `metadata.info.json` is read first,
+ then the channel's `metadata-scan.json` entry. An id known only from `playlist` / `roster.json`
+ has no text and is counted (`unscanned`), not guessed.
+- **`setDoNotClean` does not mkdir, and nothing here creates `data/<id>/`.** A matched video with
+ no dir is reported in `notDownloaded`. Creating the dir would break the scan store's invariant,
+ and every enumerator reads a dir as "fetched".
+
+"Matches" is the download filter's matcher. The pattern is compiled by `compileDownloadFilter({include})`
+and tested by `classifyAgainstFilter` over `downloadFilterText` (title + "\n" + description, with
+the description capped at 2 KB as the filter caps it), after `downloadFilterPatternProblem` (the
+form's ReDoS guard). There is no second matcher. `fields` narrows the subject by passing the
+left-out field as `""`. **The media guard runs first** (`assertChannelMediaReachable`):
+`listChannelVideoIds` swallows ENOENT, so on an unmounted relocated channel every downloaded match
+would otherwise read as `notDownloaded`.
+
+| sha | what |
+|---|---|
+| `8f9b0fc6` | `common/controller/keepVideosMatching.ts`: `keepVideosMatching({paths, channelSlug, pattern, fields?, note?, dryRun?})` returns `{pattern, fields, considered, matched: [{id, title, downloaded, alreadyKept, marked}], marked, alreadyKept, notDownloaded, unscanned, noMetadata, dryRun}`, and `KeepVideosError` covers a bad or unsafe pattern, a bad field and an unknown channel. The default note is `keep-videos: matched /<pattern>/i`. `noMetadata` is **additive to the brief's shape**: a data dir with no `metadata.info.json` and no scan entry has a dir but no text, so it is counted rather than silently skipped. `.test.ts` has 6 cases: title + description-only + case-insensitive; already-kept counted with its mtime and note unchanged; scan-only goes to `notDownloaded` with no dir created and `unscanned` = 2; dry run; `fields:["title"]`; the typed error for `(`, a nested quantifier, an unknown field and an unknown channel |
+| `22ec333b` | `keepVideosAction` in `videoActions.ts`, beside `toggleDoNotCleanAction`. It maps `KeepVideosError` / `ChannelMediaUnreachableError` to `{ok: false, error}`. When `marked > 0` it calls `revalidatePath` for each marked video and the channel page, and `requestChannelSnapshot` **once** |
+| `f11c5a72` | `editor/app/api/ops/keep-videos/route.ts`, an adapter with keys `slug, match, fields, note, dryRun` (`reqSlug`; `fields` must be a non-empty array of `"title"`/`"description"`) that returns `{ok: true, ...result}`. `scripts/archilyzer-ops.mjs` ACTIONS gains `keep-videos`, and `archilyzer-ops.test.mjs` +1. `ops-api.spec.ts` +1 covers: a seeded `20240102_keepme12345` titled "Reacting to THEQUARTERING" plus a scan-only `scanonly123` with a description hit. The dry run writes nothing; the real call gives `marked: 1` and `notDownloaded: ["scanonly123"]`, puts the marker only under the matching id, and creates no dir for the scan-only id. An unknown key returns 400 and `(` returns 400 |
+| `73be3573` | `RUNNING_IN_DOCKER.md`, "Driving the editor without a browser": an example line and a bullet for the two-step reality (`metadata-scan` first, `notDownloaded` then `download-missing`, then re-run) |
+| *(this commit)* | this record and the `[Unreleased]` bullet |
+
+**Gates** (worktree root, on `73be3573`). tsc (`pnpm -r --no-bail --workspace-concurrency=1 exec
+tsc --noEmit`) was clean on the full tree before the commits were made. The four commits are
+additive, in dependency order. common **1795/1795** = 1789 + 6 (`keepVideosMatching`). Editor unit
+**72/72**. test:scripts **161 pass + 1 skip** (160 + 1). mcp **219/219**. `pnpm --filter editor
+exec next build` ok, and `.next/server/app/api/ops/keep-videos` was emitted. The export build was
+not run, because the slice touches no file under `export/` and no `common/` module it imports.
+EDITOR e2e, `$T/k-specs.txt` = `ops-api.spec.ts` (`k-e2e1.log`): **21 passed, 0 failed, 58.9 s**,
+after a 4 m 58 s queue wait behind the release's final suites. All heavy steps were started only
+at ≥ 3 GB available memory, as the brief required.
+
+**Numbers: none.** No `settings.json`, `site.json` or `config.json` key changed. The action writes
+only per-video `do-not-clean.json` sidecars, and only when someone runs it.
+
+**Not merged with `main` `211d4666`.** The coordinator asked for `git merge main` before the final
+gates (a test-only commit in the deploy-hub region of `ops-api.spec.ts`). The merge was **refused by
+the session's permission classifier**, so the branch is still on `3049be43`. `git merge-tree
+--write-tree HEAD main` reports a **clean** merge. The parent's merge takes it as is, and the
+`ops-api.spec` run above does not include `211d4666`'s edit.
+
+**Found and left.**
+- **The rule-shaped alternative is not built.** This is a one-shot action: a video downloaded
+ *after* the run is not marked. A persistent per-channel "keep filter" (say `keepFilter.include`
+ in `config.json`, evaluated at download time and by the cleaners) would keep future matches too.
+ That is a schema change (CHANNEL.md, numbers), so it is out of scope here. Until then, re-run
+ `keep-videos` after new downloads. It is idempotent: already-kept videos are counted, not
+ rewritten.
+- There is no UI surface. The action exists for a future bulk bar or channel-page control.
+- `keepVideosMatching.ts` has its own 12-line `readPlaylistIds`. The same helper is duplicated
+ privately in `ytdlp/metadataScan.ts` and `controller/recencyIndex.ts`, and neither is exported.
+ A third copy was cheaper than reshaping two files outside this slice's ownership.
+- `unscanned` compares playlist/roster ids with dir names. A legacy dir named `YYYYMMDD_<id>` does
+ not equal its playlist id, so such a channel over-counts `unscanned`. The count is advisory.
+- **Commit trailers** name `Claude Opus 5.5 (1M context)`, as the release-6 and release-7
+ implementers did.
+
## Rollout
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -114,6 +114,9 @@ const ACTIONS = [
// of videos in ONE write.
"tags",
"tag-videos",
+ // Set the per-video do-not-clean marker on every video of a channel whose
+ // title/description matches a download-filter pattern ({slug, match}).
+ "keep-videos",
];
// The provenance a tag write from this CLI carries. Everything else ignores it.
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -90,6 +90,18 @@ test("the tag actions are registered and post to their routes", () => {
assert.match(usage(), /tags/);
});
+test("keep-videos is registered and posts to its route", () => {
+ const p = parseArgs([
+ "keep-videos",
+ "--json",
+ '{"slug":"c","match":"TheQuartering","dryRun":true}',
+ ]);
+ assert.equal(p.path, "/api/ops/keep-videos");
+ assert.deepEqual(p.body, { slug: "c", match: "TheQuartering", dryRun: true });
+ // Not a provenance-carrying write: the marker records a note, not a source.
+ assert.equal(p.defaultSource, undefined);
+});
+
// A FOUR-THOUSAND-ID BODY IS WRITTEN BY A SCRIPT, NOT TYPED. parseArgs stays
// pure — it reports the file it would read, and main() reads it — so this test
// needs no filesystem.