commit a876362bfbc5e25da67265d18f90631a961f6c6e
parent 265b55920af9ce9d52ff5b9ef23842cc5cbf0e0b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 25 Sep 2026 12:38:52 -0400
common: keepVideosMatching — the do-not-clean marker over a channel's title/description matches
The download filter's own compile + subject (title + "\n" + description),
text from metadata.info.json then the metadata-scan store; matches with no
data/<id>/ are reported (notDownloaded), never created; ids with no text are
counted (unscanned). Media guard first: an unmounted drive is not an empty
channel.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
2 files changed, 470 insertions(+), 0 deletions(-)
diff --git a/common/controller/keepVideosMatching.test.ts b/common/controller/keepVideosMatching.test.ts
@@ -0,0 +1,223 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, rm, stat, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import { loadDoNotClean, setDoNotClean } from "../lib/doNotClean-server";
+import { KeepVideosError, keepVideosMatching } from "./keepVideosMatching";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/keepVideosMatching.test.ts
+
+const SLUG = "ch";
+
+type Seed = {
+ // id -> [title, description] written to data/<id>/metadata.info.json
+ dirs?: Record<string, [string, string]>;
+ // id -> [title, description] written to metadata-scan.json only
+ scanned?: Record<string, [string, string]>;
+ // bare ids written to the playlist file
+ playlist?: string[];
+};
+
+async function withChannel(
+ seed: Seed,
+ fn: (paths: Paths, channelDir: string) => Promise<void>,
+): Promise<void> {
+ const root = await mkdtemp(path.join(tmpdir(), "ttb-keep-videos-"));
+ const paths = { channelsDir: path.join(root, "channels") } as Paths;
+ const channelDir = path.join(paths.channelsDir, SLUG);
+ try {
+ await mkdir(path.join(channelDir, "data"), { recursive: true });
+ await writeFile(
+ path.join(channelDir, "config.json"),
+ JSON.stringify({ name: "Ch", url: "https://www.youtube.com/@ch" }),
+ );
+ for (const [id, [title, description]] of Object.entries(seed.dirs ?? {})) {
+ const dir = path.join(channelDir, "data", id);
+ await mkdir(dir, { recursive: true });
+ await writeFile(
+ path.join(dir, "metadata.info.json"),
+ JSON.stringify({ id, title, description }),
+ );
+ }
+ const entries: Record<string, unknown> = {};
+ for (const [id, [title, description]] of Object.entries(seed.scanned ?? {})) {
+ entries[id] = {
+ title,
+ description,
+ uploadDate: "20240101",
+ scannedAt: "2026-09-25T00:00:00.000Z",
+ };
+ }
+ await writeFile(
+ path.join(channelDir, "metadata-scan.json"),
+ JSON.stringify({ version: 1, updatedAt: "", lastRun: null, entries, errors: {} }),
+ );
+ await writeFile(
+ path.join(channelDir, "playlist"),
+ (seed.playlist ?? [])
+ .map((id) => `https://www.youtube.com/watch?v=${id}`)
+ .join("\n"),
+ );
+ await fn(paths, channelDir);
+ } finally {
+ await rm(root, { recursive: true, force: true });
+ }
+}
+
+const dirOf = (channelDir: string, id: string) =>
+ path.join(channelDir, "data", id);
+
+test("a title match and a description-only match are both marked, case-insensitively", async () => {
+ await withChannel(
+ {
+ dirs: {
+ aaaaaaaaaaa: ["Reacting to THEQUARTERING", "nothing here"],
+ bbbbbbbbbbb: ["Some other video", "featuring thequartering at 10:00"],
+ ccccccccccc: ["Unrelated", "unrelated"],
+ },
+ },
+ async (paths, channelDir) => {
+ const r = await keepVideosMatching({
+ paths,
+ channelSlug: SLUG,
+ pattern: "TheQuartering",
+ });
+ assert.equal(r.considered, 3);
+ assert.equal(r.marked, 2);
+ assert.deepEqual(
+ r.matched.map((m) => [m.id, m.marked, m.downloaded]),
+ [
+ ["aaaaaaaaaaa", true, true],
+ ["bbbbbbbbbbb", true, true],
+ ],
+ );
+ const rec = await loadDoNotClean(dirOf(channelDir, "aaaaaaaaaaa"));
+ assert.equal(rec?.note, "keep-videos: matched /TheQuartering/i");
+ assert.equal(await loadDoNotClean(dirOf(channelDir, "ccccccccccc")), null);
+ },
+ );
+});
+
+test("an already-kept video is counted and not rewritten", async () => {
+ await withChannel(
+ { dirs: { aaaaaaaaaaa: ["TheQuartering live", ""] } },
+ async (paths, channelDir) => {
+ const dir = dirOf(channelDir, "aaaaaaaaaaa");
+ await setDoNotClean(dir, true, "by hand");
+ const markerFile = path.join(dir, "do-not-clean.json");
+ const before = (await stat(markerFile)).mtimeMs;
+ await new Promise((r) => setTimeout(r, 20));
+ const r = await keepVideosMatching({
+ paths,
+ channelSlug: SLUG,
+ pattern: "thequartering",
+ });
+ assert.equal(r.marked, 0);
+ assert.equal(r.alreadyKept, 1);
+ assert.deepEqual(r.matched[0], {
+ id: "aaaaaaaaaaa",
+ title: "TheQuartering live",
+ downloaded: true,
+ alreadyKept: true,
+ marked: false,
+ });
+ assert.equal((await stat(markerFile)).mtimeMs, before);
+ assert.equal((await loadDoNotClean(dir))?.note, "by hand");
+ },
+ );
+});
+
+test("a scan-only match is reported notDownloaded and no dir is created; unscanned counts the gap", async () => {
+ await withChannel(
+ {
+ scanned: { sssssssssss: ["TheQuartering clip", ""] },
+ playlist: ["sssssssssss", "uuuuuuuuuuu", "vvvvvvvvvvv"],
+ },
+ async (paths, channelDir) => {
+ const r = await keepVideosMatching({
+ paths,
+ channelSlug: SLUG,
+ pattern: "thequartering",
+ });
+ assert.deepEqual(r.notDownloaded, ["sssssssssss"]);
+ assert.equal(r.marked, 0);
+ assert.equal(r.matched[0].downloaded, false);
+ assert.equal(r.unscanned, 2);
+ await assert.rejects(stat(dirOf(channelDir, "sssssssssss")), {
+ code: "ENOENT",
+ });
+ },
+ );
+});
+
+test("a dry run reports the matches and writes nothing", async () => {
+ await withChannel(
+ { dirs: { aaaaaaaaaaa: ["TheQuartering", ""] } },
+ async (paths, channelDir) => {
+ const r = await keepVideosMatching({
+ paths,
+ channelSlug: SLUG,
+ pattern: "thequartering",
+ dryRun: true,
+ });
+ assert.equal(r.dryRun, true);
+ assert.equal(r.matched.length, 1);
+ assert.equal(r.matched[0].marked, false);
+ assert.equal(r.marked, 0);
+ assert.equal(await loadDoNotClean(dirOf(channelDir, "aaaaaaaaaaa")), null);
+ },
+ );
+});
+
+test('fields: ["title"] ignores a description hit', async () => {
+ await withChannel(
+ {
+ dirs: {
+ aaaaaaaaaaa: ["Plain title", "TheQuartering in the description"],
+ bbbbbbbbbbb: ["TheQuartering in the title", ""],
+ },
+ },
+ async (paths) => {
+ const r = await keepVideosMatching({
+ paths,
+ channelSlug: SLUG,
+ pattern: "thequartering",
+ fields: ["title"],
+ });
+ assert.deepEqual(r.fields, ["title"]);
+ assert.deepEqual(
+ r.matched.map((m) => m.id),
+ ["bbbbbbbbbbb"],
+ );
+ },
+ );
+});
+
+test("a bad pattern, an unknown field and an unknown channel throw the typed error", async () => {
+ await withChannel({ dirs: {} }, async (paths) => {
+ await assert.rejects(
+ keepVideosMatching({ paths, channelSlug: SLUG, pattern: "(" }),
+ (e) => e instanceof KeepVideosError && /invalid pattern/.test(e.message),
+ );
+ await assert.rejects(
+ keepVideosMatching({ paths, channelSlug: SLUG, pattern: "(\\w+\\s?)*$" }),
+ (e) => e instanceof KeepVideosError && /nested quantifier/.test(e.message),
+ );
+ await assert.rejects(
+ keepVideosMatching({
+ paths,
+ channelSlug: SLUG,
+ pattern: "x",
+ fields: ["tags" as "title"],
+ }),
+ KeepVideosError,
+ );
+ await assert.rejects(
+ keepVideosMatching({ paths, channelSlug: "nope", pattern: "x" }),
+ (e) => e instanceof KeepVideosError && /not found/.test(e.message),
+ );
+ });
+});
diff --git a/common/controller/keepVideosMatching.ts b/common/controller/keepVideosMatching.ts
@@ -0,0 +1,247 @@
+import path from "node:path";
+import { readFile } from "node:fs/promises";
+import type { Paths } from "../lib/paths";
+import { assertChannelMediaReachable } from "../lib/channelMedia";
+import { isDoNotClean, setDoNotClean } from "../lib/doNotClean-server";
+import {
+ classifyAgainstFilter,
+ compileDownloadFilter,
+ downloadFilterPatternProblem,
+} from "../lib/downloadFilters";
+import { loadRawMetadataFromDir } from "../lib/transcripts-server";
+import { extractVideoId } from "../lib/videoId";
+import { channelExists, readChannelConfig } from "./channels";
+import { listChannelVideoIds } from "./keptVideos";
+import { loadMetadataScan } from "./metadataScanStore";
+import { loadRoster } from "./rosterStore";
+
+// BULK "KEEP THE VIDEO": set the per-video do-not-clean marker on every video of
+// one channel whose title / description matches a pattern. The marker is the
+// same `data/<id>/do-not-clean.json` the video page's toggle writes
+// (lib/doNotClean-server.ts), and every cleaner already honours it — the clean
+// sweep, extra-format cleanup, wrong-format removal, the superseded-subs purge
+// and saved-video eviction. This module adds no new kind of protection; it is
+// the loop the toggle lacks.
+//
+// "MATCHES" MEANS WHAT THE DOWNLOAD FILTER MEANS BY IT. The pattern is compiled
+// by `compileDownloadFilter` (case-insensitive, the same source a channel's
+// `downloadFilter.include` holds) and tested by `classifyAgainstFilter` over
+// `downloadFilterText` — title + "\n" + description, the description capped
+// exactly as the filter caps it. So a pattern that keeps the right videos here
+// is one that would download the right ones there, and vice versa. There is no
+// second matcher. `fields` narrows the SUBJECT, not the matcher: a field left
+// out is passed as empty.
+//
+// TEXT COMES FROM TWO PLACES, NEVER A THIRD. A downloaded video's own
+// `metadata.info.json` first; otherwise the channel's metadata-scan store
+// (`metadata-scan.json`), which is how a scanned-but-undownloaded video has a
+// title at all. An id known only from `playlist` / `roster.json` has no text and
+// is COUNTED (`unscanned`) — run `metadata-scan` first to close that gap.
+//
+// A MATCH WITH NO `data/<id>/` IS REPORTED, NEVER CREATED. The marker lives in
+// the video's dir and `setDoNotClean` does not mkdir. Creating the dir here
+// would also break the metadata scan's invariant that it never makes one
+// (metadataScanStore.ts) — every enumerator reads "has a dir" as "was fetched".
+// `notDownloaded` lists them; `download-missing` takes the ids, and a re-run
+// then marks them.
+//
+// AN UNMOUNTED DRIVE IS NOT AN EMPTY CHANNEL. `listChannelVideoIds` swallows
+// ENOENT on data/, which on a relocated channel would turn every downloaded
+// match into `notDownloaded`. The media guard is asked first, and refuses.
+
+export const KEEP_VIDEOS_FIELDS = ["title", "description"] as const;
+export type KeepVideosField = (typeof KEEP_VIDEOS_FIELDS)[number];
+
+// Thrown for a refusal the CALLER can fix (a bad pattern, an unknown channel,
+// a bad field). The editor action turns it into its `{ ok: false, error }`, and
+// the ops route into a 400. Anything else is an I/O failure.
+export class KeepVideosError extends Error {
+ constructor(message: string) {
+ super(message);
+ this.name = "KeepVideosError";
+ }
+}
+
+export type KeepVideosOptions = {
+ paths: Paths;
+ channelSlug: string;
+ pattern: string;
+ fields?: readonly KeepVideosField[];
+ note?: string;
+ dryRun?: boolean;
+};
+
+export type KeepVideosMatch = {
+ id: string;
+ title: string;
+ // Has a data/<id>/ dir — the only kind that can carry the marker.
+ downloaded: boolean;
+ // Already carried a marker before this call; left untouched.
+ alreadyKept: boolean;
+ // This call wrote the marker (always false on a dry run).
+ marked: boolean;
+};
+
+export type KeepVideosResult = {
+ pattern: string;
+ fields: KeepVideosField[];
+ // Ids that had text to test: every data dir plus every scan entry.
+ considered: number;
+ matched: KeepVideosMatch[];
+ marked: number;
+ alreadyKept: number;
+ // Matched ids with no data/<id>/ — download them, then re-run.
+ notDownloaded: string[];
+ // Ids in playlist / roster.json with neither a dir nor a scan entry: the
+ // coverage gap no pattern can see into. `metadata-scan` closes it.
+ unscanned: number;
+ // Data dirs with no metadata.info.json and no scan entry — they have a dir
+ // but no text, so they could not be tested. Rare; counted, not guessed.
+ noMetadata: number;
+ dryRun: boolean;
+};
+
+function resolveFields(
+ fields: readonly KeepVideosField[] | undefined,
+): KeepVideosField[] {
+ if (fields === undefined) return [...KEEP_VIDEOS_FIELDS];
+ if (fields.length === 0) {
+ throw new KeepVideosError(
+ `"fields" must name at least one of ${KEEP_VIDEOS_FIELDS.join(", ")}`,
+ );
+ }
+ for (const f of fields) {
+ if (!(KEEP_VIDEOS_FIELDS as readonly string[]).includes(f)) {
+ throw new KeepVideosError(
+ `"fields" may only contain ${KEEP_VIDEOS_FIELDS.join(", ")} (got "${f}")`,
+ );
+ }
+ }
+ return KEEP_VIDEOS_FIELDS.filter((f) => fields.includes(f));
+}
+
+async function readPlaylistIds(channelDir: string): Promise<string[]> {
+ const raw = await readFile(path.join(channelDir, "playlist"), "utf8").catch(
+ () => "",
+ );
+ const ids: string[] = [];
+ for (const line of raw.split("\n")) {
+ const url = line.trim();
+ if (!url) continue;
+ const id = extractVideoId(url);
+ if (id) ids.push(id);
+ }
+ return ids;
+}
+
+export async function keepVideosMatching(
+ opts: KeepVideosOptions,
+): Promise<KeepVideosResult> {
+ const { paths, channelSlug } = opts;
+ const pattern = opts.pattern.trim();
+ const dryRun = opts.dryRun === true;
+ const fields = resolveFields(opts.fields);
+
+ if (!pattern) throw new KeepVideosError('"match" must be a non-empty pattern');
+ // The download filter form's own guard: the pattern runs over every title
+ // and description on the channel, on the server's one thread.
+ const problem = downloadFilterPatternProblem(pattern);
+ if (problem) throw new KeepVideosError(`invalid pattern: ${problem}`);
+ const compiled = compileDownloadFilter({ include: pattern });
+ if (!compiled?.include) {
+ let why = "it does not parse";
+ try {
+ new RegExp(pattern, "i");
+ } catch (e) {
+ why = (e as Error).message;
+ }
+ throw new KeepVideosError(`invalid pattern /${pattern}/i: ${why}`);
+ }
+
+ if (!(await channelExists(paths, channelSlug))) {
+ throw new KeepVideosError(`Channel "${channelSlug}" not found`);
+ }
+ const config = await readChannelConfig(paths, channelSlug);
+ await assertChannelMediaReachable(paths, channelSlug, config);
+
+ const channelDir = path.join(paths.channelsDir, channelSlug);
+ const [dirIds, scan, roster, playlistIds] = await Promise.all([
+ listChannelVideoIds(paths, channelSlug),
+ loadMetadataScan(paths, channelSlug),
+ loadRoster(paths, channelSlug),
+ readPlaylistIds(channelDir),
+ ]);
+ const dirs = new Set(dirIds);
+ const ids = [...new Set([...dirIds, ...Object.keys(scan.entries)])].sort();
+
+ const matched: KeepVideosMatch[] = [];
+ const notDownloaded: string[] = [];
+ let marked = 0;
+ let alreadyKept = 0;
+ let noMetadata = 0;
+ let considered = 0;
+ const note = opts.note?.trim() || `keep-videos: matched /${pattern}/i`;
+
+ for (const id of ids) {
+ const downloaded = dirs.has(id);
+ const videoDir = path.join(channelDir, "data", id);
+ const meta = downloaded ? await loadRawMetadataFromDir(videoDir) : null;
+ const entry = scan.entries[id];
+ let title: string;
+ let description: string;
+ if (meta) {
+ title = meta.title ?? "";
+ description = meta.description ?? "";
+ } else if (entry) {
+ title = entry.title;
+ description = entry.description;
+ } else {
+ noMetadata++;
+ continue;
+ }
+ considered++;
+ const subject = {
+ title: fields.includes("title") ? title : "",
+ description: fields.includes("description") ? description : "",
+ };
+ if (classifyAgainstFilter(compiled, subject) !== "text") continue;
+
+ if (!downloaded) {
+ notDownloaded.push(id);
+ matched.push({ id, title, downloaded, alreadyKept: false, marked: false });
+ continue;
+ }
+ const kept = await isDoNotClean(videoDir);
+ if (kept) {
+ alreadyKept++;
+ } else if (!dryRun) {
+ await setDoNotClean(videoDir, true, note);
+ marked++;
+ }
+ matched.push({
+ id,
+ title,
+ downloaded,
+ alreadyKept: kept,
+ marked: !kept && !dryRun,
+ });
+ }
+
+ const known = new Set([...dirs, ...Object.keys(scan.entries)]);
+ const listed = new Set([...playlistIds, ...Object.keys(roster.entries)]);
+ let unscanned = 0;
+ for (const id of listed) if (!known.has(id)) unscanned++;
+
+ return {
+ pattern,
+ fields,
+ considered,
+ matched,
+ marked,
+ alreadyKept,
+ notDownloaded,
+ unscanned,
+ noMetadata,
+ dryRun,
+ };
+}