Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit eec998c97136ee60a881f00a64f97e58315be8d2
parent 0ebc7d03acd7b9f82e9be1a84f72433da27aaf3b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri, 25 Sep 2026 13:01:11 -0400

Merge one-core/r7-keep — release 7 slice K: pnpm ops keep-videos (bulk do-not-clean by title/description regex, the download filter's matcher)

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
MRUNNING_IN_DOCKER.md | 12++++++++++++
Acommon/controller/keepVideosMatching.test.ts | 223+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/keepVideosMatching.ts | 247+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/CHANGELOG.md | 1+
Aeditor/app/api/ops/keep-videos/route.ts | 57+++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 47+++++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/ops-api.spec.ts | 73+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mplans/release-7.md | 67+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mscripts/archilyzer-ops.mjs | 3+++
Mscripts/archilyzer-ops.test.mjs | 12++++++++++++
10 files changed, 742 insertions(+), 0 deletions(-)

diff --git a/RUNNING_IN_DOCKER.md b/RUNNING_IN_DOCKER.md @@ -249,6 +249,7 @@ pnpm ops channel-config --json '{"slug":"the-quartering","patch":{"downloadFil pnpm ops channel-priority --json '{"slugs":["the-quartering"],"operation":"download","tier":"paused"}' pnpm ops lane --json '{"lane":"download","held":true}' pnpm ops refresh-report --json '{"all":true}' +pnpm ops keep-videos --json '{"slug":"paramount-tactical","match":"TheQuartering","dryRun":true}' pnpm ops get channel the-quartering pnpm ops list # every action name ``` @@ -266,6 +267,17 @@ Three things to know before you script against it: a `downloadFilter` object. That is what routes them through the form's own validators, so a bad regex is refused here with the sentence the form shows. `""` clears a field, exactly as clearing the input does. +- **`keep-videos` sets the do-not-clean marker** — the video page's "Do not + clean" toggle, over every video of one channel whose title or description + matches `match`. `match` is matched exactly as a `downloadFilterInclude` is (a + case-insensitive regex over title + description, the same compile), and + `fields: ["title"]` or `["description"]` narrows it to one half; `dryRun: true` + reports without writing. The marker lives in the video's `data/<id>/`, so it + is two steps on a fresh channel: run `metadata-scan` first (an id with no text + cannot match — the reply counts those as `unscanned`), then `keep-videos`. + Matches that were never downloaded come back in `notDownloaded` rather than + getting a directory; pass them to `download-missing` and run `keep-videos` + again to mark them. The read side needs no new routes for jobs: `/api/jobs/active`, `/api/jobs/<id>/log`, `/api/scheduler/status` and `/api/auto-queue/status` diff --git a/common/controller/keepVideosMatching.test.ts b/common/controller/keepVideosMatching.test.ts @@ -0,0 +1,223 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, rm, stat, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import { loadDoNotClean, setDoNotClean } from "../lib/doNotClean-server"; +import { KeepVideosError, keepVideosMatching } from "./keepVideosMatching"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/keepVideosMatching.test.ts + +const SLUG = "ch"; + +type Seed = { + // id -> [title, description] written to data/<id>/metadata.info.json + dirs?: Record<string, [string, string]>; + // id -> [title, description] written to metadata-scan.json only + scanned?: Record<string, [string, string]>; + // bare ids written to the playlist file + playlist?: string[]; +}; + +async function withChannel( + seed: Seed, + fn: (paths: Paths, channelDir: string) => Promise<void>, +): Promise<void> { + const root = await mkdtemp(path.join(tmpdir(), "ttb-keep-videos-")); + const paths = { channelsDir: path.join(root, "channels") } as Paths; + const channelDir = path.join(paths.channelsDir, SLUG); + try { + await mkdir(path.join(channelDir, "data"), { recursive: true }); + await writeFile( + path.join(channelDir, "config.json"), + JSON.stringify({ name: "Ch", url: "https://www.youtube.com/@ch" }), + ); + for (const [id, [title, description]] of Object.entries(seed.dirs ?? {})) { + const dir = path.join(channelDir, "data", id); + await mkdir(dir, { recursive: true }); + await writeFile( + path.join(dir, "metadata.info.json"), + JSON.stringify({ id, title, description }), + ); + } + const entries: Record<string, unknown> = {}; + for (const [id, [title, description]] of Object.entries(seed.scanned ?? {})) { + entries[id] = { + title, + description, + uploadDate: "20240101", + scannedAt: "2026-09-25T00:00:00.000Z", + }; + } + await writeFile( + path.join(channelDir, "metadata-scan.json"), + JSON.stringify({ version: 1, updatedAt: "", lastRun: null, entries, errors: {} }), + ); + await writeFile( + path.join(channelDir, "playlist"), + (seed.playlist ?? []) + .map((id) => `https://www.youtube.com/watch?v=${id}`) + .join("\n"), + ); + await fn(paths, channelDir); + } finally { + await rm(root, { recursive: true, force: true }); + } +} + +const dirOf = (channelDir: string, id: string) => + path.join(channelDir, "data", id); + +test("a title match and a description-only match are both marked, case-insensitively", async () => { + await withChannel( + { + dirs: { + aaaaaaaaaaa: ["Reacting to THEQUARTERING", "nothing here"], + bbbbbbbbbbb: ["Some other video", "featuring thequartering at 10:00"], + ccccccccccc: ["Unrelated", "unrelated"], + }, + }, + async (paths, channelDir) => { + const r = await keepVideosMatching({ + paths, + channelSlug: SLUG, + pattern: "TheQuartering", + }); + assert.equal(r.considered, 3); + assert.equal(r.marked, 2); + assert.deepEqual( + r.matched.map((m) => [m.id, m.marked, m.downloaded]), + [ + ["aaaaaaaaaaa", true, true], + ["bbbbbbbbbbb", true, true], + ], + ); + const rec = await loadDoNotClean(dirOf(channelDir, "aaaaaaaaaaa")); + assert.equal(rec?.note, "keep-videos: matched /TheQuartering/i"); + assert.equal(await loadDoNotClean(dirOf(channelDir, "ccccccccccc")), null); + }, + ); +}); + +test("an already-kept video is counted and not rewritten", async () => { + await withChannel( + { dirs: { aaaaaaaaaaa: ["TheQuartering live", ""] } }, + async (paths, channelDir) => { + const dir = dirOf(channelDir, "aaaaaaaaaaa"); + await setDoNotClean(dir, true, "by hand"); + const markerFile = path.join(dir, "do-not-clean.json"); + const before = (await stat(markerFile)).mtimeMs; + await new Promise((r) => setTimeout(r, 20)); + const r = await keepVideosMatching({ + paths, + channelSlug: SLUG, + pattern: "thequartering", + }); + assert.equal(r.marked, 0); + assert.equal(r.alreadyKept, 1); + assert.deepEqual(r.matched[0], { + id: "aaaaaaaaaaa", + title: "TheQuartering live", + downloaded: true, + alreadyKept: true, + marked: false, + }); + assert.equal((await stat(markerFile)).mtimeMs, before); + assert.equal((await loadDoNotClean(dir))?.note, "by hand"); + }, + ); +}); + +test("a scan-only match is reported notDownloaded and no dir is created; unscanned counts the gap", async () => { + await withChannel( + { + scanned: { sssssssssss: ["TheQuartering clip", ""] }, + playlist: ["sssssssssss", "uuuuuuuuuuu", "vvvvvvvvvvv"], + }, + async (paths, channelDir) => { + const r = await keepVideosMatching({ + paths, + channelSlug: SLUG, + pattern: "thequartering", + }); + assert.deepEqual(r.notDownloaded, ["sssssssssss"]); + assert.equal(r.marked, 0); + assert.equal(r.matched[0].downloaded, false); + assert.equal(r.unscanned, 2); + await assert.rejects(stat(dirOf(channelDir, "sssssssssss")), { + code: "ENOENT", + }); + }, + ); +}); + +test("a dry run reports the matches and writes nothing", async () => { + await withChannel( + { dirs: { aaaaaaaaaaa: ["TheQuartering", ""] } }, + async (paths, channelDir) => { + const r = await keepVideosMatching({ + paths, + channelSlug: SLUG, + pattern: "thequartering", + dryRun: true, + }); + assert.equal(r.dryRun, true); + assert.equal(r.matched.length, 1); + assert.equal(r.matched[0].marked, false); + assert.equal(r.marked, 0); + assert.equal(await loadDoNotClean(dirOf(channelDir, "aaaaaaaaaaa")), null); + }, + ); +}); + +test('fields: ["title"] ignores a description hit', async () => { + await withChannel( + { + dirs: { + aaaaaaaaaaa: ["Plain title", "TheQuartering in the description"], + bbbbbbbbbbb: ["TheQuartering in the title", ""], + }, + }, + async (paths) => { + const r = await keepVideosMatching({ + paths, + channelSlug: SLUG, + pattern: "thequartering", + fields: ["title"], + }); + assert.deepEqual(r.fields, ["title"]); + assert.deepEqual( + r.matched.map((m) => m.id), + ["bbbbbbbbbbb"], + ); + }, + ); +}); + +test("a bad pattern, an unknown field and an unknown channel throw the typed error", async () => { + await withChannel({ dirs: {} }, async (paths) => { + await assert.rejects( + keepVideosMatching({ paths, channelSlug: SLUG, pattern: "(" }), + (e) => e instanceof KeepVideosError && /invalid pattern/.test(e.message), + ); + await assert.rejects( + keepVideosMatching({ paths, channelSlug: SLUG, pattern: "(\\w+\\s?)*$" }), + (e) => e instanceof KeepVideosError && /nested quantifier/.test(e.message), + ); + await assert.rejects( + keepVideosMatching({ + paths, + channelSlug: SLUG, + pattern: "x", + fields: ["tags" as "title"], + }), + KeepVideosError, + ); + await assert.rejects( + keepVideosMatching({ paths, channelSlug: "nope", pattern: "x" }), + (e) => e instanceof KeepVideosError && /not found/.test(e.message), + ); + }); +}); diff --git a/common/controller/keepVideosMatching.ts b/common/controller/keepVideosMatching.ts @@ -0,0 +1,247 @@ +import path from "node:path"; +import { readFile } from "node:fs/promises"; +import type { Paths } from "../lib/paths"; +import { assertChannelMediaReachable } from "../lib/channelMedia"; +import { isDoNotClean, setDoNotClean } from "../lib/doNotClean-server"; +import { + classifyAgainstFilter, + compileDownloadFilter, + downloadFilterPatternProblem, +} from "../lib/downloadFilters"; +import { loadRawMetadataFromDir } from "../lib/transcripts-server"; +import { extractVideoId } from "../lib/videoId"; +import { channelExists, readChannelConfig } from "./channels"; +import { listChannelVideoIds } from "./keptVideos"; +import { loadMetadataScan } from "./metadataScanStore"; +import { loadRoster } from "./rosterStore"; + +// BULK "KEEP THE VIDEO": set the per-video do-not-clean marker on every video of +// one channel whose title / description matches a pattern. The marker is the +// same `data/<id>/do-not-clean.json` the video page's toggle writes +// (lib/doNotClean-server.ts), and every cleaner already honours it — the clean +// sweep, extra-format cleanup, wrong-format removal, the superseded-subs purge +// and saved-video eviction. This module adds no new kind of protection; it is +// the loop the toggle lacks. +// +// "MATCHES" MEANS WHAT THE DOWNLOAD FILTER MEANS BY IT. The pattern is compiled +// by `compileDownloadFilter` (case-insensitive, the same source a channel's +// `downloadFilter.include` holds) and tested by `classifyAgainstFilter` over +// `downloadFilterText` — title + "\n" + description, the description capped +// exactly as the filter caps it. So a pattern that keeps the right videos here +// is one that would download the right ones there, and vice versa. There is no +// second matcher. `fields` narrows the SUBJECT, not the matcher: a field left +// out is passed as empty. +// +// TEXT COMES FROM TWO PLACES, NEVER A THIRD. A downloaded video's own +// `metadata.info.json` first; otherwise the channel's metadata-scan store +// (`metadata-scan.json`), which is how a scanned-but-undownloaded video has a +// title at all. An id known only from `playlist` / `roster.json` has no text and +// is COUNTED (`unscanned`) — run `metadata-scan` first to close that gap. +// +// A MATCH WITH NO `data/<id>/` IS REPORTED, NEVER CREATED. The marker lives in +// the video's dir and `setDoNotClean` does not mkdir. Creating the dir here +// would also break the metadata scan's invariant that it never makes one +// (metadataScanStore.ts) — every enumerator reads "has a dir" as "was fetched". +// `notDownloaded` lists them; `download-missing` takes the ids, and a re-run +// then marks them. +// +// AN UNMOUNTED DRIVE IS NOT AN EMPTY CHANNEL. `listChannelVideoIds` swallows +// ENOENT on data/, which on a relocated channel would turn every downloaded +// match into `notDownloaded`. The media guard is asked first, and refuses. + +export const KEEP_VIDEOS_FIELDS = ["title", "description"] as const; +export type KeepVideosField = (typeof KEEP_VIDEOS_FIELDS)[number]; + +// Thrown for a refusal the CALLER can fix (a bad pattern, an unknown channel, +// a bad field). The editor action turns it into its `{ ok: false, error }`, and +// the ops route into a 400. Anything else is an I/O failure. +export class KeepVideosError extends Error { + constructor(message: string) { + super(message); + this.name = "KeepVideosError"; + } +} + +export type KeepVideosOptions = { + paths: Paths; + channelSlug: string; + pattern: string; + fields?: readonly KeepVideosField[]; + note?: string; + dryRun?: boolean; +}; + +export type KeepVideosMatch = { + id: string; + title: string; + // Has a data/<id>/ dir — the only kind that can carry the marker. + downloaded: boolean; + // Already carried a marker before this call; left untouched. + alreadyKept: boolean; + // This call wrote the marker (always false on a dry run). + marked: boolean; +}; + +export type KeepVideosResult = { + pattern: string; + fields: KeepVideosField[]; + // Ids that had text to test: every data dir plus every scan entry. + considered: number; + matched: KeepVideosMatch[]; + marked: number; + alreadyKept: number; + // Matched ids with no data/<id>/ — download them, then re-run. + notDownloaded: string[]; + // Ids in playlist / roster.json with neither a dir nor a scan entry: the + // coverage gap no pattern can see into. `metadata-scan` closes it. + unscanned: number; + // Data dirs with no metadata.info.json and no scan entry — they have a dir + // but no text, so they could not be tested. Rare; counted, not guessed. + noMetadata: number; + dryRun: boolean; +}; + +function resolveFields( + fields: readonly KeepVideosField[] | undefined, +): KeepVideosField[] { + if (fields === undefined) return [...KEEP_VIDEOS_FIELDS]; + if (fields.length === 0) { + throw new KeepVideosError( + `"fields" must name at least one of ${KEEP_VIDEOS_FIELDS.join(", ")}`, + ); + } + for (const f of fields) { + if (!(KEEP_VIDEOS_FIELDS as readonly string[]).includes(f)) { + throw new KeepVideosError( + `"fields" may only contain ${KEEP_VIDEOS_FIELDS.join(", ")} (got "${f}")`, + ); + } + } + return KEEP_VIDEOS_FIELDS.filter((f) => fields.includes(f)); +} + +async function readPlaylistIds(channelDir: string): Promise<string[]> { + const raw = await readFile(path.join(channelDir, "playlist"), "utf8").catch( + () => "", + ); + const ids: string[] = []; + for (const line of raw.split("\n")) { + const url = line.trim(); + if (!url) continue; + const id = extractVideoId(url); + if (id) ids.push(id); + } + return ids; +} + +export async function keepVideosMatching( + opts: KeepVideosOptions, +): Promise<KeepVideosResult> { + const { paths, channelSlug } = opts; + const pattern = opts.pattern.trim(); + const dryRun = opts.dryRun === true; + const fields = resolveFields(opts.fields); + + if (!pattern) throw new KeepVideosError('"match" must be a non-empty pattern'); + // The download filter form's own guard: the pattern runs over every title + // and description on the channel, on the server's one thread. + const problem = downloadFilterPatternProblem(pattern); + if (problem) throw new KeepVideosError(`invalid pattern: ${problem}`); + const compiled = compileDownloadFilter({ include: pattern }); + if (!compiled?.include) { + let why = "it does not parse"; + try { + new RegExp(pattern, "i"); + } catch (e) { + why = (e as Error).message; + } + throw new KeepVideosError(`invalid pattern /${pattern}/i: ${why}`); + } + + if (!(await channelExists(paths, channelSlug))) { + throw new KeepVideosError(`Channel "${channelSlug}" not found`); + } + const config = await readChannelConfig(paths, channelSlug); + await assertChannelMediaReachable(paths, channelSlug, config); + + const channelDir = path.join(paths.channelsDir, channelSlug); + const [dirIds, scan, roster, playlistIds] = await Promise.all([ + listChannelVideoIds(paths, channelSlug), + loadMetadataScan(paths, channelSlug), + loadRoster(paths, channelSlug), + readPlaylistIds(channelDir), + ]); + const dirs = new Set(dirIds); + const ids = [...new Set([...dirIds, ...Object.keys(scan.entries)])].sort(); + + const matched: KeepVideosMatch[] = []; + const notDownloaded: string[] = []; + let marked = 0; + let alreadyKept = 0; + let noMetadata = 0; + let considered = 0; + const note = opts.note?.trim() || `keep-videos: matched /${pattern}/i`; + + for (const id of ids) { + const downloaded = dirs.has(id); + const videoDir = path.join(channelDir, "data", id); + const meta = downloaded ? await loadRawMetadataFromDir(videoDir) : null; + const entry = scan.entries[id]; + let title: string; + let description: string; + if (meta) { + title = meta.title ?? ""; + description = meta.description ?? ""; + } else if (entry) { + title = entry.title; + description = entry.description; + } else { + noMetadata++; + continue; + } + considered++; + const subject = { + title: fields.includes("title") ? title : "", + description: fields.includes("description") ? description : "", + }; + if (classifyAgainstFilter(compiled, subject) !== "text") continue; + + if (!downloaded) { + notDownloaded.push(id); + matched.push({ id, title, downloaded, alreadyKept: false, marked: false }); + continue; + } + const kept = await isDoNotClean(videoDir); + if (kept) { + alreadyKept++; + } else if (!dryRun) { + await setDoNotClean(videoDir, true, note); + marked++; + } + matched.push({ + id, + title, + downloaded, + alreadyKept: kept, + marked: !kept && !dryRun, + }); + } + + const known = new Set([...dirs, ...Object.keys(scan.entries)]); + const listed = new Set([...playlistIds, ...Object.keys(roster.entries)]); + let unscanned = 0; + for (const id of listed) if (!known.has(id)) unscanned++; + + return { + pattern, + fields, + considered, + matched, + marked, + alreadyKept, + notDownloaded, + unscanned, + noMetadata, + dryRun, + }; +} diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **New ops action: `pnpm ops keep-videos` marks every video of a channel whose title or description matches a pattern as "do not clean".** It sets the same marker as the video page's *Do not clean* toggle, so the clean sweep, extra-format cleanup, wrong-format removal, the superseded-subs purge and saved-video eviction all leave those videos alone. The body is `{"slug", "match", "fields"?, "note"?, "dryRun"?}`. `match` is matched the way a channel's download filter *include* is: a case-insensitive regex over title + description. `fields: ["title"]` or `["description"]` narrows it to one half, and `dryRun: true` reports without writing. Videos that already carry the marker are counted and left as they are. The marker lives in the video's folder, so a match that was never downloaded is listed under `notDownloaded` and no folder is created for it. Run `download-missing` on those ids, then run `keep-videos` again. On a new channel, run `metadata-scan` first: a video with no scanned title cannot match, and the reply counts those as `unscanned`. - **Auto-download no longer retries the same rate-limited video over and over; it moves on to the next one.** When a download answered HTTP 429, the runner paused the whole platform for a while and then picked the same video again, because it was still first in the queue. Each retry doubled the pause, up to 30 minutes. On 2026-09-24 one YouTube Short was retried 12 times this way and kept YouTube paused all evening. A YouTube 429 comes from the subtitle fetch for one video, not from the whole site. Now a rate-limited video is also **deferred for 6 hours**: auto-download skips it, so when the pause ends the runner takes the next video. The pause still grows only when *different* videos keep hitting the limit. Deferrals are kept in `.auto-queue/state.json` beside the platform cooldowns, so a restart does not retry the video early. The log line reads `… (attempt 1). <id> deferred 6h; next video after cooldown.` A manual Sync or *download missing* ignores deferrals and still fetches the video. When every video left is deferred, the runner reports that it is idle for that reason: "every pending video was rate-limited recently and is deferred". - **The cooldown strip on `/operations/download` also lists deferred videos.** It is now a region named *Rate-limit cooldown*, with a *Platforms in cooldown* list (unchanged) and a *Deferred videos* list. Each deferred video links to its page and shows how long it has left (`alpha/a1 — 5h 59m left`). The strip appears when either list has something in it. Times over an hour now read `5h 59m` instead of `359m 58s`. - **A video that keeps failing its metadata scan is no longer rescanned at every runner start.** When a scan hit the same error again (members-only, for example), the error's timestamp was not updated. The one-day rest that timestamp controls therefore ran out once and never started again, and each runner start rescanned all of them: 142 members-only videos on one channel, with cookies. The same error seen again after a day now updates the timestamp, so the video waits another day. diff --git a/editor/app/api/ops/keep-videos/route.ts b/editor/app/api/ops/keep-videos/route.ts @@ -0,0 +1,57 @@ +import { NextResponse } from "next/server"; +import { keepVideosAction } from "../../../channels/[slug]/videos/[id]/videoActions"; +import { OpsInputError, ops, optBool, optString, opsFail, reqSlug, reqString } from "../_lib"; + +export const dynamic = "force-dynamic"; + +// POST { slug, match, fields?: ("title"|"description")[], note?, dryRun? } +// -> { ok: true, pattern, fields, considered, matched: [{ id, title, +// downloaded, alreadyKept, marked }], marked, alreadyKept, +// notDownloaded, unscanned, noMetadata, dryRun } +// +// AN ADAPTER: one call to keepVideosAction, the bulk form of the video page's +// "Do not clean" toggle. It writes the same per-video do-not-clean.json marker +// every cleaner already honours; it adds no protection of its own. +// +// `match` IS A DOWNLOAD-FILTER PATTERN, matched the way the filter matches it — +// a case-insensitive regex SOURCE over title + "\n" + description, compiled by +// the same function. A pattern that keeps the right videos here would download +// the right ones as a channel's `downloadFilterInclude`. `fields` narrows the +// subject to one half. +// +// A MATCH WITH NO data/<id>/ IS REPORTED, NOT CREATED. The marker lives in the +// video's dir, and a dir is what every enumerator reads as "fetched". Those ids +// come back in `notDownloaded`: send them to `download-missing`, then run this +// again. A fresh channel wants `metadata-scan` FIRST — an id with no dir and no +// scan entry has no text to match, and is only counted (`unscanned`). +// +// `dryRun: true` answers the same question and writes nothing. +export async function POST(request: Request) { + return ops(request, ["slug", "match", "fields", "note", "dryRun"], async (body) => { + const slug = reqSlug(body, "slug"); + const pattern = reqString(body, "match"); + const raw = body.fields; + let fields: ("title" | "description")[] | undefined; + if (raw !== undefined) { + if ( + !Array.isArray(raw) || + raw.length === 0 || + raw.some((f) => f !== "title" && f !== "description") + ) { + throw new OpsInputError( + '"fields" must be a non-empty array of "title" and/or "description"', + ); + } + fields = raw as ("title" | "description")[]; + } + const result = await keepVideosAction({ + slug, + pattern, + fields, + note: optString(body, "note"), + dryRun: optBool(body, "dryRun"), + }); + if (!result.ok) return opsFail(result.error); + return NextResponse.json({ ok: true, ...result.result }); + }); +} diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -29,6 +29,13 @@ import { } from "yt-dlp-transcript-common/lib/queueKeys"; import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"; import { setDoNotClean } from "yt-dlp-transcript-common/lib/doNotClean-server"; +import { + KeepVideosError, + keepVideosMatching, + type KeepVideosField, + type KeepVideosResult, +} from "yt-dlp-transcript-common/controller/keepVideosMatching"; +import { ChannelMediaUnreachableError } from "yt-dlp-transcript-common/lib/channelMedia"; import { setExcludedFromTruncatedCheck } from "yt-dlp-transcript-common/lib/excludeTruncatedCheck-server"; import { pruneFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions"; import { transcodeAudio } from "yt-dlp-transcript-common/controller/transcode"; @@ -640,6 +647,46 @@ export async function toggleDoNotCleanAction( return { ok: true }; } +// The BULK form of the toggle above: set the do-not-clean marker on every video +// of one channel whose title / description matches `pattern` (the download +// filter's matcher and subject — see common/controller/keepVideosMatching.ts). +// No UI calls it yet; `pnpm ops keep-videos` does. Same revalidation as the +// toggle, per marked video, and ONE snapshot request for the whole batch — the +// snapshot's cleanup buckets exclude kept ids, so it has to be re-derived, but +// once, not once per video. +export async function keepVideosAction(input: { + slug: string; + pattern: string; + fields?: KeepVideosField[]; + note?: string; + dryRun?: boolean; +}): Promise<{ ok: true; result: KeepVideosResult } | { ok: false; error: string }> { + let result: KeepVideosResult; + try { + result = await keepVideosMatching({ + paths: getPaths(), + channelSlug: input.slug, + pattern: input.pattern, + fields: input.fields, + note: input.note, + dryRun: input.dryRun, + }); + } catch (e) { + if (e instanceof KeepVideosError || e instanceof ChannelMediaUnreachableError) { + return { ok: false, error: e.message }; + } + throw e; + } + if (result.marked > 0) { + for (const m of result.matched) { + if (m.marked) revalidatePath(`/channels/${input.slug}/videos/${m.id}`); + } + revalidatePath(`/channels/${input.slug}`); + requestChannelSnapshot(getPaths(), input.slug); + } + return { ok: true, result }; +} + // Opt this video in/out of the truncated/incomplete-transcript check. When // enabled, the snapshot's incompleteTranscript + shortAudio buckets and the // video panel's "looks truncated" banner suppress this id (for videos that diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts @@ -718,6 +718,79 @@ test("tag-videos refuses a traversing slug, a bad op and an unknown key", async expect(await pathExists("test-transcripts/tags.json")).toBe(false); }); +// --- keep-videos: the bulk do-not-clean toggle -------------------------------- +// +// The marker is the video page's own do-not-clean.json; what this pins is the +// loop around it — the download filter's matcher over title + description, a +// dry run that writes nothing, and a scan-only match that is REPORTED rather +// than given a directory it was never downloaded into. +test("keep-videos marks the matching video only, after a dry run that writes nothing", async ({ + request, +}) => { + await resetData("one-youtube-channel-with-data"); + const slug = "test-youtube"; + const kept = "20240102_keepme12345"; + const other = "20240101_test1234567"; // "Synthetic Test Video" — no needle + const channel = `test-transcripts/channels/${slug}`; + await mkdir(resolvePath(`${channel}/data/${kept}`), { recursive: true }); + await writeFile( + resolvePath(`${channel}/data/${kept}/metadata.info.json`), + JSON.stringify({ id: kept, title: "Reacting to THEQUARTERING", description: "" }), + ); + await writeFile( + resolvePath(`${channel}/metadata-scan.json`), + JSON.stringify({ + version: 1, + updatedAt: "", + lastRun: null, + errors: {}, + entries: { + scanonly123: { + title: "Plain", + description: "with TheQuartering", + uploadDate: "20240103", + scannedAt: "2026-09-25T00:00:00.000Z", + }, + }, + }), + ); + const marker = (id: string) => `${channel}/data/${id}/do-not-clean.json`; + + type KeepBody = OpsResponse & { + matched?: { id: string; marked: boolean }[]; + marked?: number; + notDownloaded?: string[]; + }; + const dry = await ops(request, "keep-videos", { + slug, + match: "thequartering", + dryRun: true, + }); + expect(dry.status, JSON.stringify(dry.body)).toBe(200); + const dryBody = dry.body as KeepBody; + expect(dryBody.matched?.filter((m) => m.id === kept)).toHaveLength(1); + expect(dryBody.marked).toBe(0); + expect(await pathExists(marker(kept))).toBe(false); + + const real = await ops(request, "keep-videos", { slug, match: "thequartering" }); + expect(real.status, JSON.stringify(real.body)).toBe(200); + const body = real.body as KeepBody; + expect(body.marked).toBe(1); + expect(body.notDownloaded).toEqual(["scanonly123"]); + expect(await pathExists(marker(kept))).toBe(true); + expect(await pathExists(marker(other))).toBe(false); + // A scan-only match gets no directory — that would read as "downloaded". + expect(await pathExists(`${channel}/data/scanonly123`)).toBe(false); + + const unknown = await ops(request, "keep-videos", { slug, match: "x", pattern: "x" }); + expect(unknown.status).toBe(400); + expect(unknown.body.error).toContain("unknown key(s): pattern"); + + const badRegex = await ops(request, "keep-videos", { slug, match: "(" }); + expect(badRegex.status).toBe(400); + expect(badRegex.body.error).toContain("invalid pattern"); +}); + // --- the two build routes speak the same body --------------------------------- // // They did not. build-site took `siteIds` (a list) and build-deploy took diff --git a/plans/release-7.md b/plans/release-7.md @@ -542,4 +542,71 @@ Homepage: **15 passed, 7 skipped, 25 s**. - **Commit trailers** name `Claude Opus 5.5 (1M context)`, as in releases 5 and 6. +### Slice K, as shipped — `pnpm ops keep-videos` (2026-09-25) + +Branch `one-core/r7-keep` off `main` `3049be43`. The operator's ask: "I've just added the Paramount +Tactical channel — mark anything that includes TheQuartering in title or description as 'keep the +video', with an ops command." "Keep the video" is the existing per-video **do-not-clean** marker +(`data/<id>/do-not-clean.json`, `setDoNotClean`). The clean sweep, extra-format cleanup, +wrong-format removal, the superseded-subs purge and saved-video eviction already honour it. Before +this slice there was only the per-video toggle: no bulk form and no ops action. The slice adds the +loop around the marker and no new kind of protection. Two facts shaped it: +- **Text lives in two places only.** A downloaded video's `metadata.info.json` is read first, + then the channel's `metadata-scan.json` entry. An id known only from `playlist` / `roster.json` + has no text and is counted (`unscanned`), not guessed. +- **`setDoNotClean` does not mkdir, and nothing here creates `data/<id>/`.** A matched video with + no dir is reported in `notDownloaded`. Creating the dir would break the scan store's invariant, + and every enumerator reads a dir as "fetched". + +"Matches" is the download filter's matcher. The pattern is compiled by `compileDownloadFilter({include})` +and tested by `classifyAgainstFilter` over `downloadFilterText` (title + "\n" + description, with +the description capped at 2 KB as the filter caps it), after `downloadFilterPatternProblem` (the +form's ReDoS guard). There is no second matcher. `fields` narrows the subject by passing the +left-out field as `""`. **The media guard runs first** (`assertChannelMediaReachable`): +`listChannelVideoIds` swallows ENOENT, so on an unmounted relocated channel every downloaded match +would otherwise read as `notDownloaded`. + +| sha | what | +|---|---| +| `8f9b0fc6` | `common/controller/keepVideosMatching.ts`: `keepVideosMatching({paths, channelSlug, pattern, fields?, note?, dryRun?})` returns `{pattern, fields, considered, matched: [{id, title, downloaded, alreadyKept, marked}], marked, alreadyKept, notDownloaded, unscanned, noMetadata, dryRun}`, and `KeepVideosError` covers a bad or unsafe pattern, a bad field and an unknown channel. The default note is `keep-videos: matched /<pattern>/i`. `noMetadata` is **additive to the brief's shape**: a data dir with no `metadata.info.json` and no scan entry has a dir but no text, so it is counted rather than silently skipped. `.test.ts` has 6 cases: title + description-only + case-insensitive; already-kept counted with its mtime and note unchanged; scan-only goes to `notDownloaded` with no dir created and `unscanned` = 2; dry run; `fields:["title"]`; the typed error for `(`, a nested quantifier, an unknown field and an unknown channel | +| `22ec333b` | `keepVideosAction` in `videoActions.ts`, beside `toggleDoNotCleanAction`. It maps `KeepVideosError` / `ChannelMediaUnreachableError` to `{ok: false, error}`. When `marked > 0` it calls `revalidatePath` for each marked video and the channel page, and `requestChannelSnapshot` **once** | +| `f11c5a72` | `editor/app/api/ops/keep-videos/route.ts`, an adapter with keys `slug, match, fields, note, dryRun` (`reqSlug`; `fields` must be a non-empty array of `"title"`/`"description"`) that returns `{ok: true, ...result}`. `scripts/archilyzer-ops.mjs` ACTIONS gains `keep-videos`, and `archilyzer-ops.test.mjs` +1. `ops-api.spec.ts` +1 covers: a seeded `20240102_keepme12345` titled "Reacting to THEQUARTERING" plus a scan-only `scanonly123` with a description hit. The dry run writes nothing; the real call gives `marked: 1` and `notDownloaded: ["scanonly123"]`, puts the marker only under the matching id, and creates no dir for the scan-only id. An unknown key returns 400 and `(` returns 400 | +| `73be3573` | `RUNNING_IN_DOCKER.md`, "Driving the editor without a browser": an example line and a bullet for the two-step reality (`metadata-scan` first, `notDownloaded` then `download-missing`, then re-run) | +| *(this commit)* | this record and the `[Unreleased]` bullet | + +**Gates** (worktree root, on `73be3573`). tsc (`pnpm -r --no-bail --workspace-concurrency=1 exec +tsc --noEmit`) was clean on the full tree before the commits were made. The four commits are +additive, in dependency order. common **1795/1795** = 1789 + 6 (`keepVideosMatching`). Editor unit +**72/72**. test:scripts **161 pass + 1 skip** (160 + 1). mcp **219/219**. `pnpm --filter editor +exec next build` ok, and `.next/server/app/api/ops/keep-videos` was emitted. The export build was +not run, because the slice touches no file under `export/` and no `common/` module it imports. +EDITOR e2e, `$T/k-specs.txt` = `ops-api.spec.ts` (`k-e2e1.log`): **21 passed, 0 failed, 58.9 s**, +after a 4 m 58 s queue wait behind the release's final suites. All heavy steps were started only +at ≥ 3 GB available memory, as the brief required. + +**Numbers: none.** No `settings.json`, `site.json` or `config.json` key changed. The action writes +only per-video `do-not-clean.json` sidecars, and only when someone runs it. + +**Not merged with `main` `211d4666`.** The coordinator asked for `git merge main` before the final +gates (a test-only commit in the deploy-hub region of `ops-api.spec.ts`). The merge was **refused by +the session's permission classifier**, so the branch is still on `3049be43`. `git merge-tree +--write-tree HEAD main` reports a **clean** merge. The parent's merge takes it as is, and the +`ops-api.spec` run above does not include `211d4666`'s edit. + +**Found and left.** +- **The rule-shaped alternative is not built.** This is a one-shot action: a video downloaded + *after* the run is not marked. A persistent per-channel "keep filter" (say `keepFilter.include` + in `config.json`, evaluated at download time and by the cleaners) would keep future matches too. + That is a schema change (CHANNEL.md, numbers), so it is out of scope here. Until then, re-run + `keep-videos` after new downloads. It is idempotent: already-kept videos are counted, not + rewritten. +- There is no UI surface. The action exists for a future bulk bar or channel-page control. +- `keepVideosMatching.ts` has its own 12-line `readPlaylistIds`. The same helper is duplicated + privately in `ytdlp/metadataScan.ts` and `controller/recencyIndex.ts`, and neither is exported. + A third copy was cheaper than reshaping two files outside this slice's ownership. +- `unscanned` compares playlist/roster ids with dir names. A legacy dir named `YYYYMMDD_<id>` does + not equal its playlist id, so such a channel over-counts `unscanned`. The count is advisory. +- **Commit trailers** name `Claude Opus 5.5 (1M context)`, as the release-6 and release-7 + implementers did. + ## Rollout diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -114,6 +114,9 @@ const ACTIONS = [ // of videos in ONE write. "tags", "tag-videos", + // Set the per-video do-not-clean marker on every video of a channel whose + // title/description matches a download-filter pattern ({slug, match}). + "keep-videos", ]; // The provenance a tag write from this CLI carries. Everything else ignores it. diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs @@ -90,6 +90,18 @@ test("the tag actions are registered and post to their routes", () => { assert.match(usage(), /tags/); }); +test("keep-videos is registered and posts to its route", () => { + const p = parseArgs([ + "keep-videos", + "--json", + '{"slug":"c","match":"TheQuartering","dryRun":true}', + ]); + assert.equal(p.path, "/api/ops/keep-videos"); + assert.deepEqual(p.body, { slug: "c", match: "TheQuartering", dryRun: true }); + // Not a provenance-carrying write: the marker records a note, not a source. + assert.equal(p.defaultSource, undefined); +}); + // A FOUR-THOUSAND-ID BODY IS WRITTEN BY A SCRIPT, NOT TYPED. parseArgs stays // pure — it reports the file it would read, and main() reads it — so this test // needs no filesystem.