import path from "node:path"; import { readFile } from "node:fs/promises"; import type { Paths } from "../lib/paths"; import { assertChannelMediaReachable } from "../lib/channelMedia"; import { isDoNotClean, setDoNotClean } from "../lib/doNotClean-server"; import { classifyAgainstFilter, compileDownloadFilter, downloadFilterPatternProblem, } from "../lib/downloadFilters"; import { loadRawMetadataFromDir } from "../lib/transcripts-server"; import { extractVideoId } from "../lib/videoId"; import { channelExists, readChannelConfig } from "./channels"; import { listChannelVideoIds } from "./keptVideos"; import { loadMetadataScan } from "./metadataScanStore"; import { loadRoster } from "./rosterStore"; // BULK "KEEP THE VIDEO": set the per-video do-not-clean marker on every video of // one channel whose title / description matches a pattern. The marker is the // same `data//do-not-clean.json` the video page's toggle writes // (lib/doNotClean-server.ts), and every cleaner already honours it — the clean // sweep, extra-format cleanup, wrong-format removal, the superseded-subs purge // and saved-video eviction. This module adds no new kind of protection; it is // the loop the toggle lacks. // // "MATCHES" MEANS WHAT THE DOWNLOAD FILTER MEANS BY IT. The pattern is compiled // by `compileDownloadFilter` (case-insensitive, the same source a channel's // `downloadFilter.include` holds) and tested by `classifyAgainstFilter` over // `downloadFilterText` — title + "\n" + description, the description capped // exactly as the filter caps it. So a pattern that keeps the right videos here // is one that would download the right ones there, and vice versa. There is no // second matcher. `fields` narrows the SUBJECT, not the matcher: a field left // out is passed as empty. // // TEXT COMES FROM TWO PLACES, NEVER A THIRD. A downloaded video's own // `metadata.info.json` first; otherwise the channel's metadata-scan store // (`metadata-scan.json`), which is how a scanned-but-undownloaded video has a // title at all. An id known only from `playlist` / `roster.json` has no text and // is COUNTED (`unscanned`) — run `metadata-scan` first to close that gap. // // A MATCH WITH NO `data//` IS REPORTED, NEVER CREATED. The marker lives in // the video's dir and `setDoNotClean` does not mkdir. Creating the dir here // would also break the metadata scan's invariant that it never makes one // (metadataScanStore.ts) — every enumerator reads "has a dir" as "was fetched". // `notDownloaded` lists them; `download-missing` takes the ids, and a re-run // then marks them. // // AN UNMOUNTED DRIVE IS NOT AN EMPTY CHANNEL. `listChannelVideoIds` swallows // ENOENT on data/, which on a relocated channel would turn every downloaded // match into `notDownloaded`. The media guard is asked first, and refuses. export const KEEP_VIDEOS_FIELDS = ["title", "description"] as const; export type KeepVideosField = (typeof KEEP_VIDEOS_FIELDS)[number]; // Thrown for a refusal the CALLER can fix (a bad pattern, an unknown channel, // a bad field). The editor action turns it into its `{ ok: false, error }`, and // the ops route into a 400. Anything else is an I/O failure. export class KeepVideosError extends Error { constructor(message: string) { super(message); this.name = "KeepVideosError"; } } export type KeepVideosOptions = { paths: Paths; channelSlug: string; pattern: string; fields?: readonly KeepVideosField[]; note?: string; dryRun?: boolean; }; export type KeepVideosMatch = { id: string; title: string; // Has a data// dir — the only kind that can carry the marker. downloaded: boolean; // Already carried a marker before this call; left untouched. alreadyKept: boolean; // This call wrote the marker (always false on a dry run). marked: boolean; }; export type KeepVideosResult = { pattern: string; fields: KeepVideosField[]; // Ids that had text to test: every data dir plus every scan entry. considered: number; matched: KeepVideosMatch[]; marked: number; alreadyKept: number; // Matched ids with no data// — download them, then re-run. notDownloaded: string[]; // Ids in playlist / roster.json with neither a dir nor a scan entry: the // coverage gap no pattern can see into. `metadata-scan` closes it. unscanned: number; // Data dirs with no metadata.info.json and no scan entry — they have a dir // but no text, so they could not be tested. Rare; counted, not guessed. noMetadata: number; dryRun: boolean; }; function resolveFields( fields: readonly KeepVideosField[] | undefined, ): KeepVideosField[] { if (fields === undefined) return [...KEEP_VIDEOS_FIELDS]; if (fields.length === 0) { throw new KeepVideosError( `"fields" must name at least one of ${KEEP_VIDEOS_FIELDS.join(", ")}`, ); } for (const f of fields) { if (!(KEEP_VIDEOS_FIELDS as readonly string[]).includes(f)) { throw new KeepVideosError( `"fields" may only contain ${KEEP_VIDEOS_FIELDS.join(", ")} (got "${f}")`, ); } } return KEEP_VIDEOS_FIELDS.filter((f) => fields.includes(f)); } async function readPlaylistIds(channelDir: string): Promise { const raw = await readFile(path.join(channelDir, "playlist"), "utf8").catch( () => "", ); const ids: string[] = []; for (const line of raw.split("\n")) { const url = line.trim(); if (!url) continue; const id = extractVideoId(url); if (id) ids.push(id); } return ids; } export async function keepVideosMatching( opts: KeepVideosOptions, ): Promise { const { paths, channelSlug } = opts; const pattern = opts.pattern.trim(); const dryRun = opts.dryRun === true; const fields = resolveFields(opts.fields); if (!pattern) throw new KeepVideosError('"match" must be a non-empty pattern'); // The download filter form's own guard: the pattern runs over every title // and description on the channel, on the server's one thread. const problem = downloadFilterPatternProblem(pattern); if (problem) throw new KeepVideosError(`invalid pattern: ${problem}`); const compiled = compileDownloadFilter({ include: pattern }); if (!compiled?.include) { let why = "it does not parse"; try { new RegExp(pattern, "i"); } catch (e) { why = (e as Error).message; } throw new KeepVideosError(`invalid pattern /${pattern}/i: ${why}`); } if (!(await channelExists(paths, channelSlug))) { throw new KeepVideosError(`Channel "${channelSlug}" not found`); } const config = await readChannelConfig(paths, channelSlug); await assertChannelMediaReachable(paths, channelSlug, config); const channelDir = path.join(paths.channelsDir, channelSlug); const [dirIds, scan, roster, playlistIds] = await Promise.all([ listChannelVideoIds(paths, channelSlug), loadMetadataScan(paths, channelSlug), loadRoster(paths, channelSlug), readPlaylistIds(channelDir), ]); const dirs = new Set(dirIds); const ids = [...new Set([...dirIds, ...Object.keys(scan.entries)])].sort(); const matched: KeepVideosMatch[] = []; const notDownloaded: string[] = []; let marked = 0; let alreadyKept = 0; let noMetadata = 0; let considered = 0; const note = opts.note?.trim() || `keep-videos: matched /${pattern}/i`; for (const id of ids) { const downloaded = dirs.has(id); const videoDir = path.join(channelDir, "data", id); const meta = downloaded ? await loadRawMetadataFromDir(videoDir) : null; const entry = scan.entries[id]; let title: string; let description: string; if (meta) { title = meta.title ?? ""; description = meta.description ?? ""; } else if (entry) { title = entry.title; description = entry.description; } else { noMetadata++; continue; } considered++; const subject = { title: fields.includes("title") ? title : "", description: fields.includes("description") ? description : "", }; if (classifyAgainstFilter(compiled, subject) !== "text") continue; if (!downloaded) { notDownloaded.push(id); matched.push({ id, title, downloaded, alreadyKept: false, marked: false }); continue; } const kept = await isDoNotClean(videoDir); if (kept) { alreadyKept++; } else if (!dryRun) { await setDoNotClean(videoDir, true, note); marked++; } matched.push({ id, title, downloaded, alreadyKept: kept, marked: !kept && !dryRun, }); } const known = new Set([...dirs, ...Object.keys(scan.entries)]); const listed = new Set([...playlistIds, ...Object.keys(roster.entries)]); let unscanned = 0; for (const id of listed) if (!known.has(id)) unscanned++; return { pattern, fields, considered, matched, marked, alreadyKept, notDownloaded, unscanned, noMetadata, dryRun, }; }