Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a876362bfbc5e25da67265d18f90631a961f6c6e
parent 265b55920af9ce9d52ff5b9ef23842cc5cbf0e0b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri, 25 Sep 2026 12:38:52 -0400

common: keepVideosMatching — the do-not-clean marker over a channel's title/description matches

The download filter's own compile + subject (title + "\n" + description),
text from metadata.info.json then the metadata-scan store; matches with no
data/<id>/ are reported (notDownloaded), never created; ids with no text are
counted (unscanned). Media guard first: an unmounted drive is not an empty
channel.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/controller/keepVideosMatching.test.ts | 223+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/keepVideosMatching.ts | 247+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 470 insertions(+), 0 deletions(-)

diff --git a/common/controller/keepVideosMatching.test.ts b/common/controller/keepVideosMatching.test.ts @@ -0,0 +1,223 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, rm, stat, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import { loadDoNotClean, setDoNotClean } from "../lib/doNotClean-server"; +import { KeepVideosError, keepVideosMatching } from "./keepVideosMatching"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/keepVideosMatching.test.ts + +const SLUG = "ch"; + +type Seed = { + // id -> [title, description] written to data/<id>/metadata.info.json + dirs?: Record<string, [string, string]>; + // id -> [title, description] written to metadata-scan.json only + scanned?: Record<string, [string, string]>; + // bare ids written to the playlist file + playlist?: string[]; +}; + +async function withChannel( + seed: Seed, + fn: (paths: Paths, channelDir: string) => Promise<void>, +): Promise<void> { + const root = await mkdtemp(path.join(tmpdir(), "ttb-keep-videos-")); + const paths = { channelsDir: path.join(root, "channels") } as Paths; + const channelDir = path.join(paths.channelsDir, SLUG); + try { + await mkdir(path.join(channelDir, "data"), { recursive: true }); + await writeFile( + path.join(channelDir, "config.json"), + JSON.stringify({ name: "Ch", url: "https://www.youtube.com/@ch" }), + ); + for (const [id, [title, description]] of Object.entries(seed.dirs ?? {})) { + const dir = path.join(channelDir, "data", id); + await mkdir(dir, { recursive: true }); + await writeFile( + path.join(dir, "metadata.info.json"), + JSON.stringify({ id, title, description }), + ); + } + const entries: Record<string, unknown> = {}; + for (const [id, [title, description]] of Object.entries(seed.scanned ?? {})) { + entries[id] = { + title, + description, + uploadDate: "20240101", + scannedAt: "2026-09-25T00:00:00.000Z", + }; + } + await writeFile( + path.join(channelDir, "metadata-scan.json"), + JSON.stringify({ version: 1, updatedAt: "", lastRun: null, entries, errors: {} }), + ); + await writeFile( + path.join(channelDir, "playlist"), + (seed.playlist ?? []) + .map((id) => `https://www.youtube.com/watch?v=${id}`) + .join("\n"), + ); + await fn(paths, channelDir); + } finally { + await rm(root, { recursive: true, force: true }); + } +} + +const dirOf = (channelDir: string, id: string) => + path.join(channelDir, "data", id); + +test("a title match and a description-only match are both marked, case-insensitively", async () => { + await withChannel( + { + dirs: { + aaaaaaaaaaa: ["Reacting to THEQUARTERING", "nothing here"], + bbbbbbbbbbb: ["Some other video", "featuring thequartering at 10:00"], + ccccccccccc: ["Unrelated", "unrelated"], + }, + }, + async (paths, channelDir) => { + const r = await keepVideosMatching({ + paths, + channelSlug: SLUG, + pattern: "TheQuartering", + }); + assert.equal(r.considered, 3); + assert.equal(r.marked, 2); + assert.deepEqual( + r.matched.map((m) => [m.id, m.marked, m.downloaded]), + [ + ["aaaaaaaaaaa", true, true], + ["bbbbbbbbbbb", true, true], + ], + ); + const rec = await loadDoNotClean(dirOf(channelDir, "aaaaaaaaaaa")); + assert.equal(rec?.note, "keep-videos: matched /TheQuartering/i"); + assert.equal(await loadDoNotClean(dirOf(channelDir, "ccccccccccc")), null); + }, + ); +}); + +test("an already-kept video is counted and not rewritten", async () => { + await withChannel( + { dirs: { aaaaaaaaaaa: ["TheQuartering live", ""] } }, + async (paths, channelDir) => { + const dir = dirOf(channelDir, "aaaaaaaaaaa"); + await setDoNotClean(dir, true, "by hand"); + const markerFile = path.join(dir, "do-not-clean.json"); + const before = (await stat(markerFile)).mtimeMs; + await new Promise((r) => setTimeout(r, 20)); + const r = await keepVideosMatching({ + paths, + channelSlug: SLUG, + pattern: "thequartering", + }); + assert.equal(r.marked, 0); + assert.equal(r.alreadyKept, 1); + assert.deepEqual(r.matched[0], { + id: "aaaaaaaaaaa", + title: "TheQuartering live", + downloaded: true, + alreadyKept: true, + marked: false, + }); + assert.equal((await stat(markerFile)).mtimeMs, before); + assert.equal((await loadDoNotClean(dir))?.note, "by hand"); + }, + ); +}); + +test("a scan-only match is reported notDownloaded and no dir is created; unscanned counts the gap", async () => { + await withChannel( + { + scanned: { sssssssssss: ["TheQuartering clip", ""] }, + playlist: ["sssssssssss", "uuuuuuuuuuu", "vvvvvvvvvvv"], + }, + async (paths, channelDir) => { + const r = await keepVideosMatching({ + paths, + channelSlug: SLUG, + pattern: "thequartering", + }); + assert.deepEqual(r.notDownloaded, ["sssssssssss"]); + assert.equal(r.marked, 0); + assert.equal(r.matched[0].downloaded, false); + assert.equal(r.unscanned, 2); + await assert.rejects(stat(dirOf(channelDir, "sssssssssss")), { + code: "ENOENT", + }); + }, + ); +}); + +test("a dry run reports the matches and writes nothing", async () => { + await withChannel( + { dirs: { aaaaaaaaaaa: ["TheQuartering", ""] } }, + async (paths, channelDir) => { + const r = await keepVideosMatching({ + paths, + channelSlug: SLUG, + pattern: "thequartering", + dryRun: true, + }); + assert.equal(r.dryRun, true); + assert.equal(r.matched.length, 1); + assert.equal(r.matched[0].marked, false); + assert.equal(r.marked, 0); + assert.equal(await loadDoNotClean(dirOf(channelDir, "aaaaaaaaaaa")), null); + }, + ); +}); + +test('fields: ["title"] ignores a description hit', async () => { + await withChannel( + { + dirs: { + aaaaaaaaaaa: ["Plain title", "TheQuartering in the description"], + bbbbbbbbbbb: ["TheQuartering in the title", ""], + }, + }, + async (paths) => { + const r = await keepVideosMatching({ + paths, + channelSlug: SLUG, + pattern: "thequartering", + fields: ["title"], + }); + assert.deepEqual(r.fields, ["title"]); + assert.deepEqual( + r.matched.map((m) => m.id), + ["bbbbbbbbbbb"], + ); + }, + ); +}); + +test("a bad pattern, an unknown field and an unknown channel throw the typed error", async () => { + await withChannel({ dirs: {} }, async (paths) => { + await assert.rejects( + keepVideosMatching({ paths, channelSlug: SLUG, pattern: "(" }), + (e) => e instanceof KeepVideosError && /invalid pattern/.test(e.message), + ); + await assert.rejects( + keepVideosMatching({ paths, channelSlug: SLUG, pattern: "(\\w+\\s?)*$" }), + (e) => e instanceof KeepVideosError && /nested quantifier/.test(e.message), + ); + await assert.rejects( + keepVideosMatching({ + paths, + channelSlug: SLUG, + pattern: "x", + fields: ["tags" as "title"], + }), + KeepVideosError, + ); + await assert.rejects( + keepVideosMatching({ paths, channelSlug: "nope", pattern: "x" }), + (e) => e instanceof KeepVideosError && /not found/.test(e.message), + ); + }); +}); diff --git a/common/controller/keepVideosMatching.ts b/common/controller/keepVideosMatching.ts @@ -0,0 +1,247 @@ +import path from "node:path"; +import { readFile } from "node:fs/promises"; +import type { Paths } from "../lib/paths"; +import { assertChannelMediaReachable } from "../lib/channelMedia"; +import { isDoNotClean, setDoNotClean } from "../lib/doNotClean-server"; +import { + classifyAgainstFilter, + compileDownloadFilter, + downloadFilterPatternProblem, +} from "../lib/downloadFilters"; +import { loadRawMetadataFromDir } from "../lib/transcripts-server"; +import { extractVideoId } from "../lib/videoId"; +import { channelExists, readChannelConfig } from "./channels"; +import { listChannelVideoIds } from "./keptVideos"; +import { loadMetadataScan } from "./metadataScanStore"; +import { loadRoster } from "./rosterStore"; + +// BULK "KEEP THE VIDEO": set the per-video do-not-clean marker on every video of +// one channel whose title / description matches a pattern. The marker is the +// same `data/<id>/do-not-clean.json` the video page's toggle writes +// (lib/doNotClean-server.ts), and every cleaner already honours it — the clean +// sweep, extra-format cleanup, wrong-format removal, the superseded-subs purge +// and saved-video eviction. This module adds no new kind of protection; it is +// the loop the toggle lacks. +// +// "MATCHES" MEANS WHAT THE DOWNLOAD FILTER MEANS BY IT. The pattern is compiled +// by `compileDownloadFilter` (case-insensitive, the same source a channel's +// `downloadFilter.include` holds) and tested by `classifyAgainstFilter` over +// `downloadFilterText` — title + "\n" + description, the description capped +// exactly as the filter caps it. So a pattern that keeps the right videos here +// is one that would download the right ones there, and vice versa. There is no +// second matcher. `fields` narrows the SUBJECT, not the matcher: a field left +// out is passed as empty. +// +// TEXT COMES FROM TWO PLACES, NEVER A THIRD. A downloaded video's own +// `metadata.info.json` first; otherwise the channel's metadata-scan store +// (`metadata-scan.json`), which is how a scanned-but-undownloaded video has a +// title at all. An id known only from `playlist` / `roster.json` has no text and +// is COUNTED (`unscanned`) — run `metadata-scan` first to close that gap. +// +// A MATCH WITH NO `data/<id>/` IS REPORTED, NEVER CREATED. The marker lives in +// the video's dir and `setDoNotClean` does not mkdir. Creating the dir here +// would also break the metadata scan's invariant that it never makes one +// (metadataScanStore.ts) — every enumerator reads "has a dir" as "was fetched". +// `notDownloaded` lists them; `download-missing` takes the ids, and a re-run +// then marks them. +// +// AN UNMOUNTED DRIVE IS NOT AN EMPTY CHANNEL. `listChannelVideoIds` swallows +// ENOENT on data/, which on a relocated channel would turn every downloaded +// match into `notDownloaded`. The media guard is asked first, and refuses. + +export const KEEP_VIDEOS_FIELDS = ["title", "description"] as const; +export type KeepVideosField = (typeof KEEP_VIDEOS_FIELDS)[number]; + +// Thrown for a refusal the CALLER can fix (a bad pattern, an unknown channel, +// a bad field). The editor action turns it into its `{ ok: false, error }`, and +// the ops route into a 400. Anything else is an I/O failure. +export class KeepVideosError extends Error { + constructor(message: string) { + super(message); + this.name = "KeepVideosError"; + } +} + +export type KeepVideosOptions = { + paths: Paths; + channelSlug: string; + pattern: string; + fields?: readonly KeepVideosField[]; + note?: string; + dryRun?: boolean; +}; + +export type KeepVideosMatch = { + id: string; + title: string; + // Has a data/<id>/ dir — the only kind that can carry the marker. + downloaded: boolean; + // Already carried a marker before this call; left untouched. + alreadyKept: boolean; + // This call wrote the marker (always false on a dry run). + marked: boolean; +}; + +export type KeepVideosResult = { + pattern: string; + fields: KeepVideosField[]; + // Ids that had text to test: every data dir plus every scan entry. + considered: number; + matched: KeepVideosMatch[]; + marked: number; + alreadyKept: number; + // Matched ids with no data/<id>/ — download them, then re-run. + notDownloaded: string[]; + // Ids in playlist / roster.json with neither a dir nor a scan entry: the + // coverage gap no pattern can see into. `metadata-scan` closes it. + unscanned: number; + // Data dirs with no metadata.info.json and no scan entry — they have a dir + // but no text, so they could not be tested. Rare; counted, not guessed. + noMetadata: number; + dryRun: boolean; +}; + +function resolveFields( + fields: readonly KeepVideosField[] | undefined, +): KeepVideosField[] { + if (fields === undefined) return [...KEEP_VIDEOS_FIELDS]; + if (fields.length === 0) { + throw new KeepVideosError( + `"fields" must name at least one of ${KEEP_VIDEOS_FIELDS.join(", ")}`, + ); + } + for (const f of fields) { + if (!(KEEP_VIDEOS_FIELDS as readonly string[]).includes(f)) { + throw new KeepVideosError( + `"fields" may only contain ${KEEP_VIDEOS_FIELDS.join(", ")} (got "${f}")`, + ); + } + } + return KEEP_VIDEOS_FIELDS.filter((f) => fields.includes(f)); +} + +async function readPlaylistIds(channelDir: string): Promise<string[]> { + const raw = await readFile(path.join(channelDir, "playlist"), "utf8").catch( + () => "", + ); + const ids: string[] = []; + for (const line of raw.split("\n")) { + const url = line.trim(); + if (!url) continue; + const id = extractVideoId(url); + if (id) ids.push(id); + } + return ids; +} + +export async function keepVideosMatching( + opts: KeepVideosOptions, +): Promise<KeepVideosResult> { + const { paths, channelSlug } = opts; + const pattern = opts.pattern.trim(); + const dryRun = opts.dryRun === true; + const fields = resolveFields(opts.fields); + + if (!pattern) throw new KeepVideosError('"match" must be a non-empty pattern'); + // The download filter form's own guard: the pattern runs over every title + // and description on the channel, on the server's one thread. + const problem = downloadFilterPatternProblem(pattern); + if (problem) throw new KeepVideosError(`invalid pattern: ${problem}`); + const compiled = compileDownloadFilter({ include: pattern }); + if (!compiled?.include) { + let why = "it does not parse"; + try { + new RegExp(pattern, "i"); + } catch (e) { + why = (e as Error).message; + } + throw new KeepVideosError(`invalid pattern /${pattern}/i: ${why}`); + } + + if (!(await channelExists(paths, channelSlug))) { + throw new KeepVideosError(`Channel "${channelSlug}" not found`); + } + const config = await readChannelConfig(paths, channelSlug); + await assertChannelMediaReachable(paths, channelSlug, config); + + const channelDir = path.join(paths.channelsDir, channelSlug); + const [dirIds, scan, roster, playlistIds] = await Promise.all([ + listChannelVideoIds(paths, channelSlug), + loadMetadataScan(paths, channelSlug), + loadRoster(paths, channelSlug), + readPlaylistIds(channelDir), + ]); + const dirs = new Set(dirIds); + const ids = [...new Set([...dirIds, ...Object.keys(scan.entries)])].sort(); + + const matched: KeepVideosMatch[] = []; + const notDownloaded: string[] = []; + let marked = 0; + let alreadyKept = 0; + let noMetadata = 0; + let considered = 0; + const note = opts.note?.trim() || `keep-videos: matched /${pattern}/i`; + + for (const id of ids) { + const downloaded = dirs.has(id); + const videoDir = path.join(channelDir, "data", id); + const meta = downloaded ? await loadRawMetadataFromDir(videoDir) : null; + const entry = scan.entries[id]; + let title: string; + let description: string; + if (meta) { + title = meta.title ?? ""; + description = meta.description ?? ""; + } else if (entry) { + title = entry.title; + description = entry.description; + } else { + noMetadata++; + continue; + } + considered++; + const subject = { + title: fields.includes("title") ? title : "", + description: fields.includes("description") ? description : "", + }; + if (classifyAgainstFilter(compiled, subject) !== "text") continue; + + if (!downloaded) { + notDownloaded.push(id); + matched.push({ id, title, downloaded, alreadyKept: false, marked: false }); + continue; + } + const kept = await isDoNotClean(videoDir); + if (kept) { + alreadyKept++; + } else if (!dryRun) { + await setDoNotClean(videoDir, true, note); + marked++; + } + matched.push({ + id, + title, + downloaded, + alreadyKept: kept, + marked: !kept && !dryRun, + }); + } + + const known = new Set([...dirs, ...Object.keys(scan.entries)]); + const listed = new Set([...playlistIds, ...Object.keys(roster.entries)]); + let unscanned = 0; + for (const id of listed) if (!known.has(id)) unscanned++; + + return { + pattern, + fields, + considered, + matched, + marked, + alreadyKept, + notDownloaded, + unscanned, + noMetadata, + dryRun, + }; +}