Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 46f8f4ef3cbc46e5a20d02342f065ad3d892fd2d
parent e082d44f08184d7b1d5c3199c273033a94b6ed3a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri, 25 Sep 2026 12:38:52 -0400

ops: keep-videos — POST {slug, match, fields?, note?, dryRun?}

An adapter over keepVideosAction; pnpm ops keep-videos; ops-api.spec +1
(dry run writes nothing, only the matching dir gets the marker, a scan-only
match gets no dir, unknown key and bad regex are 400s).

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Aeditor/app/api/ops/keep-videos/route.ts | 57+++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/ops-api.spec.ts | 73+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mscripts/archilyzer-ops.mjs | 3+++
Mscripts/archilyzer-ops.test.mjs | 12++++++++++++
4 files changed, 145 insertions(+), 0 deletions(-)

diff --git a/editor/app/api/ops/keep-videos/route.ts b/editor/app/api/ops/keep-videos/route.ts @@ -0,0 +1,57 @@ +import { NextResponse } from "next/server"; +import { keepVideosAction } from "../../../channels/[slug]/videos/[id]/videoActions"; +import { OpsInputError, ops, optBool, optString, opsFail, reqSlug, reqString } from "../_lib"; + +export const dynamic = "force-dynamic"; + +// POST { slug, match, fields?: ("title"|"description")[], note?, dryRun? } +// -> { ok: true, pattern, fields, considered, matched: [{ id, title, +// downloaded, alreadyKept, marked }], marked, alreadyKept, +// notDownloaded, unscanned, noMetadata, dryRun } +// +// AN ADAPTER: one call to keepVideosAction, the bulk form of the video page's +// "Do not clean" toggle. It writes the same per-video do-not-clean.json marker +// every cleaner already honours; it adds no protection of its own. +// +// `match` IS A DOWNLOAD-FILTER PATTERN, matched the way the filter matches it — +// a case-insensitive regex SOURCE over title + "\n" + description, compiled by +// the same function. A pattern that keeps the right videos here would download +// the right ones as a channel's `downloadFilterInclude`. `fields` narrows the +// subject to one half. +// +// A MATCH WITH NO data/<id>/ IS REPORTED, NOT CREATED. The marker lives in the +// video's dir, and a dir is what every enumerator reads as "fetched". Those ids +// come back in `notDownloaded`: send them to `download-missing`, then run this +// again. A fresh channel wants `metadata-scan` FIRST — an id with no dir and no +// scan entry has no text to match, and is only counted (`unscanned`). +// +// `dryRun: true` answers the same question and writes nothing. +export async function POST(request: Request) { + return ops(request, ["slug", "match", "fields", "note", "dryRun"], async (body) => { + const slug = reqSlug(body, "slug"); + const pattern = reqString(body, "match"); + const raw = body.fields; + let fields: ("title" | "description")[] | undefined; + if (raw !== undefined) { + if ( + !Array.isArray(raw) || + raw.length === 0 || + raw.some((f) => f !== "title" && f !== "description") + ) { + throw new OpsInputError( + '"fields" must be a non-empty array of "title" and/or "description"', + ); + } + fields = raw as ("title" | "description")[]; + } + const result = await keepVideosAction({ + slug, + pattern, + fields, + note: optString(body, "note"), + dryRun: optBool(body, "dryRun"), + }); + if (!result.ok) return opsFail(result.error); + return NextResponse.json({ ok: true, ...result.result }); + }); +} diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts @@ -718,6 +718,79 @@ test("tag-videos refuses a traversing slug, a bad op and an unknown key", async expect(await pathExists("test-transcripts/tags.json")).toBe(false); }); +// --- keep-videos: the bulk do-not-clean toggle -------------------------------- +// +// The marker is the video page's own do-not-clean.json; what this pins is the +// loop around it — the download filter's matcher over title + description, a +// dry run that writes nothing, and a scan-only match that is REPORTED rather +// than given a directory it was never downloaded into. +test("keep-videos marks the matching video only, after a dry run that writes nothing", async ({ + request, +}) => { + await resetData("one-youtube-channel-with-data"); + const slug = "test-youtube"; + const kept = "20240102_keepme12345"; + const other = "20240101_test1234567"; // "Synthetic Test Video" — no needle + const channel = `test-transcripts/channels/${slug}`; + await mkdir(resolvePath(`${channel}/data/${kept}`), { recursive: true }); + await writeFile( + resolvePath(`${channel}/data/${kept}/metadata.info.json`), + JSON.stringify({ id: kept, title: "Reacting to THEQUARTERING", description: "" }), + ); + await writeFile( + resolvePath(`${channel}/metadata-scan.json`), + JSON.stringify({ + version: 1, + updatedAt: "", + lastRun: null, + errors: {}, + entries: { + scanonly123: { + title: "Plain", + description: "with TheQuartering", + uploadDate: "20240103", + scannedAt: "2026-09-25T00:00:00.000Z", + }, + }, + }), + ); + const marker = (id: string) => `${channel}/data/${id}/do-not-clean.json`; + + type KeepBody = OpsResponse & { + matched?: { id: string; marked: boolean }[]; + marked?: number; + notDownloaded?: string[]; + }; + const dry = await ops(request, "keep-videos", { + slug, + match: "thequartering", + dryRun: true, + }); + expect(dry.status, JSON.stringify(dry.body)).toBe(200); + const dryBody = dry.body as KeepBody; + expect(dryBody.matched?.filter((m) => m.id === kept)).toHaveLength(1); + expect(dryBody.marked).toBe(0); + expect(await pathExists(marker(kept))).toBe(false); + + const real = await ops(request, "keep-videos", { slug, match: "thequartering" }); + expect(real.status, JSON.stringify(real.body)).toBe(200); + const body = real.body as KeepBody; + expect(body.marked).toBe(1); + expect(body.notDownloaded).toEqual(["scanonly123"]); + expect(await pathExists(marker(kept))).toBe(true); + expect(await pathExists(marker(other))).toBe(false); + // A scan-only match gets no directory — that would read as "downloaded". + expect(await pathExists(`${channel}/data/scanonly123`)).toBe(false); + + const unknown = await ops(request, "keep-videos", { slug, match: "x", pattern: "x" }); + expect(unknown.status).toBe(400); + expect(unknown.body.error).toContain("unknown key(s): pattern"); + + const badRegex = await ops(request, "keep-videos", { slug, match: "(" }); + expect(badRegex.status).toBe(400); + expect(badRegex.body.error).toContain("invalid pattern"); +}); + // --- the two build routes speak the same body --------------------------------- // // They did not. build-site took `siteIds` (a list) and build-deploy took diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -114,6 +114,9 @@ const ACTIONS = [ // of videos in ONE write. "tags", "tag-videos", + // Set the per-video do-not-clean marker on every video of a channel whose + // title/description matches a download-filter pattern ({slug, match}). + "keep-videos", ]; // The provenance a tag write from this CLI carries. Everything else ignores it. diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs @@ -90,6 +90,18 @@ test("the tag actions are registered and post to their routes", () => { assert.match(usage(), /tags/); }); +test("keep-videos is registered and posts to its route", () => { + const p = parseArgs([ + "keep-videos", + "--json", + '{"slug":"c","match":"TheQuartering","dryRun":true}', + ]); + assert.equal(p.path, "/api/ops/keep-videos"); + assert.deepEqual(p.body, { slug: "c", match: "TheQuartering", dryRun: true }); + // Not a provenance-carrying write: the marker records a note, not a source. + assert.equal(p.defaultSource, undefined); +}); + // A FOUR-THOUSAND-ID BODY IS WRITTEN BY A SCRIPT, NOT TYPED. parseArgs stays // pure — it reports the file it would read, and main() reads it — so this test // needs no filesystem.