Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a17d0ef2127aabd94866aa60a4be102df3fef883
parent ee832ca38a2623e816facc37de0f17b756b615e0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 21 Sep 2026 13:20:01 -0400

tags S3.5: the ops door — tag-videos and tags, plus --file

Two adapters, per the rule the layer rests on: each validates a body and calls
ONE existing action, so an agent over HTTP and an operator clicking Pin get the
same write and the same refusal from the same code.

  POST /api/ops/tag-videos { tag, op, videos:[{slug,id}], source? } -> { changed }
  GET  /api/ops/tags                 -> defs + pin/suppression counts
  GET  /api/ops/tags?tag=<id>        -> that tag's assignments, with provenance
  POST /api/ops/tags { op: define|remove, tag }

Four things the shapes are saying:

  * `remove` UNPINS and nothing else; rejecting a rule's hit is `suppress`, a
    different statement, recorded as one.
  * The GET counts are what somebody ASSERTED. Rule hits are re-derived at every
    index build and never stored, so a file read cannot honestly report them —
    the built numbers are in a site's published /tags.json.
  * `op: remove` drops the DEFINITION only. Assignments survive: an assignment
    is a fact about a video, and deleting somebody's facts as a side effect of
    tidying a label is the surprise nobody wants.
  * A def posted to `define` goes through sanitizeTagsConfig, the same coercion
    the editor's own save uses — so a rule whose regex does not compile comes
    back DISABLED WITH A REASON rather than rejected or silently dropped.

`pnpm ops` gains both, a `get tags [<id>]` read (the first noun whose argument
is optional), and `--file <path>`: a four-thousand-id tag-videos body is written
by a script, not typed into a shell argument by a model. parseArgs stays pure —
it reports the file it would read and main() reads it — so the new cases in
scripts/archilyzer-ops.test.mjs need no filesystem. ARCHILYZER_AGENT names who
is asking; the CLI defaults to `agent:cli`, the route to `agent:ops`.

Co-Authored-By: Claude Opus <noreply@anthropic.com>

Diffstat:
Aeditor/app/api/ops/tag-videos/route.ts | 74++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/app/api/ops/tags/route.ts | 122+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/tags/actions.ts | 38++++++++++++++++++++++++++++++++++++++
Mscripts/archilyzer-ops.mjs | 96++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-----
Mscripts/archilyzer-ops.test.mjs | 48++++++++++++++++++++++++++++++++++++++++++++++++
5 files changed, 373 insertions(+), 5 deletions(-)

diff --git a/editor/app/api/ops/tag-videos/route.ts b/editor/app/api/ops/tag-videos/route.ts @@ -0,0 +1,74 @@ +import { NextResponse } from "next/server"; +import { isValidChannelSlug } from "yt-dlp-transcript-common/controller/channels"; +import { applyTagAssignmentsAction } from "../../../tags/actions"; +import { OpsInputError, oneOf, ops, optString, opsFail, reqString } from "../_lib"; + +export const dynamic = "force-dynamic"; + +// POST { tag, op: "add"|"remove"|"suppress"|"unsuppress", +// videos: [{ slug, id }], source? } +// -> { ok: true, changed } +// +// AN ADAPTER: one call to applyTagAssignmentsAction, which is the same action +// the /tags preview rows, the video panel and the bulk bar call. An agent +// tagging over HTTP and an operator clicking Pin perform the same write, and +// the only difference is the provenance recorded beside it. +// +// add pin +// remove UNPIN ONLY — a rule hit survives. "This tag does not belong +// here" is `suppress`, which is a different statement and is +// recorded as one. +// suppress reject a rule's hit (also unpins) +// unsuppress clear a rejection (does not pin) +// +// ONE WRITE FOR THE WHOLE BATCH, whatever its length — the store reads, +// mutates and renames tags.json once per call. Send the ids in one request +// rather than one request per id; `pnpm ops tag-videos --file ids.json` exists +// for exactly the list that is too big to type. +// +// `source` defaults to `agent:ops`: a call arriving here is by definition not +// the browser in front of the editor, and an unattributed pin is the thing the +// provenance field exists to prevent. The CLI sends `agent:$ARCHILYZER_AGENT`. +export async function POST(request: Request) { + return ops(request, ["tag", "op", "videos", "source"], async (body) => { + const tag = reqString(body, "tag"); + const op = oneOf(body, "op", [ + "add", + "remove", + "suppress", + "unsuppress", + ] as const); + const raw = body.videos; + if (!Array.isArray(raw) || raw.length === 0) { + throw new OpsInputError( + '"videos" is required and must be a non-empty array of { slug, id }', + ); + } + const videos = raw.map((entry) => { + if (!entry || typeof entry !== "object" || Array.isArray(entry)) { + throw new OpsInputError('each "videos" entry must be { slug, id }'); + } + const e = entry as Record<string, unknown>; + const slug = typeof e.slug === "string" ? e.slug.trim() : ""; + const id = typeof e.id === "string" ? e.id.trim() : ""; + if (!slug || !id) { + throw new OpsInputError('each "videos" entry needs a slug and an id'); + } + // The same door check every slug-taking route applies (see _lib's + // reqSlug): a slug reaches a path.join under channelsDir, and the readers + // swallow their own errors, so a traversing segment would fail silently. + if (!isValidChannelSlug(slug)) { + throw new OpsInputError(`"${slug}" is not a valid channel slug`); + } + return { channelSlug: slug, id }; + }); + const result = await applyTagAssignmentsAction({ + op, + tag, + videos, + source: optString(body, "source") ?? "agent:ops", + }); + if (!result.ok) return opsFail(result.error); + return NextResponse.json({ ok: true, changed: result.changed }); + }); +} diff --git a/editor/app/api/ops/tags/route.ts b/editor/app/api/ops/tags/route.ts @@ -0,0 +1,122 @@ +import { NextResponse } from "next/server"; +import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { sanitizeTagsConfig } from "yt-dlp-transcript-common/lib/curatedTags"; +import { readGlobalTags } from "../../../../lib/tagsStore"; +import { defineTagAction, removeTagDefAction } from "../../../tags/actions"; +import { OpsInputError, oneOf, ops, opsAuth, opsFail } from "../_lib"; + +export const dynamic = "force-dynamic"; + +// GET /api/ops/tags -> { ok, tags: [ def + { pinned, suppressed } ] } +// GET /api/ops/tags?tag=<id> -> { ok, tag, assignments: [...] } +// POST /api/ops/tags { op: "define"|"remove", tag } +// +// THE READ SIDE IS THE COUNTS A PERSON OR AN AGENT ASSERTED, and it says so: +// pins and suppressions are stored, RULE HITS ARE NOT. A rule is re-evaluated +// at every index build and its hits live on the records, so the only honest +// answer from a file read is "what somebody said". For the built numbers, read +// a site's published /tags.json after a build. +export async function GET(request: Request) { + const denied = opsAuth(request); + if (denied) return denied; + const config = readGlobalTags(getPaths()); + const wanted = new URL(request.url).searchParams.get("tag"); + + if (wanted) { + const assignments: Array<{ + channelSlug: string; + id: string; + state: "pinned" | "suppressed"; + source?: string; + setAt?: string; + }> = []; + for (const [key, assignment] of Object.entries(config.assignments)) { + const [channelSlug, ...rest] = key.split("/"); + const id = rest.join("/"); + const provenance = assignment.sources?.[wanted]; + const state = (assignment.manual ?? []).includes(wanted) + ? ("pinned" as const) + : (assignment.suppressed ?? []).includes(wanted) + ? ("suppressed" as const) + : null; + if (!state) continue; + assignments.push({ + channelSlug, + id, + state, + ...(provenance + ? { source: provenance.source, setAt: provenance.setAt } + : {}), + }); + } + return NextResponse.json({ + ok: true, + tag: wanted, + defined: config.tags.some((t) => t.id === wanted), + assignments, + }); + } + + const counts: Record<string, { pinned: number; suppressed: number }> = {}; + for (const assignment of Object.values(config.assignments)) { + for (const id of assignment.manual ?? []) { + counts[id] = counts[id] ?? { pinned: 0, suppressed: 0 }; + counts[id].pinned += 1; + } + for (const id of assignment.suppressed ?? []) { + counts[id] = counts[id] ?? { pinned: 0, suppressed: 0 }; + counts[id].suppressed += 1; + } + } + return NextResponse.json({ + ok: true, + tags: config.tags.map((t) => ({ + ...t, + pinned: counts[t.id]?.pinned ?? 0, + suppressed: counts[t.id]?.suppressed ?? 0, + })), + // Ids carrying assignments that no definition claims — a def deleted after + // the pins were made. They keep riding on records (losing an operator's + // fact silently would be worse) and appear in no published /tags.json. + undefinedTags: Object.keys(counts).filter( + (id) => !config.tags.some((t) => t.id === id), + ), + }); +} + +// `define` UPSERTS a whole definition — id, presentation fields and the rules — +// so an agent seeding a vocabulary sends one object per tag. `remove` takes an +// id and drops the DEFINITION ONLY: assignments survive, because an assignment +// is a fact about a video and a definition is a label for it. +export async function POST(request: Request) { + return ops(request, ["op", "tag"], async (body) => { + const op = oneOf(body, "op", ["define", "remove"] as const); + if (op === "remove") { + const id = typeof body.tag === "string" ? body.tag.trim() : ""; + if (!id) throw new OpsInputError('"tag" must be a tag id string'); + const result = await removeTagDefAction(id); + return result.ok + ? NextResponse.json({ ok: true }) + : opsFail(result.error); + } + const raw = body.tag; + if (!raw || typeof raw !== "object" || Array.isArray(raw)) { + throw new OpsInputError( + '"tag" must be a definition object { id, label?, group?, rules?… }', + ); + } + // Coerced by the frozen model, not by this route: sanitizeTagsConfig is + // what the editor's own save runs through, and a rule whose regex does not + // compile comes back disabled with a reason rather than rejected. + const [def] = sanitizeTagsConfig({ tags: [raw] }).tags; + if (!def) { + throw new OpsInputError( + 'the tag definition has no usable id — ids are lowercase letters, digits, ".", "_" and "-", starting with a letter or digit', + ); + } + const result = await defineTagAction(def); + return result.ok + ? NextResponse.json({ ok: true, tag: def }) + : opsFail(result.error); + }); +} diff --git a/editor/app/tags/actions.ts b/editor/app/tags/actions.ts @@ -88,6 +88,44 @@ export async function saveSiteTagDefsAction( return { ok: true }; } +// Upsert ONE definition, keeping the rest of the vocabulary and every +// assignment. The list-shaped save above is what the page uses; this is what a +// caller holding a single tag (the ops route, a seeding script) needs, and +// putting it here rather than in the route keeps the route an adapter. +export async function defineTagAction( + def: CuratedTagDef, +): Promise<TagActionResult> { + if (!TAG_ID_RE.test(def.id ?? "")) { + return { ok: false, error: `"${def.id}" is not a valid tag id` }; + } + const paths = getPaths(); + const current = readGlobalTags(paths); + const tags = current.tags.some((t) => t.id === def.id) + ? current.tags.map((t) => (t.id === def.id ? def : t)) + : [...current.tags, def]; + writeGlobalTags(paths, { ...current, tags }); + revalidatePath("/tags"); + return { ok: true }; +} + +// Drop a DEFINITION. Assignments survive deliberately: an assignment is a fact +// about a video, and effectiveTagsFor keeps a tag whose def has gone (it simply +// never appears in a published /tags.json, which is built from defs). Deleting +// somebody's facts as a side effect of tidying a label would be the surprise. +export async function removeTagDefAction( + id: string, +): Promise<TagActionResult> { + const paths = getPaths(); + const current = readGlobalTags(paths); + const tags = current.tags.filter((t) => t.id !== id); + if (tags.length === current.tags.length) { + return { ok: false, error: `no tag "${id}" is defined` }; + } + writeGlobalTags(paths, { ...current, tags }); + revalidatePath("/tags"); + return { ok: true }; +} + export type TagPreviewRow = { channelSlug: string; id: string; diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -8,11 +8,15 @@ // // USAGE // -// pnpm ops <action> [--json '<body>'] [--wait] [--quiet] +// pnpm ops <action> [--json '<body>' | --file <path>] [--wait] [--quiet] // pnpm ops get channel <slug> [--counts] +// pnpm ops get tags [<tagId>] // pnpm ops list // // ARCHILYZER_EDITOR_URL editor base URL (default http://localhost:3001) +// ARCHILYZER_AGENT who is asking, recorded as the provenance of a +// curated-tag write (`agent:<value>`, default +// `agent:cli`). Ignored by every other action. // WORKER_TOKEN the shared secret the editor is running with. // Unset on the SERVER => every route 503s; unset here // => every route 401s. @@ -27,6 +31,13 @@ // pnpm ops refresh-report --json '{"all":true}' // pnpm ops relocate --json '{"slugs":["x"],"locationId":"platter"}' // pnpm ops get channel the-quartering +// pnpm ops tags --json '{"op":"define","tag":{"id":"eva-collab","label":"Collab"}}' +// pnpm ops tag-videos --file ids.json +// pnpm ops get tags eva-collab +// +// --file reads the BODY from a JSON file, which is how a big one gets sent: a +// four-thousand-id tag-videos body is written by a script, not typed by a model +// into a shell argument. --json and --file are mutually exclusive. // // --wait follows /api/jobs/<jobId>/log to the end for a job-starting action and // exits 0 only if the job finished `done`. Without it the command returns as @@ -36,6 +47,8 @@ // The response JSON is printed verbatim on stdout (log lines from --wait go to // stderr), so `pnpm ops … | jq` works. +import { readFile } from "node:fs/promises"; + const DEFAULT_URL = "http://localhost:3001"; // The read-side routes, reachable as `get <noun> <arg>`. Kept tiny and explicit: @@ -46,8 +59,16 @@ const GETTERS = { // opt-in for the same reason the route makes it opt-in. channel: (slug, counts) => `/api/ops/channel/${encodeURIComponent(slug)}${counts ? "?counts=1" : ""}`, + // No argument: every definition with its pin/suppression counts. With one: a + // single tag's assignments, each carrying the provenance of the pin. + tags: (tag) => + tag ? `/api/ops/tags?tag=${encodeURIComponent(tag)}` : "/api/ops/tags", }; +// Nouns whose read takes no argument. `get channel` without a slug is a +// mistake; `get tags` without one is the whole vocabulary. +const GET_ARG_OPTIONAL = new Set(["tags"]); + const ACTIONS = [ "channel-priority", "channel-config", @@ -64,11 +85,22 @@ const ACTIONS = [ "relocate-back", "evict-clips", "lane", + // The curated-tag writers. `tags` edits the vocabulary (define/remove); + // `tag-videos` pins, unpins, suppresses or unsuppresses one tag over a batch + // of videos in ONE write. + "tags", + "tag-videos", ]; +// The provenance a tag write from this CLI carries. Everything else ignores it. +function agentSource() { + return `agent:${process.env.ARCHILYZER_AGENT || "cli"}`; +} + export function parseArgs(argv) { const positional = []; let json = null; + let file = null; let wait = false; let quiet = false; let counts = false; @@ -87,6 +119,13 @@ export function parseArgs(argv) { } } else if (arg.startsWith("--json=")) { json = arg.slice("--json=".length); + } else if (arg === "--file") { + file = argv[++i]; + if (file === undefined) { + return { error: "--file needs a path to a JSON file" }; + } + } else if (arg.startsWith("--file=")) { + file = arg.slice("--file=".length); } else if (arg === "--help" || arg === "-h") { return { help: true }; } else if (arg.startsWith("-")) { @@ -96,6 +135,9 @@ export function parseArgs(argv) { } } if (positional.length === 0) return { help: true }; + if (json !== null && file !== null) { + return { error: "--json and --file are mutually exclusive" }; + } let body = {}; if (json !== null) { try { @@ -117,7 +159,9 @@ export function parseArgs(argv) { error: `get: unknown noun "${noun ?? ""}" — known: ${Object.keys(GETTERS).join(", ")}`, }; } - if (!positional[2]) return { error: `get ${noun}: needs an argument` }; + if (!positional[2] && !GET_ARG_OPTIONAL.has(noun)) { + return { error: `get ${noun}: needs an argument` }; + } return { method: "GET", path: GETTERS[noun](positional[2], counts), @@ -136,18 +180,34 @@ export function parseArgs(argv) { error: `"${action}" takes no positional arguments — pass its body with --json`, }; } - return { method: "POST", path: `/api/ops/${action}`, body, wait, quiet }; + // The body may still arrive from --file; main() reads it, because parseArgs + // is pure (and unit-tested without a filesystem). + return { + method: "POST", + path: `/api/ops/${action}`, + body, + ...(file !== null ? { bodyFile: file } : {}), + // WHO IS ASKING, for the one pair of actions that records it. Defaulted + // here rather than in the route so a human at a terminal and a script under + // a name are told apart; the route's own default (`agent:ops`) covers a + // caller that is neither. + ...(action === "tag-videos" ? { defaultSource: agentSource() } : {}), + wait, + quiet, + }; } export function usage() { return [ - "Usage: pnpm ops <action> [--json '<body>'] [--wait]", + "Usage: pnpm ops <action> [--json '<body>' | --file <path>] [--wait]", " pnpm ops get channel <slug> [--counts]", + " pnpm ops get tags [<tagId>]", " pnpm ops list", "", `Actions: ${ACTIONS.join(", ")}`, "", - "Env: ARCHILYZER_EDITOR_URL (default http://localhost:3001), WORKER_TOKEN", + "Env: ARCHILYZER_EDITOR_URL (default http://localhost:3001), WORKER_TOKEN,", + " ARCHILYZER_AGENT (provenance of a tag write; default \"cli\")", ].join("\n"); } @@ -201,6 +261,32 @@ async function main() { console.log(ACTIONS.join("\n")); return 0; } + if (parsed.bodyFile) { + let raw; + try { + raw = await readFile(parsed.bodyFile, "utf8"); + } catch (e) { + console.error(`--file: ${e.message}`); + return 2; + } + try { + parsed.body = JSON.parse(raw); + } catch (e) { + console.error(`--file ${parsed.bodyFile} is not valid JSON: ${e.message}`); + return 2; + } + if ( + typeof parsed.body !== "object" || + parsed.body === null || + Array.isArray(parsed.body) + ) { + console.error(`--file ${parsed.bodyFile} must hold a JSON object`); + return 2; + } + } + if (parsed.defaultSource && parsed.body && parsed.body.source === undefined) { + parsed.body = { ...parsed.body, source: parsed.defaultSource }; + } const url = `${baseUrl()}${parsed.path}`; const res = await fetch(url, { method: parsed.method, diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs @@ -70,3 +70,51 @@ test("get channel --counts asks for the live on-disk counts", () => { // Off by default: the counts walk every video directory. assert.equal(parseArgs(["get", "channel", "x"]).path, "/api/ops/channel/x"); }); + +// --- curated tags ----------------------------------------------------------- + +test("the tag actions are registered and post to their routes", () => { + const defs = parseArgs(["tags", "--json", '{"op":"remove","tag":"x"}']); + assert.equal(defs.path, "/api/ops/tags"); + const videos = parseArgs([ + "tag-videos", + "--json", + '{"tag":"x","op":"add","videos":[{"slug":"c","id":"v"}]}', + ]); + assert.equal(videos.path, "/api/ops/tag-videos"); + assert.match(usage(), /tags/); +}); + +// A FOUR-THOUSAND-ID BODY IS WRITTEN BY A SCRIPT, NOT TYPED. parseArgs stays +// pure — it reports the file it would read, and main() reads it — so this test +// needs no filesystem. +test("--file names a body file instead of an inline --json", () => { + const p = parseArgs(["tag-videos", "--file", "ids.json"]); + assert.equal(p.bodyFile, "ids.json"); + assert.deepEqual(p.body, {}); + assert.equal(parseArgs(["tag-videos", "--file=ids.json"]).bodyFile, "ids.json"); + assert.match(parseArgs(["tag-videos", "--file"]).error, /needs a path/); + assert.match( + parseArgs(["tag-videos", "--file", "a.json", "--json", "{}"]).error, + /mutually exclusive/, + ); +}); + +// WHO IS ASKING is defaulted for the one action that records provenance, and +// for no other — a pin with no answer to "who said so" is what the field exists +// to prevent. +test("tag-videos carries an agent source; other actions carry none", () => { + assert.match(parseArgs(["tag-videos"]).defaultSource, /^agent:/); + assert.equal(parseArgs(["sync"]).defaultSource, undefined); + assert.equal(parseArgs(["tags"]).defaultSource, undefined); +}); + +test("get tags reads the whole vocabulary, or one tag's assignments", () => { + assert.equal(parseArgs(["get", "tags"]).path, "/api/ops/tags"); + assert.equal( + parseArgs(["get", "tags", "eva-collab"]).path, + "/api/ops/tags?tag=eva-collab", + ); + // The optional argument is per-noun: `get channel` still demands a slug. + assert.match(parseArgs(["get", "channel"]).error, /needs an argument/); +});