Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 5dd228561903b51bf4d51b0e1b3caf29f29ed90f
parent 35fd9573fe941b3e3ece72895009bc27a8f235e1
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 03:11:17 -0400

editor: the reports-prepare job kind (needsMedia, its own queue, replayable), POST /api/ops/reports-prepare and pnpm ops reports-prepare

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/jobs/jobKinds.test.ts | 3+++
Mcommon/jobs/jobKinds.ts | 17+++++++++++++++++
Aeditor/app/api/ops/reports-prepare/route.test.ts | 49+++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/app/api/ops/reports-prepare/route.ts | 23+++++++++++++++++++++++
Meditor/app/jobs/jobReplayRegistry.ts | 8++++++++
Aeditor/app/sites/lib/reportsPrepareAction.ts | 51+++++++++++++++++++++++++++++++++++++++++++++++++++
Mscripts/archilyzer-ops.mjs | 8++++++++
Mscripts/archilyzer-ops.test.mjs | 10++++++++++
8 files changed, 169 insertions(+), 0 deletions(-)

diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts @@ -84,6 +84,8 @@ const ADDED_KINDS: Record<string, { label: string; drainable: boolean }> = { }, // A list of videos persisted one at a time: a drain stops between them. "persist-videos": { label: "Persist videos", drainable: true }, + // A report site's evidence media: one pass, cancelled rather than drained. + "reports-prepare": { label: "Prepare report media", drainable: false }, }; test("added kinds carry their pinned label and drainability", () => { @@ -176,6 +178,7 @@ const STILL_MEDIA = [ "clean-audio-transcribed", "clean-extra-audio-formats", "remove-wrong-format-audio", + "reports-prepare", ]; test("release 17: the text kinds flipped from needsMedia to needsText", () => { diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -630,6 +630,23 @@ const JOB_KINDS: Record<string, JobKindMeta> = { replayable: false, queueKeyStrategy: "custom", }, + // A REPORT SITE'S EVIDENCE MEDIA (publish/reportMedia.ts): every clip its + // published reports cite, cut from the media on disk, and every cited post + // capture copied, into the site's report-media cache before its build. + // `needsMedia`: it opens saved containers and audio, which may be on another + // drive. It spans channels and runs with no `channelSlug`, so the guard in + // runManagedFunction does not ask; the step itself reports a span on an + // unreachable channel as unreachable rather than missing. Its own queue + // (`reports-prepare`): one at a time, behind nothing — not the download + // queues, not the build queue. Replayable: the spec is the site. + "reports-prepare": { + kind: "reports-prepare", + label: "Prepare report media", + drainable: false, + replayable: true, + queueKeyStrategy: "custom", + needsMedia: true, + }, // THE PER-VIDEO WRITERS THAT WERE NOT IN THIS TABLE (release 16 slice RM). // Each runs with a channelSlug and writes under `data/<id>/` — a single // video's transcription (the video page's two Transcribe buttons, and its diff --git a/editor/app/api/ops/reports-prepare/route.test.ts b/editor/app/api/ops/reports-prepare/route.test.ts @@ -0,0 +1,49 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, rm } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; + +// Run with: +// pnpm -C editor exec tsx --test "app/api/ops/reports-prepare/route.test.ts" +// +// The body's shape and the refusals that come before any job, answered from an +// empty temp corpus: no job is queued and no media is touched. + +const ROOT = await mkdtemp(path.join(os.tmpdir(), "reports-prepare-route-")); +// Set before the route (and getPaths, which caches) is first imported. +process.env.WORKER_TOKEN = "test-token"; +process.env.TRANSCRIPTS_DIR = ROOT; +process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); +const { POST } = await import("./route"); +test.after(() => rm(ROOT, { recursive: true, force: true })); + +async function post(body: Record<string, unknown>): Promise<{ status: number; error: string }> { + const res = await POST( + new Request("http://localhost/api/ops/reports-prepare", { + method: "POST", + headers: { + authorization: "Bearer test-token", + "content-type": "application/json", + }, + body: JSON.stringify(body), + }), + ); + return { status: res.status, error: ((await res.json()) as { error?: string }).error ?? "" }; +} + +test("siteId is required, a site id, and the only key", async () => { + assert.match((await post({})).error, /"siteId" is required/); + const bad = await post({ siteId: "../x" }); + assert.equal(bad.status, 400); + assert.match(bad.error, /not a valid site id/); + const unknown = await post({ siteId: "demo-site", reportId: "r1" }); + assert.equal(unknown.status, 400); + assert.match(unknown.error, /unknown key\(s\): reportId — this route accepts siteId/); +}); + +test("a site that does not exist is refused before any job", async () => { + const r = await post({ siteId: "demo-site" }); + assert.equal(r.status, 400); + assert.match(r.error, /No site "demo-site"/); +}); diff --git a/editor/app/api/ops/reports-prepare/route.ts b/editor/app/api/ops/reports-prepare/route.ts @@ -0,0 +1,23 @@ +import { reportsPrepareAction } from "../../../sites/lib/reportsPrepareAction"; +import { jobResponse, OpsInputError, ops, reqString } from "../_lib"; +import { isValidSiteId } from "yt-dlp-transcript-common/lib/site"; + +export const dynamic = "force-dynamic"; + +// POST { siteId: string } -> { ok: true, jobId } +// +// An adapter: one call to the action the site's Reports tab posts. Cuts every +// clip and copies every post capture the site's published reports cite into +// its report-media cache, as one `reports-prepare` job on its own queue. The +// job fails, naming each citation, when any lacks its media. +export async function POST(request: Request) { + return ops(request, ["siteId"], async (body) => { + const siteId = reqString(body, "siteId"); + if (!isValidSiteId(siteId)) { + throw new OpsInputError( + `"${siteId}" is not a valid site id (lowercase letters, digits and "-"; must start with a letter or digit)`, + ); + } + return jobResponse(await reportsPrepareAction(siteId)); + }); +} diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -54,6 +54,7 @@ import { redownloadShortAudioBucketAction, } from "../channels/[slug]/incompleteTranscriptActions"; import { replayFetchWindowAction } from "../channels/[slug]/videos/[id]/videoActions"; +import { reportsPrepareAction } from "../sites/lib/reportsPrepareAction"; export type ReplayHandler = (spec: JobSpec) => Promise<StreamActionResult>; @@ -89,6 +90,13 @@ function params(spec: JobSpec): { } export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { + // A report site's evidence media. The spec's slug is the SITE (the job has + // no channel); a re-run re-reads the site's reports and cuts only what its + // cache does not already hold. + "reports-prepare": (spec) => { + const { p } = params(spec); + return reportsPrepareAction(str(p.siteId) ?? spec.slug); + }, // A clip window sourced for another tool. Replay RE-DERIVES from disk like // every bucket job does: if the window (or a wider one covering it) has // arrived since, the retry says so neutrally rather than paying twice. diff --git a/editor/app/sites/lib/reportsPrepareAction.ts b/editor/app/sites/lib/reportsPrepareAction.ts @@ -0,0 +1,51 @@ +"use server"; + +import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { isValidSiteId, listSiteIds } from "yt-dlp-transcript-common/lib/site"; +import { + runManagedFunction, + type StreamActionResult, +} from "yt-dlp-transcript-common/jobs/streamCommand"; +import { + formatReportMediaProblems, + prepareReportMedia, +} from "yt-dlp-transcript-common/publish/reportMedia"; + +// Its own queue: one prepare at a time, waiting on no download and no build. +const REPORTS_PREPARE_QUEUE = "reports-prepare"; + +// PREPARE A REPORT SITE'S EVIDENCE MEDIA as a job: every clip its published +// reports cite, cut from the media on disk, and every cited post capture, +// copied into `.export-index/sites/<siteId>/report-media/` before its build. +// The work is publish/reportMedia.ts's, the same `archilyzer reports prepare` +// runs. The job FAILS when any citation lacks its media — the log names each +// one — and the manifest it writes carries the same list. +// +// The spec's `slug` is the SITE id (a spec needs one, and this job belongs to +// no channel); the replay handler reads `params.siteId`. +export async function reportsPrepareAction( + siteId: string, +): Promise<StreamActionResult> { + const paths = getPaths(); + const id = siteId.trim(); + if (!isValidSiteId(id)) { + return { ok: false, error: `"${id}" is not a valid site id` }; + } + if (!listSiteIds(paths).includes(id)) { + return { ok: false, error: `No site "${id}"` }; + } + return runManagedFunction({ + kind: "reports-prepare", + queueKey: REPORTS_PREPARE_QUEUE, + paths, + spec: { kind: "reports-prepare", slug: id, params: { siteId: id } }, + fn: async (onLog, signal) => { + const index = await prepareReportMedia({ siteId: id, paths, onLog, signal }); + if (index.problems.length === 0) return; + for (const line of formatReportMediaProblems(index.problems)) onLog(line); + throw new Error( + `${index.problems.length} problem(s): the site's evidence media is not complete`, + ); + }, + }); +} diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -39,6 +39,7 @@ // pnpm ops deploy-hub --wait // pnpm ops build-homepage --json '{"deploy":true}' --wait // pnpm ops deploy-homepage --json '{"preview":"refresh"}' --wait +// pnpm ops reports-prepare --json '{"siteId":"demo-site"}' --wait // pnpm ops get channel the-quartering // pnpm ops tags --json '{"op":"define","tag":{"id":"eva-collab","label":"Collab"}}' // pnpm ops tag-videos --file ids.json @@ -124,6 +125,9 @@ const ACTIONS = [ "relocate", "relocate-back", "evict-clips", + // A report site's evidence media: cut every cited clip and copy every cited + // post capture into the site's report-media cache ({siteId}). + "reports-prepare", "lane", // The curated-tag writers. `tags` edits the vocabulary (define/remove); // `tag-videos` pins, unpins, suppresses or unsuppresses one tag over a batch @@ -317,6 +321,10 @@ export function usage() { 'build-site, build-deploy and deploy-site all take "siteId" (one) or', ' "siteIds" (a list).', "", + 'reports-prepare cuts every clip and copies every post capture a site\'s', + ' published reports cite into its report-media cache, before its build:', + ' {"siteId"}. The job fails, naming each one, when a citation lacks media.', + "", 'retry-bucket runs one bucket of a channel\'s report as one job, past any', ' lane hold: {"slug", "bucket"}. "ids": [...] runs only those videos, and', " every one must be in the bucket (a stray id is refused, named); a job run", diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs @@ -408,6 +408,16 @@ test("capture-posts is a POST to its route, named in the usage", () => { assert.match(usage(), /unless\s+"articles": false/); }); +// A report site's evidence media: a POST to its route, the body passed through +// untouched — the route judges the site id. +test("reports-prepare is a POST to its route, named in the usage", () => { + const p = parseArgs(["reports-prepare", "--json", '{"siteId":"demo-site"}']); + assert.equal(p.method, "POST"); + assert.equal(p.path, "/api/ops/reports-prepare"); + assert.deepEqual(p.body, { siteId: "demo-site" }); + assert.match(usage(), /reports-prepare cuts every clip/); +}); + test("persist-videos is a POST to its route, named in the usage", () => { const p = parseArgs([ "persist-videos",