commit 5dd228561903b51bf4d51b0e1b3caf29f29ed90f
parent 35fd9573fe941b3e3ece72895009bc27a8f235e1
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 03:11:17 -0400
editor: the reports-prepare job kind (needsMedia, its own queue, replayable), POST /api/ops/reports-prepare and pnpm ops reports-prepare
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
8 files changed, 169 insertions(+), 0 deletions(-)
diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts
@@ -84,6 +84,8 @@ const ADDED_KINDS: Record<string, { label: string; drainable: boolean }> = {
},
// A list of videos persisted one at a time: a drain stops between them.
"persist-videos": { label: "Persist videos", drainable: true },
+ // A report site's evidence media: one pass, cancelled rather than drained.
+ "reports-prepare": { label: "Prepare report media", drainable: false },
};
test("added kinds carry their pinned label and drainability", () => {
@@ -176,6 +178,7 @@ const STILL_MEDIA = [
"clean-audio-transcribed",
"clean-extra-audio-formats",
"remove-wrong-format-audio",
+ "reports-prepare",
];
test("release 17: the text kinds flipped from needsMedia to needsText", () => {
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -630,6 +630,23 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
replayable: false,
queueKeyStrategy: "custom",
},
+ // A REPORT SITE'S EVIDENCE MEDIA (publish/reportMedia.ts): every clip its
+ // published reports cite, cut from the media on disk, and every cited post
+ // capture copied, into the site's report-media cache before its build.
+ // `needsMedia`: it opens saved containers and audio, which may be on another
+ // drive. It spans channels and runs with no `channelSlug`, so the guard in
+ // runManagedFunction does not ask; the step itself reports a span on an
+ // unreachable channel as unreachable rather than missing. Its own queue
+ // (`reports-prepare`): one at a time, behind nothing — not the download
+ // queues, not the build queue. Replayable: the spec is the site.
+ "reports-prepare": {
+ kind: "reports-prepare",
+ label: "Prepare report media",
+ drainable: false,
+ replayable: true,
+ queueKeyStrategy: "custom",
+ needsMedia: true,
+ },
// THE PER-VIDEO WRITERS THAT WERE NOT IN THIS TABLE (release 16 slice RM).
// Each runs with a channelSlug and writes under `data/<id>/` — a single
// video's transcription (the video page's two Transcribe buttons, and its
diff --git a/editor/app/api/ops/reports-prepare/route.test.ts b/editor/app/api/ops/reports-prepare/route.test.ts
@@ -0,0 +1,49 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, rm } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+
+// Run with:
+// pnpm -C editor exec tsx --test "app/api/ops/reports-prepare/route.test.ts"
+//
+// The body's shape and the refusals that come before any job, answered from an
+// empty temp corpus: no job is queued and no media is touched.
+
+const ROOT = await mkdtemp(path.join(os.tmpdir(), "reports-prepare-route-"));
+// Set before the route (and getPaths, which caches) is first imported.
+process.env.WORKER_TOKEN = "test-token";
+process.env.TRANSCRIPTS_DIR = ROOT;
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+const { POST } = await import("./route");
+test.after(() => rm(ROOT, { recursive: true, force: true }));
+
+async function post(body: Record<string, unknown>): Promise<{ status: number; error: string }> {
+ const res = await POST(
+ new Request("http://localhost/api/ops/reports-prepare", {
+ method: "POST",
+ headers: {
+ authorization: "Bearer test-token",
+ "content-type": "application/json",
+ },
+ body: JSON.stringify(body),
+ }),
+ );
+ return { status: res.status, error: ((await res.json()) as { error?: string }).error ?? "" };
+}
+
+test("siteId is required, a site id, and the only key", async () => {
+ assert.match((await post({})).error, /"siteId" is required/);
+ const bad = await post({ siteId: "../x" });
+ assert.equal(bad.status, 400);
+ assert.match(bad.error, /not a valid site id/);
+ const unknown = await post({ siteId: "demo-site", reportId: "r1" });
+ assert.equal(unknown.status, 400);
+ assert.match(unknown.error, /unknown key\(s\): reportId — this route accepts siteId/);
+});
+
+test("a site that does not exist is refused before any job", async () => {
+ const r = await post({ siteId: "demo-site" });
+ assert.equal(r.status, 400);
+ assert.match(r.error, /No site "demo-site"/);
+});
diff --git a/editor/app/api/ops/reports-prepare/route.ts b/editor/app/api/ops/reports-prepare/route.ts
@@ -0,0 +1,23 @@
+import { reportsPrepareAction } from "../../../sites/lib/reportsPrepareAction";
+import { jobResponse, OpsInputError, ops, reqString } from "../_lib";
+import { isValidSiteId } from "yt-dlp-transcript-common/lib/site";
+
+export const dynamic = "force-dynamic";
+
+// POST { siteId: string } -> { ok: true, jobId }
+//
+// An adapter: one call to the action the site's Reports tab posts. Cuts every
+// clip and copies every post capture the site's published reports cite into
+// its report-media cache, as one `reports-prepare` job on its own queue. The
+// job fails, naming each citation, when any lacks its media.
+export async function POST(request: Request) {
+ return ops(request, ["siteId"], async (body) => {
+ const siteId = reqString(body, "siteId");
+ if (!isValidSiteId(siteId)) {
+ throw new OpsInputError(
+ `"${siteId}" is not a valid site id (lowercase letters, digits and "-"; must start with a letter or digit)`,
+ );
+ }
+ return jobResponse(await reportsPrepareAction(siteId));
+ });
+}
diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts
@@ -54,6 +54,7 @@ import {
redownloadShortAudioBucketAction,
} from "../channels/[slug]/incompleteTranscriptActions";
import { replayFetchWindowAction } from "../channels/[slug]/videos/[id]/videoActions";
+import { reportsPrepareAction } from "../sites/lib/reportsPrepareAction";
export type ReplayHandler = (spec: JobSpec) => Promise<StreamActionResult>;
@@ -89,6 +90,13 @@ function params(spec: JobSpec): {
}
export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = {
+ // A report site's evidence media. The spec's slug is the SITE (the job has
+ // no channel); a re-run re-reads the site's reports and cuts only what its
+ // cache does not already hold.
+ "reports-prepare": (spec) => {
+ const { p } = params(spec);
+ return reportsPrepareAction(str(p.siteId) ?? spec.slug);
+ },
// A clip window sourced for another tool. Replay RE-DERIVES from disk like
// every bucket job does: if the window (or a wider one covering it) has
// arrived since, the retry says so neutrally rather than paying twice.
diff --git a/editor/app/sites/lib/reportsPrepareAction.ts b/editor/app/sites/lib/reportsPrepareAction.ts
@@ -0,0 +1,51 @@
+"use server";
+
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { isValidSiteId, listSiteIds } from "yt-dlp-transcript-common/lib/site";
+import {
+ runManagedFunction,
+ type StreamActionResult,
+} from "yt-dlp-transcript-common/jobs/streamCommand";
+import {
+ formatReportMediaProblems,
+ prepareReportMedia,
+} from "yt-dlp-transcript-common/publish/reportMedia";
+
+// Its own queue: one prepare at a time, waiting on no download and no build.
+const REPORTS_PREPARE_QUEUE = "reports-prepare";
+
+// PREPARE A REPORT SITE'S EVIDENCE MEDIA as a job: every clip its published
+// reports cite, cut from the media on disk, and every cited post capture,
+// copied into `.export-index/sites/<siteId>/report-media/` before its build.
+// The work is publish/reportMedia.ts's, the same `archilyzer reports prepare`
+// runs. The job FAILS when any citation lacks its media — the log names each
+// one — and the manifest it writes carries the same list.
+//
+// The spec's `slug` is the SITE id (a spec needs one, and this job belongs to
+// no channel); the replay handler reads `params.siteId`.
+export async function reportsPrepareAction(
+ siteId: string,
+): Promise<StreamActionResult> {
+ const paths = getPaths();
+ const id = siteId.trim();
+ if (!isValidSiteId(id)) {
+ return { ok: false, error: `"${id}" is not a valid site id` };
+ }
+ if (!listSiteIds(paths).includes(id)) {
+ return { ok: false, error: `No site "${id}"` };
+ }
+ return runManagedFunction({
+ kind: "reports-prepare",
+ queueKey: REPORTS_PREPARE_QUEUE,
+ paths,
+ spec: { kind: "reports-prepare", slug: id, params: { siteId: id } },
+ fn: async (onLog, signal) => {
+ const index = await prepareReportMedia({ siteId: id, paths, onLog, signal });
+ if (index.problems.length === 0) return;
+ for (const line of formatReportMediaProblems(index.problems)) onLog(line);
+ throw new Error(
+ `${index.problems.length} problem(s): the site's evidence media is not complete`,
+ );
+ },
+ });
+}
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -39,6 +39,7 @@
// pnpm ops deploy-hub --wait
// pnpm ops build-homepage --json '{"deploy":true}' --wait
// pnpm ops deploy-homepage --json '{"preview":"refresh"}' --wait
+// pnpm ops reports-prepare --json '{"siteId":"demo-site"}' --wait
// pnpm ops get channel the-quartering
// pnpm ops tags --json '{"op":"define","tag":{"id":"eva-collab","label":"Collab"}}'
// pnpm ops tag-videos --file ids.json
@@ -124,6 +125,9 @@ const ACTIONS = [
"relocate",
"relocate-back",
"evict-clips",
+ // A report site's evidence media: cut every cited clip and copy every cited
+ // post capture into the site's report-media cache ({siteId}).
+ "reports-prepare",
"lane",
// The curated-tag writers. `tags` edits the vocabulary (define/remove);
// `tag-videos` pins, unpins, suppresses or unsuppresses one tag over a batch
@@ -317,6 +321,10 @@ export function usage() {
'build-site, build-deploy and deploy-site all take "siteId" (one) or',
' "siteIds" (a list).',
"",
+ 'reports-prepare cuts every clip and copies every post capture a site\'s',
+ ' published reports cite into its report-media cache, before its build:',
+ ' {"siteId"}. The job fails, naming each one, when a citation lacks media.',
+ "",
'retry-bucket runs one bucket of a channel\'s report as one job, past any',
' lane hold: {"slug", "bucket"}. "ids": [...] runs only those videos, and',
" every one must be in the bucket (a stray id is refused, named); a job run",
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -408,6 +408,16 @@ test("capture-posts is a POST to its route, named in the usage", () => {
assert.match(usage(), /unless\s+"articles": false/);
});
+// A report site's evidence media: a POST to its route, the body passed through
+// untouched — the route judges the site id.
+test("reports-prepare is a POST to its route, named in the usage", () => {
+ const p = parseArgs(["reports-prepare", "--json", '{"siteId":"demo-site"}']);
+ assert.equal(p.method, "POST");
+ assert.equal(p.path, "/api/ops/reports-prepare");
+ assert.deepEqual(p.body, { siteId: "demo-site" });
+ assert.match(usage(), /reports-prepare cuts every clip/);
+});
+
test("persist-videos is a POST to its route, named in the usage", () => {
const p = parseArgs([
"persist-videos",