commit 35fd9573fe941b3e3ece72895009bc27a8f235e1
parent f0e07f6de96956648310e0e569cdb61583636fd0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 03:11:17 -0400
common: archilyzer reports prepare <site> — every cited moment's clip or capture into .export-index/sites/<site>/report-media/, an index.json manifest and every problem listed
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
4 files changed, 640 insertions(+), 0 deletions(-)
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -151,6 +151,17 @@ export const COMMANDS: Command[] = [
},
},
{
+ path: ["reports", "prepare"],
+ usage:
+ "<id> cut every clip and copy every post capture the site's published reports cite into its report-media cache, before its build (exit 1 when a citation lacks media; default id: SITE_ID)",
+ maxPositionals: 1,
+ run: async ({ positionals, env }) => {
+ const siteId = siteIdFrom(positionals, env, "reports prepare");
+ if (!siteId) return 2;
+ return (await import("./reports-prepare")).main({ siteId, signal: interrupted() });
+ },
+ },
+ {
path: ["source", "publish"],
usage:
"[--force] [--check] [--keep-scratch] the scrubbed git mirror, raw tree, history pages (stagit, when installed) and tarball into homepage/public, behind the denied-literal gate (--check: audit and count, write nothing)",
diff --git a/common/bin/reports-prepare.ts b/common/bin/reports-prepare.ts
@@ -0,0 +1,30 @@
+// `archilyzer reports prepare <siteId>` — cut and copy the evidence media of a
+// site's published reports into `.export-index/sites/<siteId>/report-media/`,
+// before the site's build. The work is publish/reportMedia.ts's, the same the
+// editor's `reports-prepare` job runs; this file prints its log and its
+// problems.
+//
+// Exit 0 when every cited moment has its media; 1 when any problem is listed
+// (the manifest is written either way, problems included); 2 for a site that
+// does not exist.
+
+import { formatReportMediaProblems, prepareReportMedia } from "../publish/reportMedia";
+
+type Out = { log: (s: string) => void; error: (s: string) => void };
+
+export async function main(
+ opts: { siteId: string; signal?: AbortSignal },
+ out: Out = console,
+): Promise<number> {
+ let index;
+ try {
+ index = await prepareReportMedia({ siteId: opts.siteId, signal: opts.signal, onLog: out.log });
+ } catch (err) {
+ out.error(`reports prepare: ${(err as Error).message}`);
+ return opts.signal?.aborted ? 1 : 2;
+ }
+ if (index.problems.length === 0) return 0;
+ out.error(`reports prepare ${opts.siteId}: ${index.problems.length} problem(s):`);
+ for (const line of formatReportMediaProblems(index.problems)) out.error(` ${line}`);
+ return 1;
+}
diff --git a/common/publish/reportMedia.test.ts b/common/publish/reportMedia.test.ts
@@ -0,0 +1,229 @@
+// `reports prepare` over a temp site: two reports' citations become clips and
+// post captures in the site's report-media cache, with a manifest and every
+// problem listed — through the real report checker, the real tier lookup and
+// real ffmpeg over a few frames of lavfi.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test publish/reportMedia.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { execFileSync } from "node:child_process";
+import { existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+// Every path getPaths() can resolve to a place this file's code may write is
+// pinned under ROOT before anything calls it.
+const ROOT = mkdtempSync(path.join(tmpdir(), "reports-prepare-"));
+Object.assign(process.env, {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+});
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { prepareReportMedia, readReportMediaIndex, reportMediaDir, citedMoments } = await import("./reportMedia");
+const { citedCaptureSourceDir } = await import("./citedPostCaptures");
+const { main: prepareMain } = await import("../bin/reports-prepare");
+
+const paths = getPaths();
+const SITE = "demo-site";
+const CH = "demo-channel";
+const X = "demo-x";
+
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const ff = (args: string[]) => execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", ...args]);
+const videoDir = (slug: string, id: string) => path.join(paths.channelsDir, slug, "data", id);
+
+// A PNG header is all the copier reads of a screenshot.
+function png(w: number, h: number): Buffer {
+ const b = Buffer.alloc(33);
+ b.writeUInt32BE(0x89504e47, 0);
+ b.writeUInt32BE(0x0d0a1a0a, 4);
+ b.writeUInt32BE(13, 8);
+ b.write("IHDR", 12, "latin1");
+ b.writeUInt32BE(w, 16);
+ b.writeUInt32BE(h, 20);
+ return b;
+}
+
+function report(id: string, citations: Record<string, unknown>) {
+ return {
+ format: "archilyzer-report",
+ version: 1,
+ id,
+ kind: "sweep",
+ title: `Report ${id}`,
+ citations,
+ sections: [
+ {
+ id: "s1",
+ title: "One",
+ body: Object.keys(citations).map((c) => `[${c}](cite:${c})`).join(" "),
+ },
+ ],
+ };
+}
+
+// The corpus: a video channel with a fetched window and a recording's sound,
+// an X channel with two captured posts, and a channel no site has.
+writeJson(path.join(paths.channelsDir, CH, "config.json"), {
+ handling: "transcribe",
+ name: "Demo",
+ url: "https://example.test/demo",
+});
+writeJson(path.join(paths.channelsDir, X, "config.json"), {
+ handling: "transcribe",
+ name: "Demo (X)",
+ url: "https://x.com/demo",
+ sourceKind: "social",
+ platform: "twitter",
+});
+mkdirSync(path.join(videoDir(CH, "abc123"), "clips"), { recursive: true });
+ff([
+ "-f", "lavfi", "-i", "testsrc=size=160x90:rate=10:duration=6",
+ "-f", "lavfi", "-i", "sine=frequency=440:duration=6",
+ "-c:v", "libx264", "-preset", "ultrafast", "-pix_fmt", "yuv420p", "-c:a", "aac", "-shortest",
+ path.join(videoDir(CH, "abc123"), "clips", "0.00-6.00.mp4"),
+]);
+mkdirSync(videoDir(CH, "pod1"), { recursive: true });
+ff(["-f", "lavfi", "-i", "sine=frequency=220:duration=6", "-c:a", "aac", path.join(videoDir(CH, "pod1"), "audio.m4a")]);
+for (const id of ["111", "222"]) {
+ const dir = citedCaptureSourceDir(paths.channelsDir, X, id);
+ mkdirSync(dir, { recursive: true });
+ writeFileSync(path.join(dir, "shot.png"), png(600, 400));
+ writeFileSync(path.join(dir, `${id}_1.jpg`), `media of ${id}`);
+ writeFileSync(path.join(dir, "capture.json"), "{}");
+}
+
+writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ title: "Demo",
+ channels: [{ slug: CH }, { slug: X }],
+ publish: "cited",
+ reports: ["r1", "r2", "r-gone"],
+});
+writeJson(
+ path.join(paths.sitesDir, SITE, "reports", "r1", "report.json"),
+ report("r1", {
+ c01: { kind: "video", channel: CH, id: "abc123", start: 1, end: 2, pad: { before: 0.5 }, quote: "one" },
+ c02: { kind: "audio", channel: CH, id: "pod1", start: 1, end: 3, quote: "two" },
+ p01: { kind: "post", channel: X, id: "111", quote: "three" },
+ c03: { kind: "video", channel: CH, id: "missing1", start: 10, end: 12, quote: "four" },
+ c04: { kind: "video", channel: "elsewhere", id: "xyz", start: 1, end: 2, quote: "five" },
+ }),
+);
+// The same moment as r1's c01, with more context after it.
+writeJson(
+ path.join(paths.sitesDir, SITE, "reports", "r2", "report.json"),
+ report("r2", { k1: { kind: "video", channel: CH, id: "abc123", start: 1, end: 2, pad: { after: 1 }, quote: "one" } }),
+);
+// A draft: in reports/, not in site.json. Never read.
+writeJson(
+ path.join(paths.sitesDir, SITE, "reports", "draft", "report.json"),
+ report("draft", { d1: { kind: "post", channel: X, id: "222", quote: "uncited" } }),
+);
+
+const PUBLIC = { social: { x: { visibility: "public" } } };
+const VIDEO_KEY = `${CH}/abc123/1.00-2.00`;
+const AUDIO_KEY = `${CH}/pod1/1.00-3.00`;
+const POST_KEY = `${X}/111`;
+
+test("citedMoments: one moment per span, cited by both reports, with the wider pad on each side", () => {
+ const moments = citedMoments([
+ report("r1", { a: { kind: "video", channel: CH, id: "abc123", start: 1, end: 2, pad: { before: 0.5 }, quote: "q" } }),
+ report("r2", { b: { kind: "audio", channel: CH, id: "abc123", start: 1.001, end: 2, pad: { after: 1 }, quote: "q" } }),
+ ] as never);
+ assert.equal(moments.length, 1);
+ assert.equal(moments[0].key, VIDEO_KEY);
+ assert.equal(moments[0].kind, "video");
+ assert.deepEqual(moments[0].pad, { before: 0.5, after: 1 });
+ assert.deepEqual(moments[0].citedBy, ["r1#a", "r2#b"]);
+});
+
+test("prepare: clips, the cited capture, a manifest, and every problem", async () => {
+ const lines: string[] = [];
+ const index = await prepareReportMedia({ siteId: SITE, settings: PUBLIC, onLog: (l) => lines.push(l) });
+ const dir = reportMediaDir(paths, SITE);
+
+ assert.deepEqual(Object.keys(index.moments), [VIDEO_KEY, POST_KEY, AUDIO_KEY].sort());
+
+ const video = index.moments[VIDEO_KEY];
+ assert.equal(video.kind, "video");
+ assert.match(video.file, /^[0-9a-f]{32}\.mp4$/);
+ assert.equal(video.width, 160);
+ assert.equal(video.height, 90);
+ // 1–2 widened by r1's 0.5 before and r2's 1 after: 0.5–3.
+ assert.ok(Math.abs((video.durationSec ?? 0) - 2.5) < 0.15, `duration ${video.durationSec}`);
+ assert.equal(readFileSync(path.join(dir, video.file)).length, video.bytes);
+
+ const audio = index.moments[AUDIO_KEY];
+ assert.equal(audio.kind, "audio");
+ assert.match(audio.file, /\.m4a$/);
+ assert.equal(audio.width, null);
+
+ const post = index.moments[POST_KEY];
+ assert.equal(post.kind, "post");
+ assert.equal(post.file, `posts/${X}/111/shot.png`);
+ assert.equal(post.width, 600);
+ assert.ok(post.kind === "post" && post.media.map((m) => m.file).join() === `posts/${X}/111/111_1.jpg`);
+ // Only the cited post: not the draft's, not the record.
+ assert.deepEqual(readdirSync(path.join(dir, "posts", X)), ["111"]);
+ assert.deepEqual(readdirSync(path.join(dir, "posts", X, "111")).sort(), ["111_1.jpg", "shot.png"]);
+
+ const byKind = (k: string) => index.problems.filter((p) => p.kind === k);
+ assert.equal(byKind("missing-report").length, 1);
+ assert.equal(byKind("missing-report")[0].report, "r-gone");
+ assert.deepEqual(byKind("missing-media").map((p) => p.moment), [`${CH}/missing1/10.00-12.00`]);
+ assert.deepEqual(byKind("missing-media")[0].citations, ["r1#c03"]);
+ assert.deepEqual(byKind("not-in-site").map((p) => p.moment), ["elsewhere/xyz/1.00-2.00"]);
+ assert.equal(index.problems.length, 3);
+
+ // The manifest on disk is what was returned.
+ assert.deepEqual(await readReportMediaIndex(paths, SITE), index);
+ assert.ok(lines.some((l) => l.includes(VIDEO_KEY) && l.includes("corpus-window")));
+});
+
+test("a second run cuts nothing it already has; the CLI exits 1 and names each problem", async () => {
+ const first = await readReportMediaIndex(paths, SITE);
+ const out: string[] = [];
+ const err: string[] = [];
+ const code = await prepareMain({ siteId: SITE }, { log: (s) => out.push(s), error: (s) => err.push(s) });
+ assert.equal(code, 1);
+ const second = await readReportMediaIndex(paths, SITE);
+ assert.equal(second?.moments[VIDEO_KEY].file, first?.moments[VIDEO_KEY].file);
+ assert.ok(out.some((l) => l.includes(VIDEO_KEY) && l.includes("cached")));
+ assert.ok(err.some((l) => l.startsWith(" missing-media:") && l.includes("missing1") && l.includes("r1#c03")));
+ assert.ok(err.some((l) => l.includes("missing-report")));
+ assert.equal(await prepareMain({ siteId: "no-such-site" }, { log: () => {}, error: () => {} }), 2);
+});
+
+test("a post the site may not carry is a problem, and its capture leaves the cache", async () => {
+ const index = await prepareReportMedia({ siteId: SITE, settings: { social: { x: { visibility: "private" } } } });
+ assert.equal(index.moments[POST_KEY], undefined);
+ assert.deepEqual(
+ index.problems.filter((p) => p.kind === "not-visible").map((p) => p.moment),
+ [POST_KEY],
+ );
+ assert.equal(existsSync(path.join(reportMediaDir(paths, SITE), "posts", X)), false);
+});
+
+test("a clean site: no problems, and a clip no longer cited leaves the cache", async () => {
+ const site = path.join(paths.sitesDir, SITE, "site.json");
+ writeJson(site, { title: "Demo", channels: [{ slug: CH }, { slug: X }], reports: ["r2"] });
+ const index = await prepareReportMedia({ siteId: SITE, settings: PUBLIC });
+ assert.deepEqual(index.problems, []);
+ assert.deepEqual(Object.keys(index.moments), [VIDEO_KEY]);
+ const files = readdirSync(reportMediaDir(paths, SITE)).sort();
+ const clip = index.moments[VIDEO_KEY].file;
+ assert.deepEqual(files, [clip.replace(/\.mp4$/, ".json"), clip, "index.json"].sort());
+ assert.equal(await prepareMain({ siteId: SITE }, { log: () => {}, error: () => {} }), 0);
+});
diff --git a/common/publish/reportMedia.ts b/common/publish/reportMedia.ts
@@ -0,0 +1,370 @@
+// PREPARE A SITE'S EVIDENCE MEDIA — every clip and post capture its published
+// reports cite, cut and copied on the host, before the site's build
+// (plans/report-sites.md, "Evidence media"). `archilyzer reports prepare
+// <siteId>` and the editor's `reports-prepare` job both run `prepareReportMedia`.
+//
+// What it reads: the site's `reports` (site.json, in order), each
+// `sites/<siteId>/reports/<reportId>/report.json`, parsed and validated by the
+// report document's own checker (lib/report/validate.ts). What it does, per
+// cited MOMENT (lib/citations/moments.ts — two citations of one span share one
+// page and so one clip, cut with the wider of their pads):
+//
+// video / audio span lib/evidenceClip-server.ts finds the span's media on
+// disk (clip window, saved container, audio) and cuts it
+// into the cache, or answers why it cannot;
+// post ./citedPostCaptures.ts copies the post's screenshot
+// and media — only cited posts — into the cache.
+//
+// A citation must resolve against the site's own `channels` (the pool a cited
+// site's citations may draw on), and a post must be one the site may carry
+// (lib/postsVisibility.ts — a public site never shows a private platform's
+// posts, cited or not).
+//
+// What it writes: `.export-index/sites/<siteId>/report-media/` —
+// `<hash>.mp4` / `<hash>.m4a` (+ `<hash>.json`) the clips
+// `posts/<channel>/<id>/…` the cited captures
+// `index.json` the manifest:
+// { format, version, siteId, preparedAt,
+// moments: { <momentKey>: { kind, file, bytes, sha256, width, height,
+// durationSec, media? } },
+// problems: [ { kind, message, report?, citations?, moment?, path? } ] }
+// — and nothing else: a clip or capture no moment names is removed, so the
+// cache holds exactly what the site cites. The build copies from here and
+// never runs ffmpeg (it has none).
+//
+// A RUN WITH PROBLEMS STILL WRITES THE MANIFEST (the editor lists the problems
+// from it) and FAILS: a citation without media, a clip over the size limit, an
+// invalid or missing report. Nothing here fetches — the editor's fetch-window
+// and persist (for spans) and Capture posts (for posts) fill what is missing.
+//
+// AN UNMOUNTED DRIVE IS NOT A MISSING CLIP. A span whose media is not found on
+// a channel whose media tier is not reachable (lib/channelMedia.ts) is
+// reported as unreachable, with the reason, rather than as missing.
+
+import { readdir, rm } from "node:fs/promises";
+import path from "node:path";
+import { readJsonFile, writeJsonAtomic } from "../lib/jsonFile-server";
+import { getPaths, type Paths } from "../lib/paths";
+import { getSettings } from "../lib/settings";
+import { getSite, listSiteIds, siteChannelSlugs, siteDir, siteIndexDir, type Site } from "../lib/site";
+import { postsVisibleTo } from "../lib/postsVisibility";
+import { inspectChannelMedia } from "../lib/channelMedia";
+import type { ChannelConfig } from "../lib/channelConfig";
+import { readChannelConfig } from "../controller/channels";
+import { momentKey, momentOf, momentProblem, type Moment } from "../lib/citations/moments";
+import type { CitationPad } from "../lib/citations/schema";
+import { parseReport } from "../lib/report/validate";
+import type { Report } from "../lib/report/schema";
+import {
+ evidenceSpan,
+ isAudioOnlyPlatform,
+ prepareEvidenceClip,
+ widerPad,
+ type EvidenceKind,
+ type EvidenceMedia,
+} from "../lib/evidenceClip-server";
+import { copyCitedPostCaptures, REPORT_POSTS_DIRNAME, type CopiedFile } from "./citedPostCaptures";
+
+export const REPORT_MEDIA_FORMAT = "archilyzer-report-media";
+export const REPORT_MEDIA_VERSION = 1;
+export const REPORT_MEDIA_DIRNAME = "report-media";
+export const REPORT_MEDIA_INDEX_FILENAME = "index.json";
+
+// `.export-index/sites/<siteId>/report-media/`: the prepared media, not served.
+export function reportMediaDir(paths: Paths, siteId: string): string {
+ return path.join(siteIndexDir(paths, siteId), REPORT_MEDIA_DIRNAME);
+}
+
+export function reportMediaIndexFile(paths: Paths, siteId: string): string {
+ return path.join(reportMediaDir(paths, siteId), REPORT_MEDIA_INDEX_FILENAME);
+}
+
+// `sites/<siteId>/reports/<reportId>/`: a report's directory.
+export function siteReportDir(paths: Paths, siteId: string, reportId: string): string {
+ return path.join(siteDir(paths, siteId), "reports", reportId);
+}
+
+export function siteReportFile(paths: Paths, siteId: string, reportId: string): string {
+ return path.join(siteReportDir(paths, siteId, reportId), "report.json");
+}
+
+export type ReportMediaEntry =
+ | EvidenceMedia
+ | {
+ kind: "post";
+ // The screenshot.
+ file: string;
+ bytes: number;
+ sha256: string;
+ width: number | null;
+ height: number | null;
+ durationSec: null;
+ // The post's attached media, copied beside it.
+ media: CopiedFile[];
+ };
+
+export type ReportMediaProblemKind =
+ | "missing-report"
+ | "invalid-report"
+ | "not-in-site"
+ | "not-visible"
+ | "missing-media"
+ | "unreachable"
+ | "too-big"
+ | "cut-failed";
+
+export type ReportMediaProblem = {
+ kind: ReportMediaProblemKind;
+ message: string;
+ report?: string;
+ // `<reportId>#<citationId>` for every citation of the moment.
+ citations?: string[];
+ moment?: string;
+ // A JSON path in the report (an invalid report's problems).
+ path?: string;
+};
+
+export type ReportMediaIndex = {
+ format: typeof REPORT_MEDIA_FORMAT;
+ version: typeof REPORT_MEDIA_VERSION;
+ siteId: string;
+ preparedAt: string;
+ moments: Record<string, ReportMediaEntry>;
+ problems: ReportMediaProblem[];
+};
+
+// One cited moment and everything that cites it.
+type CitedMoment = {
+ key: string;
+ moment: Moment;
+ // A span cited as video anywhere is a video clip; else audio.
+ kind: EvidenceKind | "post";
+ pad?: CitationPad;
+ citedBy: string[];
+};
+
+// The published reports of a site, read and validated. A report that does not
+// parse contributes its problems and no moments; one that parses with value
+// problems contributes both — its media is still prepared, so fixing the
+// document does not wait on a re-cut.
+export async function loadSiteReports(
+ paths: Paths,
+ site: Site,
+): Promise<{ reports: Report[]; problems: ReportMediaProblem[] }> {
+ const reports: Report[] = [];
+ const problems: ReportMediaProblem[] = [];
+ for (const id of site.reports ?? []) {
+ const file = siteReportFile(paths, site.siteId, id);
+ const read = await readJsonFile(file);
+ if (!read.ok) {
+ problems.push({
+ kind: read.reason === "absent" ? "missing-report" : "invalid-report",
+ report: id,
+ message:
+ read.reason === "absent"
+ ? `the site lists report "${id}", but sites/${site.siteId}/reports/${id}/report.json does not exist`
+ : `sites/${site.siteId}/reports/${id}/report.json is not readable JSON`,
+ });
+ continue;
+ }
+ const parsed = parseReport(read.value, { id });
+ for (const p of parsed.problems) {
+ problems.push({ kind: "invalid-report", report: id, path: p.path, message: p.message });
+ }
+ if (parsed.ok) reports.push(parsed.value);
+ }
+ return { reports, problems };
+}
+
+// Every moment the reports cite, keyed, in first-cited order.
+export function citedMoments(reports: readonly Report[]): CitedMoment[] {
+ const byKey = new Map<string, CitedMoment>();
+ for (const report of reports) {
+ for (const [cid, c] of Object.entries(report.citations ?? {})) {
+ const moment = momentOf(c);
+ // A kind without a page, or one validation already names as unsafe.
+ if (!moment || momentProblem(moment)) continue;
+ const key = momentKey(moment);
+ const kind: CitedMoment["kind"] = c.kind === "post" ? "post" : c.kind === "audio" ? "audio" : "video";
+ const pad = c.kind === "video" || c.kind === "audio" ? c.pad : undefined;
+ const seen = byKey.get(key);
+ if (!seen) {
+ byKey.set(key, { key, moment, kind, pad, citedBy: [`${report.id}#${cid}`] });
+ continue;
+ }
+ seen.citedBy.push(`${report.id}#${cid}`);
+ seen.pad = widerPad(seen.pad, pad);
+ if (kind === "video") seen.kind = "video";
+ }
+ }
+ return [...byKey.values()];
+}
+
+export type PrepareReportMediaOptions = {
+ siteId: string;
+ paths?: Paths;
+ onLog?: (line: string) => void;
+ signal?: AbortSignal;
+ // `social.x.visibility` and friends; default the live settings.
+ settings?: { social?: { x?: { visibility?: unknown } } };
+ now?: () => Date;
+};
+
+const mib = (n: number) => `${(n / 1024 / 1024).toFixed(1)} MiB`;
+
+export async function prepareReportMedia(opts: PrepareReportMediaOptions): Promise<ReportMediaIndex> {
+ const paths = opts.paths ?? getPaths();
+ const log = opts.onLog ?? (() => {});
+ const { siteId } = opts;
+ if (!listSiteIds(paths).includes(siteId)) {
+ throw new Error(`no site "${siteId}" (no sites/${siteId}/site.json)`);
+ }
+ const site = getSite(siteId, paths);
+ const settings = opts.settings ?? getSettings();
+ const cacheDir = reportMediaDir(paths, siteId);
+
+ const { reports, problems } = await loadSiteReports(paths, site);
+ log(`${siteId}: ${site.reports?.length ?? 0} published report(s), ${reports.length} readable.`);
+ const moments = citedMoments(reports);
+ log(`${moments.length} cited moment(s) with media.`);
+
+ const pool = siteChannelSlugs(site);
+ const configs = new Map<string, ChannelConfig | null>();
+ const configOf = async (slug: string) => {
+ if (!configs.has(slug)) configs.set(slug, await readChannelConfig(paths, slug));
+ return configs.get(slug) ?? null;
+ };
+ const entries: Record<string, ReportMediaEntry> = {};
+ const fail = (m: CitedMoment, kind: ReportMediaProblemKind, message: string) => {
+ problems.push({ kind, moment: m.key, citations: m.citedBy, message });
+ log(` ✗ ${m.key}: ${message}`);
+ };
+
+ const posts: CitedMoment[] = [];
+ for (const m of moments) {
+ if (opts.signal?.aborted) throw new Error("prepare cancelled");
+ const slug = m.moment.channel;
+ if (!pool.has(slug)) {
+ fail(m, "not-in-site", `channel "${slug}" is not one of this site's channels`);
+ continue;
+ }
+ const config = await configOf(slug);
+ if (m.kind === "post") {
+ if (!postsVisibleTo(site, config, settings)) {
+ fail(m, "not-visible", `this site may not carry posts of "${slug}" (the post visibility rule)`);
+ continue;
+ }
+ posts.push(m);
+ continue;
+ }
+ if (m.moment.kind !== "span") continue;
+ const span = evidenceSpan({ start: m.moment.start, end: m.moment.end, pad: m.pad });
+ const r = await prepareEvidenceClip({
+ channelsDir: paths.channelsDir,
+ slug,
+ id: m.moment.id,
+ kind: m.kind,
+ span,
+ cacheDir,
+ audioOnlyRecord: isAudioOnlyPlatform(config?.platform),
+ ffmpegBin: paths.ffmpegBin,
+ ffprobeBin: paths.ffprobeBin,
+ signal: opts.signal,
+ });
+ if (r.ok) {
+ entries[m.key] = r.media;
+ log(
+ ` ${r.cached ? "=" : "+"} ${m.key} ← ${r.source.kind} ${r.source.name}: ` +
+ `${r.media.file} (${mib(r.media.bytes)}${r.cached ? ", cached" : ""})`,
+ );
+ continue;
+ }
+ if (r.reason === "missing") {
+ const where = await inspectChannelMedia(paths, slug, config, { fresh: true }).catch(() => null);
+ if (where && where.status !== "ok" && where.status !== "in-place") {
+ fail(m, "unreachable", `${r.message}; the channel's media is ${where.status}${where.detail ? ` (${where.detail})` : ""}`);
+ continue;
+ }
+ fail(m, "missing-media", `${r.message} — fetch the window or persist the video in the editor`);
+ continue;
+ }
+ fail(m, r.reason === "too-big" ? "too-big" : r.reason === "no-audio" ? "missing-media" : "cut-failed", r.message);
+ }
+
+ if (opts.signal?.aborted) throw new Error("prepare cancelled");
+ const copiedPosts = await copyCitedPostCaptures({
+ channelsDir: paths.channelsDir,
+ destRoot: cacheDir,
+ posts: posts.map((m) => ({ channel: m.moment.channel, id: m.moment.id })),
+ });
+ const postMoment = new Map(posts.map((m) => [m.key, m]));
+ for (const c of copiedPosts.copied) {
+ const key = `${c.channel}/${c.id}`;
+ entries[key] = { kind: "post", ...c.shot, durationSec: null, media: c.media };
+ log(` + ${key}: screenshot${c.media.length ? ` and ${c.media.length} media file(s)` : ""}`);
+ }
+ for (const miss of copiedPosts.missing) {
+ const m = postMoment.get(`${miss.channel}/${miss.id}`);
+ if (m) fail(m, "missing-media", miss.message);
+ }
+
+ // The cache holds what the manifest names: a clip no moment names goes.
+ await pruneClips(cacheDir, entries);
+
+ const index: ReportMediaIndex = {
+ format: REPORT_MEDIA_FORMAT,
+ version: REPORT_MEDIA_VERSION,
+ siteId,
+ preparedAt: (opts.now?.() ?? new Date()).toISOString(),
+ moments: Object.fromEntries(Object.keys(entries).sort().map((k) => [k, entries[k]])),
+ problems,
+ };
+ await writeJsonAtomic(reportMediaIndexFile(paths, siteId), index, { mkdir: true });
+ const bytes = Object.values(entries).reduce(
+ (n, e) => n + e.bytes + (e.kind === "post" ? e.media.reduce((s, f) => s + f.bytes, 0) : 0),
+ 0,
+ );
+ log(
+ `${Object.keys(entries).length} of ${moments.length} moment(s) prepared (${mib(bytes)}); ` +
+ `${problems.length} problem(s).`,
+ );
+ return index;
+}
+
+const CLIP_FILE_RE = /^[0-9a-f]{32}\.(?:mp4|m4a|json)$/;
+
+async function pruneClips(cacheDir: string, entries: Record<string, ReportMediaEntry>): Promise<void> {
+ const keep = new Set<string>([REPORT_MEDIA_INDEX_FILENAME, REPORT_POSTS_DIRNAME]);
+ for (const e of Object.values(entries)) {
+ if (e.kind === "post") continue;
+ keep.add(e.file);
+ keep.add(e.file.replace(/\.[^.]+$/, ".json"));
+ }
+ for (const name of await readdir(cacheDir).catch(() => [] as string[])) {
+ if (keep.has(name)) continue;
+ // Only what this module writes: a clip, its sidecar, a temp file of either.
+ if (CLIP_FILE_RE.test(name) || /^[0-9a-f]{32}\.(?:mp4|m4a|json)\.tmp-/.test(name)) {
+ await rm(path.join(cacheDir, name), { force: true });
+ }
+ }
+}
+
+// The problems as lines, for a log or a terminal.
+export function formatReportMediaProblems(problems: readonly ReportMediaProblem[]): string[] {
+ return problems.map((p) => {
+ const where = p.moment
+ ? `${p.moment}${p.citations?.length ? ` (cited by ${p.citations.join(", ")})` : ""}`
+ : `${p.report ?? "?"}${p.path ? ` at ${p.path}` : ""}`;
+ return `${p.kind}: ${where}: ${p.message}`;
+ });
+}
+
+// Read a prepared manifest, or null when there is none (never prepared, or
+// unreadable).
+export async function readReportMediaIndex(paths: Paths, siteId: string): Promise<ReportMediaIndex | null> {
+ const read = await readJsonFile(reportMediaIndexFile(paths, siteId));
+ if (!read.ok) return null;
+ const v = read.value as Partial<ReportMediaIndex> | null;
+ if (!v || v.format !== REPORT_MEDIA_FORMAT || v.version !== REPORT_MEDIA_VERSION) return null;
+ return v as ReportMediaIndex;
+}