Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 35fd9573fe941b3e3ece72895009bc27a8f235e1
parent f0e07f6de96956648310e0e569cdb61583636fd0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 03:11:17 -0400

common: archilyzer reports prepare <site> — every cited moment's clip or capture into .export-index/sites/<site>/report-media/, an index.json manifest and every problem listed

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/bin/archilyzer.ts | 11+++++++++++
Acommon/bin/reports-prepare.ts | 30++++++++++++++++++++++++++++++
Acommon/publish/reportMedia.test.ts | 229+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/publish/reportMedia.ts | 370+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
4 files changed, 640 insertions(+), 0 deletions(-)

diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts @@ -151,6 +151,17 @@ export const COMMANDS: Command[] = [ }, }, { + path: ["reports", "prepare"], + usage: + "<id> cut every clip and copy every post capture the site's published reports cite into its report-media cache, before its build (exit 1 when a citation lacks media; default id: SITE_ID)", + maxPositionals: 1, + run: async ({ positionals, env }) => { + const siteId = siteIdFrom(positionals, env, "reports prepare"); + if (!siteId) return 2; + return (await import("./reports-prepare")).main({ siteId, signal: interrupted() }); + }, + }, + { path: ["source", "publish"], usage: "[--force] [--check] [--keep-scratch] the scrubbed git mirror, raw tree, history pages (stagit, when installed) and tarball into homepage/public, behind the denied-literal gate (--check: audit and count, write nothing)", diff --git a/common/bin/reports-prepare.ts b/common/bin/reports-prepare.ts @@ -0,0 +1,30 @@ +// `archilyzer reports prepare <siteId>` — cut and copy the evidence media of a +// site's published reports into `.export-index/sites/<siteId>/report-media/`, +// before the site's build. The work is publish/reportMedia.ts's, the same the +// editor's `reports-prepare` job runs; this file prints its log and its +// problems. +// +// Exit 0 when every cited moment has its media; 1 when any problem is listed +// (the manifest is written either way, problems included); 2 for a site that +// does not exist. + +import { formatReportMediaProblems, prepareReportMedia } from "../publish/reportMedia"; + +type Out = { log: (s: string) => void; error: (s: string) => void }; + +export async function main( + opts: { siteId: string; signal?: AbortSignal }, + out: Out = console, +): Promise<number> { + let index; + try { + index = await prepareReportMedia({ siteId: opts.siteId, signal: opts.signal, onLog: out.log }); + } catch (err) { + out.error(`reports prepare: ${(err as Error).message}`); + return opts.signal?.aborted ? 1 : 2; + } + if (index.problems.length === 0) return 0; + out.error(`reports prepare ${opts.siteId}: ${index.problems.length} problem(s):`); + for (const line of formatReportMediaProblems(index.problems)) out.error(` ${line}`); + return 1; +} diff --git a/common/publish/reportMedia.test.ts b/common/publish/reportMedia.test.ts @@ -0,0 +1,229 @@ +// `reports prepare` over a temp site: two reports' citations become clips and +// post captures in the site's report-media cache, with a manifest and every +// problem listed — through the real report checker, the real tier lookup and +// real ffmpeg over a few frames of lavfi. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test publish/reportMedia.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { execFileSync } from "node:child_process"; +import { existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +// Every path getPaths() can resolve to a place this file's code may write is +// pinned under ROOT before anything calls it. +const ROOT = mkdtempSync(path.join(tmpdir(), "reports-prepare-")); +Object.assign(process.env, { + TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), + SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), + SITES_DIR: path.join(ROOT, "transcripts", "sites"), + SETTINGS_FILE: path.join(ROOT, "settings.json"), + EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), + EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), + EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), + ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), +}); +after(() => rmSync(ROOT, { recursive: true, force: true })); + +const { getPaths } = await import("../lib/paths"); +const { prepareReportMedia, readReportMediaIndex, reportMediaDir, citedMoments } = await import("./reportMedia"); +const { citedCaptureSourceDir } = await import("./citedPostCaptures"); +const { main: prepareMain } = await import("../bin/reports-prepare"); + +const paths = getPaths(); +const SITE = "demo-site"; +const CH = "demo-channel"; +const X = "demo-x"; + +const writeJson = (file: string, value: unknown) => { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, JSON.stringify(value, null, 2)); +}; +const ff = (args: string[]) => execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", ...args]); +const videoDir = (slug: string, id: string) => path.join(paths.channelsDir, slug, "data", id); + +// A PNG header is all the copier reads of a screenshot. +function png(w: number, h: number): Buffer { + const b = Buffer.alloc(33); + b.writeUInt32BE(0x89504e47, 0); + b.writeUInt32BE(0x0d0a1a0a, 4); + b.writeUInt32BE(13, 8); + b.write("IHDR", 12, "latin1"); + b.writeUInt32BE(w, 16); + b.writeUInt32BE(h, 20); + return b; +} + +function report(id: string, citations: Record<string, unknown>) { + return { + format: "archilyzer-report", + version: 1, + id, + kind: "sweep", + title: `Report ${id}`, + citations, + sections: [ + { + id: "s1", + title: "One", + body: Object.keys(citations).map((c) => `[${c}](cite:${c})`).join(" "), + }, + ], + }; +} + +// The corpus: a video channel with a fetched window and a recording's sound, +// an X channel with two captured posts, and a channel no site has. +writeJson(path.join(paths.channelsDir, CH, "config.json"), { + handling: "transcribe", + name: "Demo", + url: "https://example.test/demo", +}); +writeJson(path.join(paths.channelsDir, X, "config.json"), { + handling: "transcribe", + name: "Demo (X)", + url: "https://x.com/demo", + sourceKind: "social", + platform: "twitter", +}); +mkdirSync(path.join(videoDir(CH, "abc123"), "clips"), { recursive: true }); +ff([ + "-f", "lavfi", "-i", "testsrc=size=160x90:rate=10:duration=6", + "-f", "lavfi", "-i", "sine=frequency=440:duration=6", + "-c:v", "libx264", "-preset", "ultrafast", "-pix_fmt", "yuv420p", "-c:a", "aac", "-shortest", + path.join(videoDir(CH, "abc123"), "clips", "0.00-6.00.mp4"), +]); +mkdirSync(videoDir(CH, "pod1"), { recursive: true }); +ff(["-f", "lavfi", "-i", "sine=frequency=220:duration=6", "-c:a", "aac", path.join(videoDir(CH, "pod1"), "audio.m4a")]); +for (const id of ["111", "222"]) { + const dir = citedCaptureSourceDir(paths.channelsDir, X, id); + mkdirSync(dir, { recursive: true }); + writeFileSync(path.join(dir, "shot.png"), png(600, 400)); + writeFileSync(path.join(dir, `${id}_1.jpg`), `media of ${id}`); + writeFileSync(path.join(dir, "capture.json"), "{}"); +} + +writeJson(path.join(paths.sitesDir, SITE, "site.json"), { + title: "Demo", + channels: [{ slug: CH }, { slug: X }], + publish: "cited", + reports: ["r1", "r2", "r-gone"], +}); +writeJson( + path.join(paths.sitesDir, SITE, "reports", "r1", "report.json"), + report("r1", { + c01: { kind: "video", channel: CH, id: "abc123", start: 1, end: 2, pad: { before: 0.5 }, quote: "one" }, + c02: { kind: "audio", channel: CH, id: "pod1", start: 1, end: 3, quote: "two" }, + p01: { kind: "post", channel: X, id: "111", quote: "three" }, + c03: { kind: "video", channel: CH, id: "missing1", start: 10, end: 12, quote: "four" }, + c04: { kind: "video", channel: "elsewhere", id: "xyz", start: 1, end: 2, quote: "five" }, + }), +); +// The same moment as r1's c01, with more context after it. +writeJson( + path.join(paths.sitesDir, SITE, "reports", "r2", "report.json"), + report("r2", { k1: { kind: "video", channel: CH, id: "abc123", start: 1, end: 2, pad: { after: 1 }, quote: "one" } }), +); +// A draft: in reports/, not in site.json. Never read. +writeJson( + path.join(paths.sitesDir, SITE, "reports", "draft", "report.json"), + report("draft", { d1: { kind: "post", channel: X, id: "222", quote: "uncited" } }), +); + +const PUBLIC = { social: { x: { visibility: "public" } } }; +const VIDEO_KEY = `${CH}/abc123/1.00-2.00`; +const AUDIO_KEY = `${CH}/pod1/1.00-3.00`; +const POST_KEY = `${X}/111`; + +test("citedMoments: one moment per span, cited by both reports, with the wider pad on each side", () => { + const moments = citedMoments([ + report("r1", { a: { kind: "video", channel: CH, id: "abc123", start: 1, end: 2, pad: { before: 0.5 }, quote: "q" } }), + report("r2", { b: { kind: "audio", channel: CH, id: "abc123", start: 1.001, end: 2, pad: { after: 1 }, quote: "q" } }), + ] as never); + assert.equal(moments.length, 1); + assert.equal(moments[0].key, VIDEO_KEY); + assert.equal(moments[0].kind, "video"); + assert.deepEqual(moments[0].pad, { before: 0.5, after: 1 }); + assert.deepEqual(moments[0].citedBy, ["r1#a", "r2#b"]); +}); + +test("prepare: clips, the cited capture, a manifest, and every problem", async () => { + const lines: string[] = []; + const index = await prepareReportMedia({ siteId: SITE, settings: PUBLIC, onLog: (l) => lines.push(l) }); + const dir = reportMediaDir(paths, SITE); + + assert.deepEqual(Object.keys(index.moments), [VIDEO_KEY, POST_KEY, AUDIO_KEY].sort()); + + const video = index.moments[VIDEO_KEY]; + assert.equal(video.kind, "video"); + assert.match(video.file, /^[0-9a-f]{32}\.mp4$/); + assert.equal(video.width, 160); + assert.equal(video.height, 90); + // 1–2 widened by r1's 0.5 before and r2's 1 after: 0.5–3. + assert.ok(Math.abs((video.durationSec ?? 0) - 2.5) < 0.15, `duration ${video.durationSec}`); + assert.equal(readFileSync(path.join(dir, video.file)).length, video.bytes); + + const audio = index.moments[AUDIO_KEY]; + assert.equal(audio.kind, "audio"); + assert.match(audio.file, /\.m4a$/); + assert.equal(audio.width, null); + + const post = index.moments[POST_KEY]; + assert.equal(post.kind, "post"); + assert.equal(post.file, `posts/${X}/111/shot.png`); + assert.equal(post.width, 600); + assert.ok(post.kind === "post" && post.media.map((m) => m.file).join() === `posts/${X}/111/111_1.jpg`); + // Only the cited post: not the draft's, not the record. + assert.deepEqual(readdirSync(path.join(dir, "posts", X)), ["111"]); + assert.deepEqual(readdirSync(path.join(dir, "posts", X, "111")).sort(), ["111_1.jpg", "shot.png"]); + + const byKind = (k: string) => index.problems.filter((p) => p.kind === k); + assert.equal(byKind("missing-report").length, 1); + assert.equal(byKind("missing-report")[0].report, "r-gone"); + assert.deepEqual(byKind("missing-media").map((p) => p.moment), [`${CH}/missing1/10.00-12.00`]); + assert.deepEqual(byKind("missing-media")[0].citations, ["r1#c03"]); + assert.deepEqual(byKind("not-in-site").map((p) => p.moment), ["elsewhere/xyz/1.00-2.00"]); + assert.equal(index.problems.length, 3); + + // The manifest on disk is what was returned. + assert.deepEqual(await readReportMediaIndex(paths, SITE), index); + assert.ok(lines.some((l) => l.includes(VIDEO_KEY) && l.includes("corpus-window"))); +}); + +test("a second run cuts nothing it already has; the CLI exits 1 and names each problem", async () => { + const first = await readReportMediaIndex(paths, SITE); + const out: string[] = []; + const err: string[] = []; + const code = await prepareMain({ siteId: SITE }, { log: (s) => out.push(s), error: (s) => err.push(s) }); + assert.equal(code, 1); + const second = await readReportMediaIndex(paths, SITE); + assert.equal(second?.moments[VIDEO_KEY].file, first?.moments[VIDEO_KEY].file); + assert.ok(out.some((l) => l.includes(VIDEO_KEY) && l.includes("cached"))); + assert.ok(err.some((l) => l.startsWith(" missing-media:") && l.includes("missing1") && l.includes("r1#c03"))); + assert.ok(err.some((l) => l.includes("missing-report"))); + assert.equal(await prepareMain({ siteId: "no-such-site" }, { log: () => {}, error: () => {} }), 2); +}); + +test("a post the site may not carry is a problem, and its capture leaves the cache", async () => { + const index = await prepareReportMedia({ siteId: SITE, settings: { social: { x: { visibility: "private" } } } }); + assert.equal(index.moments[POST_KEY], undefined); + assert.deepEqual( + index.problems.filter((p) => p.kind === "not-visible").map((p) => p.moment), + [POST_KEY], + ); + assert.equal(existsSync(path.join(reportMediaDir(paths, SITE), "posts", X)), false); +}); + +test("a clean site: no problems, and a clip no longer cited leaves the cache", async () => { + const site = path.join(paths.sitesDir, SITE, "site.json"); + writeJson(site, { title: "Demo", channels: [{ slug: CH }, { slug: X }], reports: ["r2"] }); + const index = await prepareReportMedia({ siteId: SITE, settings: PUBLIC }); + assert.deepEqual(index.problems, []); + assert.deepEqual(Object.keys(index.moments), [VIDEO_KEY]); + const files = readdirSync(reportMediaDir(paths, SITE)).sort(); + const clip = index.moments[VIDEO_KEY].file; + assert.deepEqual(files, [clip.replace(/\.mp4$/, ".json"), clip, "index.json"].sort()); + assert.equal(await prepareMain({ siteId: SITE }, { log: () => {}, error: () => {} }), 0); +}); diff --git a/common/publish/reportMedia.ts b/common/publish/reportMedia.ts @@ -0,0 +1,370 @@ +// PREPARE A SITE'S EVIDENCE MEDIA — every clip and post capture its published +// reports cite, cut and copied on the host, before the site's build +// (plans/report-sites.md, "Evidence media"). `archilyzer reports prepare +// <siteId>` and the editor's `reports-prepare` job both run `prepareReportMedia`. +// +// What it reads: the site's `reports` (site.json, in order), each +// `sites/<siteId>/reports/<reportId>/report.json`, parsed and validated by the +// report document's own checker (lib/report/validate.ts). What it does, per +// cited MOMENT (lib/citations/moments.ts — two citations of one span share one +// page and so one clip, cut with the wider of their pads): +// +// video / audio span lib/evidenceClip-server.ts finds the span's media on +// disk (clip window, saved container, audio) and cuts it +// into the cache, or answers why it cannot; +// post ./citedPostCaptures.ts copies the post's screenshot +// and media — only cited posts — into the cache. +// +// A citation must resolve against the site's own `channels` (the pool a cited +// site's citations may draw on), and a post must be one the site may carry +// (lib/postsVisibility.ts — a public site never shows a private platform's +// posts, cited or not). +// +// What it writes: `.export-index/sites/<siteId>/report-media/` — +// `<hash>.mp4` / `<hash>.m4a` (+ `<hash>.json`) the clips +// `posts/<channel>/<id>/…` the cited captures +// `index.json` the manifest: +// { format, version, siteId, preparedAt, +// moments: { <momentKey>: { kind, file, bytes, sha256, width, height, +// durationSec, media? } }, +// problems: [ { kind, message, report?, citations?, moment?, path? } ] } +// — and nothing else: a clip or capture no moment names is removed, so the +// cache holds exactly what the site cites. The build copies from here and +// never runs ffmpeg (it has none). +// +// A RUN WITH PROBLEMS STILL WRITES THE MANIFEST (the editor lists the problems +// from it) and FAILS: a citation without media, a clip over the size limit, an +// invalid or missing report. Nothing here fetches — the editor's fetch-window +// and persist (for spans) and Capture posts (for posts) fill what is missing. +// +// AN UNMOUNTED DRIVE IS NOT A MISSING CLIP. A span whose media is not found on +// a channel whose media tier is not reachable (lib/channelMedia.ts) is +// reported as unreachable, with the reason, rather than as missing. + +import { readdir, rm } from "node:fs/promises"; +import path from "node:path"; +import { readJsonFile, writeJsonAtomic } from "../lib/jsonFile-server"; +import { getPaths, type Paths } from "../lib/paths"; +import { getSettings } from "../lib/settings"; +import { getSite, listSiteIds, siteChannelSlugs, siteDir, siteIndexDir, type Site } from "../lib/site"; +import { postsVisibleTo } from "../lib/postsVisibility"; +import { inspectChannelMedia } from "../lib/channelMedia"; +import type { ChannelConfig } from "../lib/channelConfig"; +import { readChannelConfig } from "../controller/channels"; +import { momentKey, momentOf, momentProblem, type Moment } from "../lib/citations/moments"; +import type { CitationPad } from "../lib/citations/schema"; +import { parseReport } from "../lib/report/validate"; +import type { Report } from "../lib/report/schema"; +import { + evidenceSpan, + isAudioOnlyPlatform, + prepareEvidenceClip, + widerPad, + type EvidenceKind, + type EvidenceMedia, +} from "../lib/evidenceClip-server"; +import { copyCitedPostCaptures, REPORT_POSTS_DIRNAME, type CopiedFile } from "./citedPostCaptures"; + +export const REPORT_MEDIA_FORMAT = "archilyzer-report-media"; +export const REPORT_MEDIA_VERSION = 1; +export const REPORT_MEDIA_DIRNAME = "report-media"; +export const REPORT_MEDIA_INDEX_FILENAME = "index.json"; + +// `.export-index/sites/<siteId>/report-media/`: the prepared media, not served. +export function reportMediaDir(paths: Paths, siteId: string): string { + return path.join(siteIndexDir(paths, siteId), REPORT_MEDIA_DIRNAME); +} + +export function reportMediaIndexFile(paths: Paths, siteId: string): string { + return path.join(reportMediaDir(paths, siteId), REPORT_MEDIA_INDEX_FILENAME); +} + +// `sites/<siteId>/reports/<reportId>/`: a report's directory. +export function siteReportDir(paths: Paths, siteId: string, reportId: string): string { + return path.join(siteDir(paths, siteId), "reports", reportId); +} + +export function siteReportFile(paths: Paths, siteId: string, reportId: string): string { + return path.join(siteReportDir(paths, siteId, reportId), "report.json"); +} + +export type ReportMediaEntry = + | EvidenceMedia + | { + kind: "post"; + // The screenshot. + file: string; + bytes: number; + sha256: string; + width: number | null; + height: number | null; + durationSec: null; + // The post's attached media, copied beside it. + media: CopiedFile[]; + }; + +export type ReportMediaProblemKind = + | "missing-report" + | "invalid-report" + | "not-in-site" + | "not-visible" + | "missing-media" + | "unreachable" + | "too-big" + | "cut-failed"; + +export type ReportMediaProblem = { + kind: ReportMediaProblemKind; + message: string; + report?: string; + // `<reportId>#<citationId>` for every citation of the moment. + citations?: string[]; + moment?: string; + // A JSON path in the report (an invalid report's problems). + path?: string; +}; + +export type ReportMediaIndex = { + format: typeof REPORT_MEDIA_FORMAT; + version: typeof REPORT_MEDIA_VERSION; + siteId: string; + preparedAt: string; + moments: Record<string, ReportMediaEntry>; + problems: ReportMediaProblem[]; +}; + +// One cited moment and everything that cites it. +type CitedMoment = { + key: string; + moment: Moment; + // A span cited as video anywhere is a video clip; else audio. + kind: EvidenceKind | "post"; + pad?: CitationPad; + citedBy: string[]; +}; + +// The published reports of a site, read and validated. A report that does not +// parse contributes its problems and no moments; one that parses with value +// problems contributes both — its media is still prepared, so fixing the +// document does not wait on a re-cut. +export async function loadSiteReports( + paths: Paths, + site: Site, +): Promise<{ reports: Report[]; problems: ReportMediaProblem[] }> { + const reports: Report[] = []; + const problems: ReportMediaProblem[] = []; + for (const id of site.reports ?? []) { + const file = siteReportFile(paths, site.siteId, id); + const read = await readJsonFile(file); + if (!read.ok) { + problems.push({ + kind: read.reason === "absent" ? "missing-report" : "invalid-report", + report: id, + message: + read.reason === "absent" + ? `the site lists report "${id}", but sites/${site.siteId}/reports/${id}/report.json does not exist` + : `sites/${site.siteId}/reports/${id}/report.json is not readable JSON`, + }); + continue; + } + const parsed = parseReport(read.value, { id }); + for (const p of parsed.problems) { + problems.push({ kind: "invalid-report", report: id, path: p.path, message: p.message }); + } + if (parsed.ok) reports.push(parsed.value); + } + return { reports, problems }; +} + +// Every moment the reports cite, keyed, in first-cited order. +export function citedMoments(reports: readonly Report[]): CitedMoment[] { + const byKey = new Map<string, CitedMoment>(); + for (const report of reports) { + for (const [cid, c] of Object.entries(report.citations ?? {})) { + const moment = momentOf(c); + // A kind without a page, or one validation already names as unsafe. + if (!moment || momentProblem(moment)) continue; + const key = momentKey(moment); + const kind: CitedMoment["kind"] = c.kind === "post" ? "post" : c.kind === "audio" ? "audio" : "video"; + const pad = c.kind === "video" || c.kind === "audio" ? c.pad : undefined; + const seen = byKey.get(key); + if (!seen) { + byKey.set(key, { key, moment, kind, pad, citedBy: [`${report.id}#${cid}`] }); + continue; + } + seen.citedBy.push(`${report.id}#${cid}`); + seen.pad = widerPad(seen.pad, pad); + if (kind === "video") seen.kind = "video"; + } + } + return [...byKey.values()]; +} + +export type PrepareReportMediaOptions = { + siteId: string; + paths?: Paths; + onLog?: (line: string) => void; + signal?: AbortSignal; + // `social.x.visibility` and friends; default the live settings. + settings?: { social?: { x?: { visibility?: unknown } } }; + now?: () => Date; +}; + +const mib = (n: number) => `${(n / 1024 / 1024).toFixed(1)} MiB`; + +export async function prepareReportMedia(opts: PrepareReportMediaOptions): Promise<ReportMediaIndex> { + const paths = opts.paths ?? getPaths(); + const log = opts.onLog ?? (() => {}); + const { siteId } = opts; + if (!listSiteIds(paths).includes(siteId)) { + throw new Error(`no site "${siteId}" (no sites/${siteId}/site.json)`); + } + const site = getSite(siteId, paths); + const settings = opts.settings ?? getSettings(); + const cacheDir = reportMediaDir(paths, siteId); + + const { reports, problems } = await loadSiteReports(paths, site); + log(`${siteId}: ${site.reports?.length ?? 0} published report(s), ${reports.length} readable.`); + const moments = citedMoments(reports); + log(`${moments.length} cited moment(s) with media.`); + + const pool = siteChannelSlugs(site); + const configs = new Map<string, ChannelConfig | null>(); + const configOf = async (slug: string) => { + if (!configs.has(slug)) configs.set(slug, await readChannelConfig(paths, slug)); + return configs.get(slug) ?? null; + }; + const entries: Record<string, ReportMediaEntry> = {}; + const fail = (m: CitedMoment, kind: ReportMediaProblemKind, message: string) => { + problems.push({ kind, moment: m.key, citations: m.citedBy, message }); + log(` ✗ ${m.key}: ${message}`); + }; + + const posts: CitedMoment[] = []; + for (const m of moments) { + if (opts.signal?.aborted) throw new Error("prepare cancelled"); + const slug = m.moment.channel; + if (!pool.has(slug)) { + fail(m, "not-in-site", `channel "${slug}" is not one of this site's channels`); + continue; + } + const config = await configOf(slug); + if (m.kind === "post") { + if (!postsVisibleTo(site, config, settings)) { + fail(m, "not-visible", `this site may not carry posts of "${slug}" (the post visibility rule)`); + continue; + } + posts.push(m); + continue; + } + if (m.moment.kind !== "span") continue; + const span = evidenceSpan({ start: m.moment.start, end: m.moment.end, pad: m.pad }); + const r = await prepareEvidenceClip({ + channelsDir: paths.channelsDir, + slug, + id: m.moment.id, + kind: m.kind, + span, + cacheDir, + audioOnlyRecord: isAudioOnlyPlatform(config?.platform), + ffmpegBin: paths.ffmpegBin, + ffprobeBin: paths.ffprobeBin, + signal: opts.signal, + }); + if (r.ok) { + entries[m.key] = r.media; + log( + ` ${r.cached ? "=" : "+"} ${m.key} ← ${r.source.kind} ${r.source.name}: ` + + `${r.media.file} (${mib(r.media.bytes)}${r.cached ? ", cached" : ""})`, + ); + continue; + } + if (r.reason === "missing") { + const where = await inspectChannelMedia(paths, slug, config, { fresh: true }).catch(() => null); + if (where && where.status !== "ok" && where.status !== "in-place") { + fail(m, "unreachable", `${r.message}; the channel's media is ${where.status}${where.detail ? ` (${where.detail})` : ""}`); + continue; + } + fail(m, "missing-media", `${r.message} — fetch the window or persist the video in the editor`); + continue; + } + fail(m, r.reason === "too-big" ? "too-big" : r.reason === "no-audio" ? "missing-media" : "cut-failed", r.message); + } + + if (opts.signal?.aborted) throw new Error("prepare cancelled"); + const copiedPosts = await copyCitedPostCaptures({ + channelsDir: paths.channelsDir, + destRoot: cacheDir, + posts: posts.map((m) => ({ channel: m.moment.channel, id: m.moment.id })), + }); + const postMoment = new Map(posts.map((m) => [m.key, m])); + for (const c of copiedPosts.copied) { + const key = `${c.channel}/${c.id}`; + entries[key] = { kind: "post", ...c.shot, durationSec: null, media: c.media }; + log(` + ${key}: screenshot${c.media.length ? ` and ${c.media.length} media file(s)` : ""}`); + } + for (const miss of copiedPosts.missing) { + const m = postMoment.get(`${miss.channel}/${miss.id}`); + if (m) fail(m, "missing-media", miss.message); + } + + // The cache holds what the manifest names: a clip no moment names goes. + await pruneClips(cacheDir, entries); + + const index: ReportMediaIndex = { + format: REPORT_MEDIA_FORMAT, + version: REPORT_MEDIA_VERSION, + siteId, + preparedAt: (opts.now?.() ?? new Date()).toISOString(), + moments: Object.fromEntries(Object.keys(entries).sort().map((k) => [k, entries[k]])), + problems, + }; + await writeJsonAtomic(reportMediaIndexFile(paths, siteId), index, { mkdir: true }); + const bytes = Object.values(entries).reduce( + (n, e) => n + e.bytes + (e.kind === "post" ? e.media.reduce((s, f) => s + f.bytes, 0) : 0), + 0, + ); + log( + `${Object.keys(entries).length} of ${moments.length} moment(s) prepared (${mib(bytes)}); ` + + `${problems.length} problem(s).`, + ); + return index; +} + +const CLIP_FILE_RE = /^[0-9a-f]{32}\.(?:mp4|m4a|json)$/; + +async function pruneClips(cacheDir: string, entries: Record<string, ReportMediaEntry>): Promise<void> { + const keep = new Set<string>([REPORT_MEDIA_INDEX_FILENAME, REPORT_POSTS_DIRNAME]); + for (const e of Object.values(entries)) { + if (e.kind === "post") continue; + keep.add(e.file); + keep.add(e.file.replace(/\.[^.]+$/, ".json")); + } + for (const name of await readdir(cacheDir).catch(() => [] as string[])) { + if (keep.has(name)) continue; + // Only what this module writes: a clip, its sidecar, a temp file of either. + if (CLIP_FILE_RE.test(name) || /^[0-9a-f]{32}\.(?:mp4|m4a|json)\.tmp-/.test(name)) { + await rm(path.join(cacheDir, name), { force: true }); + } + } +} + +// The problems as lines, for a log or a terminal. +export function formatReportMediaProblems(problems: readonly ReportMediaProblem[]): string[] { + return problems.map((p) => { + const where = p.moment + ? `${p.moment}${p.citations?.length ? ` (cited by ${p.citations.join(", ")})` : ""}` + : `${p.report ?? "?"}${p.path ? ` at ${p.path}` : ""}`; + return `${p.kind}: ${where}: ${p.message}`; + }); +} + +// Read a prepared manifest, or null when there is none (never prepared, or +// unreadable). +export async function readReportMediaIndex(paths: Paths, siteId: string): Promise<ReportMediaIndex | null> { + const read = await readJsonFile(reportMediaIndexFile(paths, siteId)); + if (!read.ok) return null; + const v = read.value as Partial<ReportMediaIndex> | null; + if (!v || v.format !== REPORT_MEDIA_FORMAT || v.version !== REPORT_MEDIA_VERSION) return null; + return v as ReportMediaIndex; +}