// PREPARE A SITE'S EVIDENCE MEDIA — every clip and post capture its published // reports cite, cut and copied on the host, before the site's build // (plans/report-sites.md, "Evidence media"). `archilyzer reports prepare // ` and the editor's `reports-prepare` job both run `prepareReportMedia`. // // What it reads: the site's `reports` (site.json, in order), each // `sites//reports//report.json`, parsed and validated by the // report document's own checker (lib/report/validate.ts). What it does, per // cited MOMENT (lib/citations/moments.ts — two citations of one span share one // page and so one clip, cut with the wider of their pads): // // video / audio span lib/evidenceClip-server.ts finds the span's media on // disk (clip window, saved container, audio) and cuts it // into the cache, or answers why it cannot; // post ./citedPostCaptures.ts copies the post's screenshot // and media — only cited posts — into the cache. // // A citation must resolve against the site's own `channels` (the pool a cited // site's citations may draw on), and a post must be one the site may carry // (lib/postsVisibility.ts — a public site never shows a private platform's // posts, cited or not). // // What it writes: `.export-index/sites//report-media/` — // `.mp4` / `.m4a` (+ `.json`) the clips // `posts///…` the cited captures // `index.json` the manifest: // { format, version, siteId, preparedAt, // moments: { : { kind, file, bytes, sha256, width, height, // durationSec, media? } }, // problems: [ { kind, message, report?, citations?, moment?, path? } ] } // — and nothing else: a clip or capture no moment names is removed, so the // cache holds exactly what the site cites. The build copies from here and // never runs ffmpeg (it has none). // // A RUN WITH PROBLEMS STILL WRITES THE MANIFEST (the editor lists the problems // from it) and FAILS: a citation without media, a clip over the size limit, an // invalid or missing report. Nothing here fetches — the editor's fetch-windows // (for spans: `missingEvidenceWindows` below says which), persist, and Capture // posts (for posts) fill what is missing. // // AN UNMOUNTED DRIVE IS NOT A MISSING CLIP. A span whose media is not found on // a channel whose media tier is not reachable (lib/channelMedia.ts) is // reported as unreachable, with the reason, rather than as missing. import type { Dirent } from "node:fs"; import { readdir, rm } from "node:fs/promises"; import path from "node:path"; import { readJsonFile, writeJsonAtomic } from "../lib/jsonFile-server"; import { getPaths, type Paths } from "../lib/paths"; import { getSettings } from "../lib/settings"; import { getSite, listSiteIds, siteChannelSlugs, siteDir, siteIndexDir, type Site } from "../lib/site"; import { postsVisibleTo } from "../lib/postsVisibility"; import { inspectChannelMedia } from "../lib/channelMedia"; import type { ChannelConfig } from "../lib/channelConfig"; import { readChannelConfig } from "../controller/channels"; import { momentKey, momentOf, momentProblem, type Moment } from "../lib/citations/moments"; import type { CitationPad } from "../lib/citations/schema"; import { parseReport } from "../lib/report/validate"; import { isReportId, type Report } from "../lib/report/schema"; import { isPermanentlyGone } from "../lib/availability"; import { loadAvailability } from "../lib/availability-server"; import { MAX_CLIP_WINDOW_SECONDS } from "../lib/clipWindow"; import { citedEvidenceSpan, isAudioOnlyPlatform, prepareEvidenceClip, resolveEvidenceSource, widerPad, type EvidenceKind, type EvidenceMedia, } from "../lib/evidenceClip-server"; import { copyCitedPostCaptures, REPORT_POSTS_DIRNAME, type CopiedFile } from "./citedPostCaptures"; export const REPORT_MEDIA_FORMAT = "archilyzer-report-media"; export const REPORT_MEDIA_VERSION = 1; export const REPORT_MEDIA_DIRNAME = "report-media"; export const REPORT_MEDIA_INDEX_FILENAME = "index.json"; // `.export-index/sites//report-media/`: the prepared media, not served. export function reportMediaDir(paths: Paths, siteId: string): string { return path.join(siteIndexDir(paths, siteId), REPORT_MEDIA_DIRNAME); } export function reportMediaIndexFile(paths: Paths, siteId: string): string { return path.join(reportMediaDir(paths, siteId), REPORT_MEDIA_INDEX_FILENAME); } // `sites//reports//`: a report's directory. export function siteReportDir(paths: Paths, siteId: string, reportId: string): string { return path.join(siteDir(paths, siteId), "reports", reportId); } export function siteReportFile(paths: Paths, siteId: string, reportId: string): string { return path.join(siteReportDir(paths, siteId, reportId), "report.json"); } // Every report directory of a site, published or draft: a directory under // `sites//reports/` whose name is a report id. Anything else there (a // stray file, a bad name) is not a report. Sorted. Whether one is PUBLISHED is // the site's `reports` list (site.json), not anything on disk. export async function listReportDirs(paths: Paths, siteId: string): Promise { let entries: Dirent[]; try { entries = await readdir(path.join(siteDir(paths, siteId), "reports"), { withFileTypes: true }); } catch { return []; } return entries .filter((e) => e.isDirectory() && isReportId(e.name)) .map((e) => e.name) .sort((a, b) => a.localeCompare(b)); } export type ReportMediaEntry = | EvidenceMedia | { kind: "post"; // The screenshot. file: string; bytes: number; sha256: string; width: number | null; height: number | null; durationSec: null; // The post's attached media, copied beside it. media: CopiedFile[]; }; export type ReportMediaProblemKind = | "missing-report" | "invalid-report" | "not-in-site" | "not-visible" | "missing-media" | "unreachable" | "too-big" | "cut-failed"; export type ReportMediaProblem = { kind: ReportMediaProblemKind; message: string; report?: string; // `#` for every citation of the moment. citations?: string[]; moment?: string; // A JSON path in the report (an invalid report's problems). path?: string; }; export type ReportMediaIndex = { format: typeof REPORT_MEDIA_FORMAT; version: typeof REPORT_MEDIA_VERSION; siteId: string; preparedAt: string; moments: Record; problems: ReportMediaProblem[]; }; // One cited moment and everything that cites it. type CitedMoment = { key: string; moment: Moment; // A span cited as video anywhere is a video clip; else audio. kind: EvidenceKind | "post"; pad?: CitationPad; citedBy: string[]; }; // The published reports of a site, read and validated. A report that does not // parse contributes its problems and no moments; one that parses with value // problems contributes both — its media is still prepared, so fixing the // document does not wait on a re-cut. export async function loadSiteReports( paths: Paths, site: Site, ): Promise<{ reports: Report[]; problems: ReportMediaProblem[] }> { const reports: Report[] = []; const problems: ReportMediaProblem[] = []; for (const id of site.reports ?? []) { const file = siteReportFile(paths, site.siteId, id); const read = await readJsonFile(file); if (!read.ok) { problems.push({ kind: read.reason === "absent" ? "missing-report" : "invalid-report", report: id, message: read.reason === "absent" ? `the site lists report "${id}", but sites/${site.siteId}/reports/${id}/report.json does not exist` : `sites/${site.siteId}/reports/${id}/report.json is not readable JSON`, }); continue; } const parsed = parseReport(read.value, { id }); for (const p of parsed.problems) { problems.push({ kind: "invalid-report", report: id, path: p.path, message: p.message }); } if (parsed.ok) reports.push(parsed.value); } return { reports, problems }; } // Every moment the reports cite, keyed, in first-cited order. export function citedMoments(reports: readonly Report[]): CitedMoment[] { const byKey = new Map(); for (const report of reports) { for (const [cid, c] of Object.entries(report.citations ?? {})) { const moment = momentOf(c); // A kind without a page, or one validation already names as unsafe. if (!moment || momentProblem(moment)) continue; const key = momentKey(moment); const kind: CitedMoment["kind"] = c.kind === "post" ? "post" : c.kind === "audio" ? "audio" : "video"; const pad = c.kind === "video" || c.kind === "audio" ? c.pad : undefined; const seen = byKey.get(key); if (!seen) { byKey.set(key, { key, moment, kind, pad, citedBy: [`${report.id}#${cid}`] }); continue; } seen.citedBy.push(`${report.id}#${cid}`); seen.pad = widerPad(seen.pad, pad); if (kind === "video") seen.kind = "video"; } } return [...byKey.values()]; } export type PrepareReportMediaOptions = { siteId: string; paths?: Paths; onLog?: (line: string) => void; signal?: AbortSignal; // `social.x.visibility` and friends; default the live settings. settings?: { social?: { x?: { visibility?: unknown } } }; now?: () => Date; }; const mib = (n: number) => `${(n / 1024 / 1024).toFixed(1)} MiB`; export async function prepareReportMedia(opts: PrepareReportMediaOptions): Promise { const paths = opts.paths ?? getPaths(); const log = opts.onLog ?? (() => {}); const { siteId } = opts; if (!listSiteIds(paths).includes(siteId)) { throw new Error(`no site "${siteId}" (no sites/${siteId}/site.json)`); } const site = getSite(siteId, paths); const settings = opts.settings ?? getSettings(); const cacheDir = reportMediaDir(paths, siteId); const { reports, problems } = await loadSiteReports(paths, site); log(`${siteId}: ${site.reports?.length ?? 0} published report(s), ${reports.length} readable.`); const moments = citedMoments(reports); log(`${moments.length} cited moment(s) with media.`); const pool = siteChannelSlugs(site); const configs = new Map(); const configOf = async (slug: string) => { if (!configs.has(slug)) configs.set(slug, await readChannelConfig(paths, slug)); return configs.get(slug) ?? null; }; const entries: Record = {}; const fail = (m: CitedMoment, kind: ReportMediaProblemKind, message: string) => { problems.push({ kind, moment: m.key, citations: m.citedBy, message }); log(` ✗ ${m.key}: ${message}`); }; const posts: CitedMoment[] = []; for (const m of moments) { if (opts.signal?.aborted) throw new Error("prepare cancelled"); const slug = m.moment.channel; if (!pool.has(slug)) { fail(m, "not-in-site", `channel "${slug}" is not one of this site's channels`); continue; } const config = await configOf(slug); if (m.kind === "post") { if (!postsVisibleTo(site, config, settings)) { fail(m, "not-visible", `this site may not carry posts of "${slug}" (the post visibility rule)`); continue; } posts.push(m); continue; } if (m.moment.kind !== "span") continue; const span = await citedEvidenceSpan(paths.channelsDir, slug, m.moment.id, { start: m.moment.start, end: m.moment.end, pad: m.pad }); const r = await prepareEvidenceClip({ channelsDir: paths.channelsDir, slug, id: m.moment.id, kind: m.kind, span, cacheDir, audioOnlyRecord: isAudioOnlyPlatform(config?.platform), ffmpegBin: paths.ffmpegBin, ffprobeBin: paths.ffprobeBin, signal: opts.signal, }); if (r.ok) { entries[m.key] = r.media; log( ` ${r.cached ? "=" : "+"} ${m.key} ← ${r.source.kind} ${r.source.name}: ` + `${r.media.file} (${mib(r.media.bytes)}${r.cached ? ", cached" : ""})`, ); continue; } if (r.reason === "missing") { const where = await inspectChannelMedia(paths, slug, config, { fresh: true }).catch(() => null); if (where && where.status !== "ok" && where.status !== "in-place") { fail(m, "unreachable", `${r.message}; the channel's media is ${where.status}${where.detail ? ` (${where.detail})` : ""}`); continue; } fail(m, "missing-media", `${r.message} — fetch the window or persist the video in the editor`); continue; } fail(m, r.reason === "too-big" ? "too-big" : r.reason === "no-audio" ? "missing-media" : "cut-failed", r.message); } if (opts.signal?.aborted) throw new Error("prepare cancelled"); const copiedPosts = await copyCitedPostCaptures({ channelsDir: paths.channelsDir, destRoot: cacheDir, posts: posts.map((m) => ({ channel: m.moment.channel, id: m.moment.id })), }); const postMoment = new Map(posts.map((m) => [m.key, m])); for (const c of copiedPosts.copied) { const key = `${c.channel}/${c.id}`; entries[key] = { kind: "post", ...c.shot, durationSec: null, media: c.media }; log(` + ${key}: screenshot${c.media.length ? ` and ${c.media.length} media file(s)` : ""}`); } for (const miss of copiedPosts.missing) { const m = postMoment.get(`${miss.channel}/${miss.id}`); if (m) fail(m, "missing-media", miss.message); } // The cache holds what the manifest names: a clip no moment names goes. await pruneClips(cacheDir, entries); const index: ReportMediaIndex = { format: REPORT_MEDIA_FORMAT, version: REPORT_MEDIA_VERSION, siteId, preparedAt: (opts.now?.() ?? new Date()).toISOString(), moments: Object.fromEntries(Object.keys(entries).sort().map((k) => [k, entries[k]])), problems, }; await writeJsonAtomic(reportMediaIndexFile(paths, siteId), index, { mkdir: true }); const bytes = Object.values(entries).reduce( (n, e) => n + e.bytes + (e.kind === "post" ? e.media.reduce((s, f) => s + f.bytes, 0) : 0), 0, ); log( `${Object.keys(entries).length} of ${moments.length} moment(s) prepared (${mib(bytes)}); ` + `${problems.length} problem(s).`, ); return index; } const CLIP_FILE_RE = /^[0-9a-f]{32}\.(?:mp4|m4a|json)$/; async function pruneClips(cacheDir: string, entries: Record): Promise { const keep = new Set([REPORT_MEDIA_INDEX_FILENAME, REPORT_POSTS_DIRNAME]); for (const e of Object.values(entries)) { if (e.kind === "post") continue; keep.add(e.file); keep.add(e.file.replace(/\.[^.]+$/, ".json")); } for (const name of await readdir(cacheDir).catch(() => [] as string[])) { if (keep.has(name)) continue; // Only what this module writes: a clip, its sidecar, a temp file of either. if (CLIP_FILE_RE.test(name) || /^[0-9a-f]{32}\.(?:mp4|m4a|json)\.tmp-/.test(name)) { await rm(path.join(cacheDir, name), { force: true }); } } } // The problems as lines, for a log or a terminal. export function formatReportMediaProblems(problems: readonly ReportMediaProblem[]): string[] { return problems.map((p) => { const where = p.moment ? `${p.moment}${p.citations?.length ? ` (cited by ${p.citations.join(", ")})` : ""}` : `${p.report ?? "?"}${p.path ? ` at ${p.path}` : ""}`; return `${p.kind}: ${where}: ${p.message}`; }); } // Read a prepared manifest, or null when there is none (never prepared, or // unreadable). export async function readReportMediaIndex(paths: Paths, siteId: string): Promise { const read = await readJsonFile(reportMediaIndexFile(paths, siteId)); if (!read.ok) return null; const v = read.value as Partial | null; if (!v || v.format !== REPORT_MEDIA_FORMAT || v.version !== REPORT_MEDIA_VERSION) return null; return v as ReportMediaIndex; } // --------------------------------------------------------------------------- // WHICH WINDOWS A SITE STILL NEEDS — the list the editor's `fetch-windows` job // takes, so "fetch what the reports cite" is one request rather than a loop. // // The same enumeration prepare runs (loadSiteReports, citedMoments, // citedEvidenceSpan), and the same question prepare asks first // (resolveEvidenceSource) — WITHOUT cutting anything. A span prepare would // call `missing-media` comes back in `missing` as a window to fetch; a span // that cannot be fetched as a window comes back in `unfetchable` with the // reason, so a dry run tells the operator what no fetch will fix: // // not-in-site the channel is not one of the site's (prepare's not-in-site) // unreachable the channel's media is on a drive that is not answering — // not missing, and a fetch would write beside the wrong thing // gone availability.json says deleted, private or members-only // audio-only a feed record: its media is the episode's audio, which a // download fetches; a window selector has no picture to pick // too-long longer than one window may be — persist the video instead // // Spans already on disk count in `onDisk`. Posts are not windows and are left // to Capture posts. // --------------------------------------------------------------------------- export type EvidenceWindow = { slug: string; id: string; from: number; to: number; // `#` of the first citation of the moment. clipId: string; // Every citation of the moment. citations: string[]; // The first citation's quote, for the window's provenance. reason?: string; }; export type UnfetchableReason = "not-in-site" | "unreachable" | "gone" | "audio-only" | "too-long"; export type UnfetchableEvidenceWindow = EvidenceWindow & { why: UnfetchableReason; message: string; }; export type MissingEvidenceWindows = { siteId: string; missing: EvidenceWindow[]; unfetchable: UnfetchableEvidenceWindow[]; // Cited spans whose media is already on disk. onDisk: number; // The reports that could not be read (prepare's missing/invalid-report). problems: ReportMediaProblem[]; }; // A window fetched for a span must hold it: its name is two decimals, so the // start rounds down and the end up. const floor2 = (n: number) => Math.floor(n * 100 + 1e-6) / 100; const ceil2 = (n: number) => Math.ceil(n * 100 - 1e-6) / 100; export async function missingEvidenceWindows(paths: Paths, siteId: string): Promise { if (!listSiteIds(paths).includes(siteId)) { throw new Error(`no site "${siteId}" (no sites/${siteId}/site.json)`); } const site = getSite(siteId, paths); const { reports, problems } = await loadSiteReports(paths, site); const quotes = new Map(); for (const report of reports) { for (const [cid, c] of Object.entries(report.citations ?? {})) { if (typeof c.quote === "string" && c.quote.trim()) quotes.set(`${report.id}#${cid}`, c.quote.trim()); } } const pool = siteChannelSlugs(site); const configs = new Map(); const configOf = async (slug: string) => { if (!configs.has(slug)) configs.set(slug, await readChannelConfig(paths, slug)); return configs.get(slug) ?? null; }; const out: MissingEvidenceWindows = { siteId, missing: [], unfetchable: [], onDisk: 0, problems }; for (const m of citedMoments(reports)) { if (m.kind === "post" || m.moment.kind !== "span") continue; const slug = m.moment.channel; const id = m.moment.id; const span = await citedEvidenceSpan(paths.channelsDir, slug, id, { start: m.moment.start, end: m.moment.end, pad: m.pad }); const clipId = m.citedBy[0]; const reason = quotes.get(clipId); const window: EvidenceWindow = { slug, id, from: floor2(span.from), to: ceil2(span.to), clipId, citations: m.citedBy, ...(reason ? { reason } : {}), }; const unfetchable = (why: UnfetchableReason, message: string) => out.unfetchable.push({ ...window, why, message }); if (!pool.has(slug)) { unfetchable("not-in-site", `channel "${slug}" is not one of this site's channels`); continue; } const config = await configOf(slug); const audioOnly = isAudioOnlyPlatform(config?.platform); const source = await resolveEvidenceSource({ channelsDir: paths.channelsDir, slug, id, span, audio: m.kind === "audio" || audioOnly, ffprobeBin: paths.ffprobeBin, }); if (source) { out.onDisk += 1; continue; } const where = await inspectChannelMedia(paths, slug, config, { fresh: true }).catch(() => null); if (where && where.status !== "ok" && where.status !== "in-place") { unfetchable("unreachable", `the channel's media is ${where.status}${where.detail ? ` (${where.detail})` : ""}`); continue; } const availability = (await loadAvailability(path.join(paths.channelsDir, slug, "data", id)))?.availability; if (isPermanentlyGone(availability)) { unfetchable("gone", `the source says the video is ${availability}`); continue; } if (audioOnly) { unfetchable("audio-only", "a feed record's media is its audio download, not a window"); continue; } if (window.to - window.from > MAX_CLIP_WINDOW_SECONDS) { unfetchable( "too-long", `${Math.round(window.to - window.from)}s is longer than a window may be (${MAX_CLIP_WINDOW_SECONDS}s) — persist the video`, ); continue; } out.missing.push(window); } return out; }