// THE REPORTS STAGE OF COMPOSE — a site's published reports, resolved against // the corpus and written as the views the export's Reports pages read // (lib/report/views.ts names every file), with the media and stills they cite // (plans/report-sites.md, "Compose and the contract"). // // Runs for EVERY site's compose, full or cited: it first removes what an // earlier compose — of this site or another, public/ is shared — left under // `reports/`, `m/` and `media/`, so a site with no reports ships none. // // What it reads: // - the site's `reports` (site.json, in order), each report.json parsed and // validated by the document's own checker (./reportMedia.ts // loadSiteReports); any problem fails the stage; // - per cited video/audio record: transcript.cues.json when it is fresh, // else metadata.info.json and the raw transcript; when those cues are // empty, the English VTT tracks in turn, `en-orig` first (a served `en` // track can parse to no cues); // - per cited post: the channel's posts archive (lib/posts-server.ts), and // the post visibility rule (lib/postsVisibility.ts); // - the media `archilyzer reports prepare` cut and copied for the site // (./reportMedia.ts — the manifest and the cache beside it). This stage // never cuts: the build has no ffmpeg. // // What it computes: every citation's VERIFICATION, overwriting whatever the // document carried (lib/citations/verify.ts): a span's quote against its cue // window, a post's against its text; a `source` or `page` citation keeps none. // A quote that drifted fails the stage. // // What it writes, under the public dir: // reports/index.json the report index // reports//page.json each report's page view // reports//citations.{json,csv} its citations as data (the JSON is // an `archilyzer-citations` set; // never a source's `saved` copy) // reports// each cited source still // reports//history/history.json, the report's revisions and a // history/repo/… dumb-HTTP clone of them, when // `reports export` has committed // one (./reportHistory.ts) // reports//feed.{xml,json} its timeline as RSS 2.0 and JSON // Feed 1.1 (lib/report/feeds.ts), for // a report with entries on a site // with a public `siteUrl` (a feed's // links are absolute: none without) // reports//report.{html,pdf,md}, the report's exports, when // evidence-pack.zip `archilyzer reports export` made // them from the report as it is now // and each is within the publish // limit (./reportExportFiles.ts) // m/index.json, m//moment.json one moment view per cited moment, // "cited in" across every report // media/clips///-.mp4 a span's prepared clip (.m4a for // a clip cut as audio) // media/posts/// a cited post's prepared capture // — ONLY what the published reports cite. A citation whose media was not // prepared fails the stage with the list, unless `allowMissingMedia`: its page // then renders without a clip. // // THE STAGE FAILS BEFORE IT WRITES: every problem is collected first and // thrown together (ComposeReportsError), so a failed compose leaves no half // set of reports behind. import { copyFile, mkdir, readdir, readFile, rm, stat, writeFile } from "node:fs/promises"; import path from "node:path"; import { publishFileSizeProblem } from "../lib/builtExport"; import type { Paths } from "../lib/paths"; import { siteChannelSlugs, type Site } from "../lib/site"; import { isCitedSite, parseSiteUrl } from "../lib/siteSchema"; import { getSettings } from "../lib/settings"; import { postsVisibleTo } from "../lib/postsVisibility"; import { assertChannelTextReadable } from "../lib/channelMedia"; import type { ChannelConfig } from "../lib/channelConfig"; import { readChannelConfig } from "../controller/channels"; import { isCuesJsonFresh, readNormalizedTranscript } from "../controller/normalizeTranscript"; import { loadRawMetadataFromDir, platformLabelStale, summarize, } from "../lib/transcripts-server"; import type { TranscriptSummary } from "../lib/transcripts"; import { parseVtt, type Cue } from "../lib/vtt"; import { parseTranscriptJson } from "../lib/whisper"; import { WHISPER_FILENAME, englishVttsByPreference, readEnglishVttCues } from "../lib/videoStatus"; import { platformMomentUrl } from "../lib/momentUrl"; import { archiveOrgCitationLinks, type ArchiveOrgProvenance } from "../lib/archiveOrg"; import { loadArchiveOrgProvenance } from "../lib/archiveOrg-server"; import { WAYBACK_PROVENANCE_FILENAME, waybackCitationLinks, type WaybackProvenance } from "../lib/wayback"; import { loadWaybackProvenance } from "../lib/wayback-server"; import type { Platform } from "../lib/platform"; import { readAllPosts } from "../lib/posts-server"; import type { Post } from "../lib/posts"; import { readJsonFile } from "../lib/jsonFile-server"; import { momentKeyOf, momentPath, parseMomentKey, type SpanMoment } from "../lib/citations/moments"; import { CITATIONS_VERSION, type Citation, type PostCitation, type SpanCitation } from "../lib/citations/schema"; import { cueWindowText, quoteDrifted, quoteVerification, QUOTE_DRIFT_THRESHOLD } from "../lib/citations/verify"; import { buildCitedIn } from "../lib/report/citedIn"; import { renderReportJsonFeed, renderReportRss, reportFeedUrls } from "../lib/report/feeds"; import { reportHistoryPagePath } from "../lib/report/revisions"; import type { Report } from "../lib/report/schema"; import { reportCitationNumbers } from "../lib/report/uses"; import { MOMENT_INDEX_FORMAT, MOMENT_PAGE_FORMAT, MOMENTS_INDEX_PATH, REPORT_INDEX_FORMAT, REPORT_VIEWS_VERSION, REPORTS_INDEX_PATH, buildReportPageView, citedInViews, evidenceClipPath, momentViewPath, orderedCitations, reportCitationsDownloadPath, reportExportDownloadPath, reportFeedPath, reportIndexEntry, reportViewPath, REPORT_EXPORT_FORMATS, type CitationView, type CueLineView, type MomentPageView, type MomentPostView, type RecordView, type ReportIndexEntry, type ReportIndexView, type ReportDownloads, type ReportFeeds, type ReportPageView, } from "../lib/report/views"; import { citedEvidenceSpan, isAudioOnlyPlatform, type EvidenceSpan } from "../lib/evidenceClip-server"; import { citedMoments, loadSiteReports, readReportMediaIndex, reportMediaDir, siteReportDir, type ReportMediaEntry, } from "./reportMedia"; import { publishableReportExports, reportFileSha256, type PublishableReportExports } from "./reportExportFiles"; import { publishReportHistory, readReportHistoryView, reportHistoryGitDir, reportHistoryRef } from "./reportHistory"; import type { ReportHistoryRef, ReportHistoryView } from "../lib/report/revisions"; // The public dir's entries this stage owns. Every compose removes them first. export const REPORT_PUBLIC_ENTRIES: readonly string[] = ["reports", "m", "media"]; // The transcript lines a moment page shows either side of its span, and at // most how many: bounded context, never the record (plans/report-sites.md, // Risks 5). export const MOMENT_CUE_CONTEXT_SECONDS = 15; export const MOMENT_CUE_LINES_MAX = 80; export const CITATIONS_CSV_COLUMNS = [ "citation", "number", "kind", "channel", "id", "start", "end", "quote", "speaker", "date", "originalUrl", "momentPath", "quoteScore", ] as const; export type ComposeReportsProblemKind = | "missing-report" | "invalid-report" | "not-in-site" | "not-visible" | "unreadable" | "missing-record" | "no-cues" | "quote-drift" | "missing-post" | "missing-still" | "report-video" | "missing-media" | "stale-media"; export type ComposeReportsProblem = { kind: ComposeReportsProblemKind; message: string; report?: string; // `#`. citation?: string; moment?: string; // A JSON path in the report (an invalid report's problems). path?: string; }; // The kinds `allowMissingMedia` lets through: the page renders without a clip. const MEDIA_PROBLEMS: ReadonlySet = new Set(["missing-media", "stale-media"]); export class ComposeReportsError extends Error { constructor(readonly problems: ComposeReportsProblem[]) { super( `the site's reports cannot be composed (${problems.length} problem(s)):\n` + formatComposeReportsProblems(problems).map((l) => ` ${l}`).join("\n"), ); this.name = "ComposeReportsError"; } } export function formatComposeReportsProblems(problems: readonly ComposeReportsProblem[]): string[] { return problems.map((p) => { const where = p.citation ?? p.moment ?? `${p.report ?? "?"}${p.path ? ` at ${p.path}` : ""}`; return `${p.kind}: ${where}: ${p.message}`; }); } export type ComposeReportsOptions = { paths: Paths; site: Site; // Default: paths.exportPublicDir. publicDir?: string; // Let a citation without prepared media through (`--allow-missing-media`). allowMissingMedia?: boolean; // `social.x.visibility` and friends; default the live settings. settings?: { social?: { x?: { visibility?: unknown } } }; now?: () => Date; log?: (line: string) => void; }; export type ComposedReports = { // The index's entries, in the site's order. reports: ReportIndexEntry[]; // Every moment page written. moments: string[]; // Media problems let through by `allowMissingMedia`. allowed: ComposeReportsProblem[]; // The reports whose revision history was published. histories?: string[]; }; // ─── Reading the corpus ─── export type CitedRecord = { summary: Pick; cues: Cue[]; // Every transcript of the record that has cues, by file name: the // normalized cues, the whisper output, and each English VTT (`en-orig` // first). A served `en` track can be a rewrite of what was said, so a // quote is checked against each and the best match is the one shown. tracks: { name: string; cues: Cue[] }[]; // An archive.org record's provenance (its torrent, a mirror's original); // null for every other record. archiveOrg: ArchiveOrgProvenance | null; // A Wayback Machine capture's provenance (lib/wayback.ts): what the record // is an archived copy of. Null for every other record. wayback: WaybackProvenance | null; }; async function readCues(file: string, kind: "vtt" | "whisper"): Promise { try { const raw = await readFile(file, "utf8"); return kind === "vtt" ? parseVtt(raw) : parseTranscriptJson(raw); } catch { return []; } } // A cited record: its summary and its cues, or null when the data dir holds // neither a normalized transcript nor metadata. export async function readCitedRecord( channelsDir: string, slug: string, id: string, channelName?: string, ): Promise { const dir = path.join(channelsDir, slug, "data", id); const entries = await readdir(dir).catch(() => [] as string[]); if (entries.length === 0) return null; let summary: CitedRecord["summary"] | null = null; let cues: Cue[] = []; const fresh = await isCuesJsonFresh(dir); if (fresh.fresh) { const n = await readNormalizedTranscript(fresh.cuesPath); // A summary frozen before its platform was known is re-derived below. if (n && !platformLabelStale(n)) { summary = n; cues = n.cues ?? []; } } if (!summary) { const meta = await loadRawMetadataFromDir(dir); if (!meta) return null; summary = summarize(slug, id, meta, channelName); if (entries.includes(WHISPER_FILENAME)) cues = await readCues(path.join(dir, WHISPER_FILENAME), "whisper"); } // The caption-track rule (videoStatus.ts): the first English VTT, `en-orig` // first, that has cues. if (cues.length === 0) cues = (await readEnglishVttCues(dir, entries))?.cues ?? []; const tracks: CitedRecord["tracks"] = []; if (fresh.fresh) { const n = await readNormalizedTranscript(fresh.cuesPath); if (n?.cues?.length) tracks.push({ name: path.basename(fresh.cuesPath), cues: n.cues }); } if (entries.includes(WHISPER_FILENAME)) { const w = await readCues(path.join(dir, WHISPER_FILENAME), "whisper"); if (w.length > 0) tracks.push({ name: WHISPER_FILENAME, cues: w }); } for (const name of englishVttsByPreference(entries)) { const v = await readCues(path.join(dir, name), "vtt"); if (v.length > 0) tracks.push({ name, cues: v }); } if (tracks.length === 0 && cues.length > 0) tracks.push({ name: "cues", cues }); const archiveOrg = summary.platform === "archiveorg" ? await loadArchiveOrgProvenance(dir) : null; const wayback = entries.includes(WAYBACK_PROVENANCE_FILENAME) ? await loadWaybackProvenance(dir) : null; return { summary, cues, tracks, archiveOrg, wayback }; } // THE SPAN QUOTE CHECK — the one compose runs, and `archilyzer reports // verify-quotes` and `reports check` report: the quote against the cue window // of EVERY transcript the record has (lib/citations/verify.ts), the best match // is the verification (its method names the track). `tracks` is each track's // own score, so a caller can say what the `en-orig` track — the words as // spoken — makes of a quote a served `en` rewrite matched. export type SpanQuoteCheck = { verification: ReturnType; // The best track's cues (what a moment page shows), or null when the record // had no track and its default cues were used. cues: Cue[] | null; tracks: { name: string; score: number }[]; }; export function checkSpanQuote( record: Pick, c: { quote: string; start: number; end: number }, now: string, ): SpanQuoteCheck { let best: { name: string; cues: Cue[]; v: ReturnType } | null = null; const tracks: SpanQuoteCheck["tracks"] = []; for (const t of record.tracks) { const v = quoteVerification(c.quote, cueWindowText(t.cues, c.start, c.end), now); tracks.push({ name: t.name, score: v.quoteScore ?? 0 }); if (!best || (v.quoteScore ?? 0) > (best.v.quoteScore ?? 0)) best = { name: t.name, cues: t.cues, v }; } return best ? { verification: { ...best.v, method: `${best.v.method}; text: ${best.name}` }, cues: best.cues, tracks } : { verification: quoteVerification(c.quote, cueWindowText(record.cues, c.start, c.end), now), cues: null, tracks }; } // The sentence compose fails a drifted quote with. export function quoteDriftMessage(score: number | undefined): string { return ( `the quote matches ${Math.round((score ?? 0) * 100)}% of what the record says there ` + `(at least ${Math.round(QUOTE_DRIFT_THRESHOLD * 100)}% is required): quote it verbatim, or fix the span` ); } const isoDay = (uploadDate: string | undefined): string | undefined => uploadDate && /^\d{8}$/.test(uploadDate) ? `${uploadDate.slice(0, 4)}-${uploadDate.slice(4, 6)}-${uploadDate.slice(6, 8)}` : undefined; // A record in this site's corpus — the viewer's `?v=` link (a FULL site only). function corpusLink(slug: string, params: Record): string { return `/?${new URLSearchParams({ v: slug, ...params }).toString()}`; } const cueLine = (text: string) => text.replace(/\s+/g, " ").trim(); const IMAGE_EXTS = new Set([".png", ".jpg", ".jpeg", ".gif", ".webp", ".avif"]); function csvCell(v: unknown): string { if (v === undefined || v === null) return ""; const s = String(v); return /[",\r\n]/.test(s) ? `"${s.replace(/"/g, '""')}"` : s; } // A report's citations as CSV: one row per citation it cites, in number order. export function citationsCsv(view: ReportPageView): string { const rows: string[] = [CITATIONS_CSV_COLUMNS.join(",")]; for (const c of orderedCitations(view)) { const span = c.kind === "video" || c.kind === "audio" ? c : null; const recorded = c.kind === "video" || c.kind === "audio" || c.kind === "post" ? c : null; const row: Record<(typeof CITATIONS_CSV_COLUMNS)[number], unknown> = { citation: c.id, number: c.number, kind: c.kind, channel: recorded?.record.channel, id: recorded?.record.id, start: span?.start, end: span?.end, quote: c.quote, speaker: c.speaker, date: c.date ?? recorded?.record.date, originalUrl: recorded ? recorded.record.originalUrl : c.href, momentPath: recorded?.href, quoteScore: c.verification?.quoteScore, }; rows.push(CITATIONS_CSV_COLUMNS.map((k) => csvCell(row[k])).join(",")); } return rows.join("\r\n") + "\r\n"; } // A report's citations as an `archilyzer-citations` set (CITATIONS.md): the // citations it cites, verification computed, and the sources they quote — // never a source's `saved` copy. export function citationSet(report: Report, view: ReportPageView): unknown { const cited = orderedCitations(view).map((c) => c.id); const all = report.citations ?? {}; const citations = Object.fromEntries(cited.map((id) => [id, all[id]])); const sourceIds = new Set(); for (const id of cited) if (all[id].kind === "source") sourceIds.add((all[id] as { source: string }).source); if (report.subject) sourceIds.add(report.subject.source); const sources = Object.fromEntries( [...sourceIds] .filter((id) => report.sources?.[id]) .map((id) => { const { saved: _saved, ...rest } = report.sources![id]; void _saved; return [id, rest]; }), ); return { format: "archilyzer-citations", version: CITATIONS_VERSION, ...(Object.keys(sources).length > 0 ? { sources } : {}), citations, }; } const json = (value: unknown) => `${JSON.stringify(value, null, 2)}\n`; async function writeOut(publicDir: string, urlPath: string, data: string): Promise { const file = path.join(publicDir, ...urlPath.split("/").filter(Boolean)); await mkdir(path.dirname(file), { recursive: true }); await writeFile(file, data); } async function copyOut(publicDir: string, src: string, urlPath: string): Promise { const file = path.join(publicDir, ...urlPath.split("/").filter(Boolean)); await mkdir(path.dirname(file), { recursive: true }); await copyFile(src, file); } const isFile = async (p: string) => (await stat(p).catch(() => null))?.isFile() === true; const sameSpan = (a: EvidenceSpan, b: EvidenceSpan) => Math.abs(a.from - b.from) < 0.001 && Math.abs(a.to - b.to) < 0.001; // ─── Resolving: the reports as views, nothing written ─── export type ResolveSiteReportsOptions = Omit & { // Each report's downloads, as its view carries them. downloads?: (reportId: string) => ReportDownloads | undefined; // Each report's newest revision, as its view carries it. history?: (reportId: string) => ReportHistoryRef | undefined; // Each report's feeds, as its view carries them (only one with entries). feeds?: (reportId: string) => ReportFeeds | undefined; }; // The site's reports resolved against the corpus — verified, their views and // moment pages built, their prepared media matched — and nothing written. // compose writes what this answers; publish/reportExports.ts renders the same // views as files. Throws ComposeReportsError with every problem. export type ResolvedSiteReports = { // The reports (verification computed) and their views, in the site's order. reports: Report[]; views: ReportPageView[]; index: ReportIndexView; moments: MomentPageView[]; // moment key → its prepared media, in `cacheDir`. mediaOf: Map; cacheDir: string; allowed: ComposeReportsProblem[]; }; export async function resolveSiteReports(opts: ResolveSiteReportsOptions): Promise { const { paths, site } = opts; const log = opts.log ?? (() => {}); const now = (opts.now?.() ?? new Date()).toISOString(); const cited = isCitedSite(site); const problems: ComposeReportsProblem[] = []; const loaded = await loadSiteReports(paths, site); for (const p of loaded.problems) { problems.push({ kind: p.kind === "missing-report" ? "missing-report" : "invalid-report", message: p.message, report: p.report, path: p.path, }); } // Compose works on its own copy: verification is overwritten below. const reports: Report[] = loaded.problems.length > 0 ? [] : structuredClone(loaded.reports); // moment key → the cues its quote was verified against (first citation wins). const momentCues = new Map(); const settings = opts.settings ?? getSettings(); const pool = siteChannelSlugs(site); const configs = new Map(); const configOf = async (slug: string) => { if (!configs.has(slug)) configs.set(slug, await readChannelConfig(paths, slug).catch(() => null)); return configs.get(slug) ?? null; }; // Per channel, whether its text can be read (a legacy or migrating channel // cannot), as the problem's sentence or null. const unreadable = new Map(); const textProblem = async (slug: string): Promise => { if (!unreadable.has(slug)) { try { await assertChannelTextReadable(paths, slug, await configOf(slug)); unreadable.set(slug, null); } catch (e) { unreadable.set(slug, (e as Error).message); } } return unreadable.get(slug) ?? null; }; const records = new Map(); const recordOf = async (slug: string, id: string) => { const k = `${slug}/${id}`; if (!records.has(k)) records.set(k, await readCitedRecord(paths.channelsDir, slug, id, (await configOf(slug))?.name)); return records.get(k) ?? null; }; const postsByChannel = new Map>(); const postOf = async (slug: string, id: string) => { if (!postsByChannel.has(slug)) { const posts = await readAllPosts(path.join(paths.channelsDir, slug)).catch(() => [] as Post[]); postsByChannel.set(slug, new Map(posts.map((p) => [p.id, p]))); } return postsByChannel.get(slug)!.get(id) ?? null; }; // Resolve and verify every citation the reports cite. A channel outside the // site's pool, or a post the site may not carry, is refused before its // record is read. const refusedChannel = new Set(); for (const report of reports) { // The report's video and its poster: on disk, and small enough to publish. if (report.video) { const dir = siteReportDir(paths, site.siteId, report.id); for (const rel of [report.video.src, report.video.poster]) { if (!rel) continue; const st = await stat(path.join(dir, rel)).catch(() => null); const why = !st?.isFile() ? `the video file ${rel} does not exist` : publishFileSizeProblem(rel, st.size); if (why) problems.push({ kind: "report-video", report: report.id, message: why }); } } const all = report.citations ?? {}; // Only what the report cites: a citation it defines but never cites is // not in its view, and is neither checked nor published. const used = new Set(reportCitationNumbers(report).keys()); for (const [cid, c] of Object.entries(all)) { const ref = `${report.id}#${cid}`; if (!used.has(cid)) continue; if (c.kind === "source" || c.kind === "page") { delete c.verification; if (c.kind === "source" && c.image) { const src = path.join(siteReportDir(paths, site.siteId, report.id), c.image); if (!(await isFile(src))) { problems.push({ kind: "missing-still", citation: ref, report: report.id, message: `the still ${c.image} does not exist` }); } } continue; } if (!pool.has(c.channel)) { problems.push({ kind: "not-in-site", citation: ref, report: report.id, message: `channel "${c.channel}" is not one of this site's channels` }); refusedChannel.add(c.channel); continue; } const text = await textProblem(c.channel); if (text) { problems.push({ kind: "unreadable", citation: ref, report: report.id, message: text }); continue; } if (c.kind === "post") { if (!postsVisibleTo(site, await configOf(c.channel), settings)) { problems.push({ kind: "not-visible", citation: ref, report: report.id, message: `this site may not carry posts of "${c.channel}" (the post visibility rule)` }); continue; } const post = await postOf(c.channel, c.id); if (!post) { problems.push({ kind: "missing-post", citation: ref, report: report.id, message: `no post ${c.id} in the posts archive of "${c.channel}"` }); continue; } c.verification = quoteVerification(c.quote, post.text, now); } else { const record = await recordOf(c.channel, c.id); if (!record) { problems.push({ kind: "missing-record", citation: ref, report: report.id, message: `no record ${c.channel}/${c.id} (no metadata or transcript in its data dir)` }); continue; } if (record.cues.length === 0) { problems.push({ kind: "no-cues", citation: ref, report: report.id, message: `${c.channel}/${c.id} has no transcript cues to check the quote against` }); continue; } const checked = checkSpanQuote(record, c, now); c.verification = checked.verification; const mk = momentKeyOf(c); if (checked.cues && mk && !momentCues.has(mk)) momentCues.set(mk, checked.cues); } if (quoteDrifted(c.verification)) { problems.push({ kind: "quote-drift", citation: ref, report: report.id, message: quoteDriftMessage(c.verification.quoteScore), }); } } } // The prepared media, per moment. const media = await readReportMediaIndex(paths, site.siteId); const cacheDir = reportMediaDir(paths, site.siteId); const citedIn = buildCitedIn(reports); const momentInfo = new Map(citedMoments(reports).map((m) => [m.key, m])); const mediaOf = new Map(); for (const key of Object.keys(citedIn)) { const m = momentInfo.get(key); if (!m || refusedChannel.has(m.moment.channel)) continue; const entry = media?.siteId === site.siteId ? media.moments[key] : undefined; const citedBy = citedIn[key].map((e) => `${e.reportId}#${e.citationId}`); const missing = (kind: ComposeReportsProblemKind, message: string) => problems.push({ kind, moment: key, message: `${message} (cited by ${[...new Set(citedBy)].join(", ")})` }); if (!entry) { missing("missing-media", "no prepared media — run Prepare evidence media (archilyzer reports prepare)"); continue; } const files = entry.kind === "post" ? [entry.file, ...entry.media.map((f) => f.file)] : [entry.file]; const absent = []; for (const f of files) if (!(await isFile(path.join(cacheDir, f)))) absent.push(f); if (absent.length > 0) { missing("missing-media", `the prepared media is gone from the cache (${absent.join(", ")}) — prepare again`); continue; } if (entry.kind !== "post" && m.moment.kind === "span") { // The clip was cut for a span and pad; a report changed since must not // ship the old cut under the new moment's page. const sidecar = await readJsonFile(path.join(cacheDir, entry.file.replace(/\.[^.]+$/, ".json"))); const cut = sidecar.ok ? (sidecar.value as { span?: EvidenceSpan }).span : undefined; const want = await citedEvidenceSpan(paths.channelsDir, m.moment.channel, m.moment.id, { start: m.moment.start, end: m.moment.end, pad: m.pad }); if (cut && !sameSpan(cut, want)) { missing("stale-media", `the clip was cut for ${cut.from}–${cut.to} s, the reports now cite ${want.from}–${want.to} s — prepare again`); continue; } } mediaOf.set(key, entry); } const allowed = opts.allowMissingMedia ? problems.filter((p) => MEDIA_PROBLEMS.has(p.kind)) : []; const fatal = problems.filter((p) => !allowed.includes(p)); if (fatal.length > 0) throw new ComposeReportsError(fatal); for (const line of formatComposeReportsProblems(allowed)) log(`[reports] allowed (--allow-missing-media): ${line}`); // ─── The views ─── const recordView = async (c: SpanCitation | PostCitation): Promise => { const config = await configOf(c.channel); if (c.kind === "post") { const post = (await postOf(c.channel, c.id))!; return defined({ channel: c.channel, channelTitle: config?.name ?? post.authorName, id: c.id, date: post.createdAt.slice(0, 10), platform: post.platform, originalUrl: post.url, corpusUrl: cited ? undefined : corpusLink(`${c.channel}/${c.id}`, { vm: "post" }), }); } const { summary, archiveOrg, wayback } = (await recordOf(c.channel, c.id))!; const audioOnly = isAudioOnlyPlatform(config?.platform); const seconds = Math.max(0, Math.floor(c.start)); if (wayback) { // An archived copy (Wayback Machine): the original, named as the // original and as possibly gone, then the copy, which plays. const links = waybackCitationLinks(wayback, { originalMomentUrl: platformMomentUrl(wayback.originalUrl, null, c.start), }); return defined({ channel: c.channel, channelTitle: config?.name ?? (summary.channel || undefined), id: c.id, title: summary.title, date: isoDay(summary.uploadDate), platform: summary.platform, originalUrl: links.original.url, originalLabel: links.original.label, downloads: [links.copy], corpusUrl: cited ? undefined : corpusLink(summary.slug ?? `${c.channel}/${summary.id}`, seconds > 0 ? { t: String(seconds) } : {}), }); } if (summary.platform === "archiveorg") { // archive.org: the original (YouTube at the second, for a mirror; else // the archive.org page) plus the downloads a reader can check it from. const links = archiveOrgCitationLinks({ webpageUrl: summary.webpageUrl, provenance: archiveOrg, seconds: c.start }); return defined({ channel: c.channel, channelTitle: config?.name ?? (summary.channel || undefined), id: c.id, title: summary.title, date: isoDay(summary.uploadDate), platform: summary.platform, originalUrl: links.original?.url ?? (summary.webpageUrl || undefined), originalLabel: links.original?.label, downloads: links.downloads.length > 0 ? links.downloads : undefined, corpusUrl: cited ? undefined : corpusLink(summary.slug ?? `${c.channel}/${summary.id}`, seconds > 0 ? { t: String(seconds) } : {}), }); } return defined({ channel: c.channel, channelTitle: config?.name ?? (summary.channel || undefined), id: c.id, title: summary.title, date: isoDay(summary.uploadDate), platform: summary.platform, originalUrl: audioOnly ? summary.webpageUrl || undefined : (platformMomentUrl(summary.webpageUrl, summary.platform as Platform, c.start) ?? undefined), // The viewer keys a record by its PUBLISHED slug, which on a platform // with two ids is not the data dir's name. corpusUrl: cited ? undefined : corpusLink(summary.slug ?? `${c.channel}/${summary.id}`, seconds > 0 ? { t: String(seconds) } : {}), }); }; // buildReportPageView's resolver is synchronous: resolve every cited // record first, by the citation it is resolved for. const recordViews = new WeakMap(); for (const report of reports) { const all = report.citations ?? {}; for (const id of reportCitationNumbers(report).keys()) { const c = all[id]; if (c.kind === "video" || c.kind === "audio" || c.kind === "post") recordViews.set(c, await recordView(c)); } } const postShot = (c: { channel: string; id: string }): string | undefined => { const entry = mediaOf.get(`${c.channel}/${c.id}`); return entry?.kind === "post" ? `/media/${entry.file}` : undefined; }; const views: ReportPageView[] = reports.map((report) => buildReportPageView(report, { record: (c) => recordViews.get(c)!, post: (c) => { const post = postsByChannel.get(c.channel)?.get(c.id); return post ? { author: postAuthor(post), text: post.text, shot: postShot(c) } : undefined; }, downloads: opts.downloads?.(report.id), history: opts.history?.(report.id), feeds: opts.feeds?.(report.id), }), ); const index: ReportIndexView = { format: REPORT_INDEX_FORMAT, version: REPORT_VIEWS_VERSION, reports: views.map(reportIndexEntry), }; const moments: MomentPageView[] = []; for (const [key, entries] of Object.entries(citedIn)) { const first = entries[0]; const report = reports.find((r) => r.id === first.reportId)!; const c = report.citations![first.citationId] as Citation; if (c.kind !== "video" && c.kind !== "audio" && c.kind !== "post") continue; const view = views.find((v) => v.id === report.id)!.citations[first.citationId] as Extract< CitationView, { kind: "video" | "audio" | "post" } >; const entry = mediaOf.get(key); const common = { format: MOMENT_PAGE_FORMAT, version: REPORT_VIEWS_VERSION, key, record: view.record, quote: c.quote, ...(c.speaker ? { speaker: c.speaker } : {}), ...(c.date ? { date: c.date } : {}), ...(c.verification ? { verification: c.verification } : {}), citedIn: citedInViews(entries, views), } as const; if (c.kind === "post") { const post = postsByChannel.get(c.channel)!.get(c.id)!; const postView: MomentPostView = { author: postAuthor(post), text: post.text, ...(entry?.kind === "post" ? { shot: `/media/${entry.file}`, ...(entry.media.length > 0 ? { media: entry.media.map((f) => ({ src: `/media/${f.file}`, kind: IMAGE_EXTS.has(path.extname(f.file).toLowerCase()) ? ("image" as const) : ("video" as const), })), } : {}), } : {}), }; moments.push({ ...common, kind: "post", post: postView }); continue; } const m = parseMomentKey(key) as SpanMoment; const info = momentInfo.get(key)!; const span = await citedEvidenceSpan(paths.channelsDir, m.channel, m.id, { start: m.start, end: m.end, pad: info.pad }); // The transcript the moment's first citation matched best (see the // verification above), else the record's default cues. const cues = momentCues.get(key) ?? records.get(`${c.channel}/${c.id}`)!.cues; const from = m.start - MOMENT_CUE_CONTEXT_SECONDS; const to = m.end + MOMENT_CUE_CONTEXT_SECONDS; const lines: CueLineView[] = cues .filter((q) => q.end > from && q.start < to) .slice(0, MOMENT_CUE_LINES_MAX) .map((q) => ({ start: q.start, end: q.end, text: cueLine(q.text), inSpan: q.end > m.start && q.start < m.end })); const audio = entry?.kind === "audio"; moments.push({ ...common, // A span whose clip had to be cut as sound is heard, not watched. kind: audio ? "audio" : info.kind === "audio" ? "audio" : "video", start: m.start, end: m.end, ...(entry && entry.kind !== "post" ? { clip: { src: clipPath(m, entry.kind), start: span.from, end: span.to } } : {}), cues: lines, }); } moments.sort((a, b) => (a.key < b.key ? -1 : a.key > b.key ? 1 : 0)); return { reports, views, index, moments, mediaOf, cacheDir, allowed }; } // ─── The stage ─── export async function composeReports(opts: ComposeReportsOptions): Promise { const { paths, site } = opts; const publicDir = opts.publicDir ?? paths.exportPublicDir; const log = opts.log ?? (() => {}); // Whatever an earlier compose left. rm removes a link, never its target (a // worktree's public/ entries may be links into the primary checkout). for (const entry of REPORT_PUBLIC_ENTRIES) { await rm(path.join(publicDir, entry), { recursive: true, force: true }); } if ((site.reports ?? []).length === 0) { log("[reports] none published."); return { reports: [], moments: [], allowed: [] }; } // Each report's exports (publish/reportExports.ts), made from the report // as it is now and small enough to publish; the citations as files always. const exportsOf = new Map(); for (const id of site.reports ?? []) { const found = await publishableReportExports(paths, site.siteId, id); exportsOf.set(id, found); for (const note of found.notes) log(`[reports] ${id}: ${note}`); } // Each report's revision history (publish/reportHistory.ts), read before // anything is written; a report never exported has none. const histories = new Map(); const historyProblems: ComposeReportsProblem[] = []; for (const id of site.reports ?? []) { const gitDir = reportHistoryGitDir(paths, site.siteId, id); try { const view = await readReportHistoryView(gitDir, { reportId: id, siteUrl: site.siteUrl }); if (!view) continue; const ref = reportHistoryRef(view, await reportFileSha256(paths, site.siteId, id)); if (!ref.current) log(`[reports] ${id}: report.json changed since revision ${ref.revision} — export again to commit it`); histories.set(id, { view, ref, gitDir }); } catch (e) { historyProblems.push({ kind: "unreadable", report: id, message: `its revision history (${gitDir}) cannot be read: ${String((e as Error)?.message ?? e).split("\n")[0]}`, }); } } if (historyProblems.length > 0) throw new ComposeReportsError(historyProblems); // A timeline's feeds link absolutely: a site with no public URL publishes none. const siteUrl = parseSiteUrl(site.siteUrl); const { reports, views, index, moments, mediaOf, cacheDir, allowed } = await resolveSiteReports({ ...opts, history: (id) => histories.get(id)?.ref, feeds: siteUrl ? (id) => reportFeedUrls(siteUrl, id) : undefined, downloads: (id) => ({ ...Object.fromEntries( REPORT_EXPORT_FORMATS.filter((f) => exportsOf.get(id)?.files[f]).map((f) => [f, reportExportDownloadPath(id, f)]), ), json: reportCitationsDownloadPath(id, "json"), csv: reportCitationsDownloadPath(id, "csv"), }), }); // ─── Writing ─── await writeOut(publicDir, REPORTS_INDEX_PATH, json(index)); for (let i = 0; i < reports.length; i++) { const report = reports[i]; const view = views[i]; await writeOut(publicDir, reportViewPath(report.id), json(view)); await writeOut(publicDir, reportCitationsDownloadPath(report.id, "json"), json(citationSet(report, view))); await writeOut(publicDir, reportCitationsDownloadPath(report.id, "csv"), citationsCsv(view)); if (siteUrl && view.feeds) { await writeOut(publicDir, reportFeedPath(report.id, "rss"), renderReportRss(view, { siteUrl })); await writeOut(publicDir, reportFeedPath(report.id, "json"), json(renderReportJsonFeed(view, { siteUrl }))); log(`[reports] ${report.id}: feeds of ${view.entries?.length ?? 0} timeline entr${view.entries?.length === 1 ? "y" : "ies"}.`); } if (report.video && view.video) { const dir = siteReportDir(paths, site.siteId, report.id); await copyOut(publicDir, path.join(dir, report.video.src), view.video.src); if (report.video.poster && view.video.poster) { await copyOut(publicDir, path.join(dir, report.video.poster), view.video.poster); } } for (const c of orderedCitations(view)) { if (c.kind !== "source" || !c.image) continue; const rel = (report.citations![c.id] as { image: string }).image; await copyOut(publicDir, path.join(siteReportDir(paths, site.siteId, report.id), rel), c.image); } const exported = exportsOf.get(report.id)?.files ?? {}; for (const f of REPORT_EXPORT_FORMATS) { const src = exported[f]; if (src) await copyOut(publicDir, src, reportExportDownloadPath(report.id, f)); } const history = histories.get(report.id); if (history) { const files = await publishReportHistory({ gitDir: history.gitDir, publicDir, view: history.view }); log(`[reports] ${report.id}: revision ${history.ref.revision}, history published (${files.length} repository files).`); } } await writeOut( publicDir, MOMENTS_INDEX_PATH, json({ format: MOMENT_INDEX_FORMAT, version: REPORT_VIEWS_VERSION, moments: moments.map((m) => m.key) }), ); let copied = 0; for (const m of moments) { await writeOut(publicDir, momentViewPath(m.key), json(m)); const entry = mediaOf.get(m.key); if (!entry) continue; if (entry.kind === "post") { for (const f of [entry.file, ...entry.media.map((x) => x.file)]) { await copyOut(publicDir, path.join(cacheDir, f), `/media/${f}`); copied++; } } else { await copyOut(publicDir, path.join(cacheDir, entry.file), clipPath(parseMomentKey(m.key) as SpanMoment, entry.kind)); copied++; } } log( `[reports] ${reports.length} report(s), ${moments.length} moment page(s), ${copied} media file(s)` + `${allowed.length ? `, ${allowed.length} without media` : ""}.`, ); return { reports: index.reports, moments: moments.map((m) => m.key), allowed, histories: [...histories.keys()] }; } // A clip's published path: the moment's (lib/report/views.ts), `.m4a` for a // clip cut as sound. export function clipPath(m: SpanMoment, kind: "video" | "audio"): string { const p = evidenceClipPath(m); return kind === "audio" ? p.replace(/\.mp4$/, ".m4a") : p; } function postAuthor(post: Post): string { const handle = post.author.startsWith("@") ? post.author : `@${post.author}`; return post.authorName ? `${post.authorName} (${handle})` : handle; } const defined = (o: T): T => Object.fromEntries(Object.entries(o).filter(([, v]) => v !== undefined)) as T; // The site-root routes the reports add to a sitemap: the index, each report // (and its history page, when it has one), each moment page. export function reportRoutes(composed: Pick): string[] { if (composed.reports.length === 0) return []; const histories = new Set(composed.histories ?? []); return [ "/reports/", ...composed.reports.flatMap((r) => (histories.has(r.id) ? [r.href, reportHistoryPagePath(r.id)] : [r.href])), ...composed.moments.map((k) => momentPath(k)), ]; }