Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit d4b2cd48ec0eaaeec4c74bf65cc467830198ea87
parent a38ef94914b2d5ecfc92a5799231fcc96dd28a52
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 13:56:23 -0400

reports: export a report as files (report.html, report.pdf, report.md, evidence-pack.zip)

`archilyzer reports export <site> [--report <id>] [--formats html,pdf,md,zip]`
writes each published report, resolved exactly as compose resolves it, into
.export-index/sites/<site>/report-exports/<id>/ with an export.json manifest
(sizes, checksums, the report.json sha256, the footer, notes). report.html is
one self-contained file (inline CSS, no script, stills and post screenshots
recompressed into data URIs, clips linked); report.pdf prints it through
headless Chromium and is skipped with a note where there is none;
report.md is plain Markdown with numbered references; the evidence pack is a
deterministic zip of an HTML variant playing its own media/. `reports prepare`
exports at its end when nothing is missing.

Compose is split into resolveSiteReports (the views, nothing written) and the
write; it publishes an export only when it was made from the report.json as
it is now and is within the shared 24 MiB publish limit
(PUBLISH_MAX_FILE_BYTES, now also the source mirror's and the evidence
clips'). The footer carries the report's date and sha256; exportFooterFor is
where the revision history fills in revision and commit.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/bin/archilyzer.ts | 26+++++++++++++++++++++++++-
Acommon/bin/reports-export.ts | 38++++++++++++++++++++++++++++++++++++++
Mcommon/bin/reports-prepare.ts | 36+++++++++++++++++++++++++++++-------
Mcommon/lib/builtExport.ts | 15+++++++++++++++
Mcommon/lib/evidenceClip-server.ts | 3++-
Acommon/lib/report/exportHtml.ts | 624+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/report/exportMarkdown.ts | 214+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/report/views.ts | 28+++++++++++++++++++++++++---
Mcommon/publish/composeReports.ts | 95++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------------
Acommon/publish/reportExportFiles.ts | 134+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/publish/reportExports.ts | 592+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/publish/reportMedia.test.ts | 28+++++++++++++++++++++++++++-
Mcommon/publish/source.ts | 3++-
Mcommon/social/playwrightRuntime.ts | 11++++++++++-
14 files changed, 1814 insertions(+), 33 deletions(-)

diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts @@ -159,7 +159,7 @@ export const COMMANDS: Command[] = [ { path: ["reports", "prepare"], usage: - "<id> cut every clip and copy every post capture the site's published reports cite into its report-media cache, before its build (exit 1 when a citation lacks media; default id: SITE_ID)", + "<id> cut every clip and copy every post capture the site's published reports cite into its report-media cache, before its build, then export the reports as files (reports export) when nothing is missing (exit 1 when a citation lacks media or an export fails; default id: SITE_ID)", maxPositionals: 1, run: async ({ positionals, env }) => { const siteId = siteIdFrom(positionals, env, "reports prepare"); @@ -168,6 +168,30 @@ export const COMMANDS: Command[] = [ }, }, { + path: ["reports", "export"], + usage: + "<id> [--report <reportId>] [--formats html,pdf,md,zip] [--allow-missing-media] write the site's published reports as files (report.html, report.pdf, report.md, evidence-pack.zip) into its report-exports staging, where compose publishes them from (exit 1 on a problem; a PDF skipped for want of a browser is a note; default id: SITE_ID)", + flags: { report: "string", formats: "string", "allow-missing-media": "boolean" }, + maxPositionals: 1, + run: async ({ positionals, flags, env }) => { + const siteId = siteIdFrom(positionals, env, "reports export"); + if (!siteId) return 2; + const { parseReportExportFormats } = await import("../publish/reportExports"); + const formats = typeof flags.formats === "string" ? parseReportExportFormats(flags.formats) : undefined; + if (formats === null) { + console.error("reports export: --formats is a comma-separated list of html, pdf, md, zip"); + return 2; + } + return (await import("./reports-export")).main({ + siteId, + signal: interrupted(), + ...(typeof flags.report === "string" ? { reportId: flags.report } : {}), + ...(formats ? { formats } : {}), + allowMissingMedia: flags["allow-missing-media"] === true, + }); + }, + }, + { path: ["reports", "convert"], usage: "<sweep|ask|manifest> <in> --out <report.json> [--channels-dir <dir>] [--id <id>] [--title <title>] a /sweep report (markdown), an /ask answer or a report-to-video manifest as a report.json, written only when it validates (--channels-dir: widen spans from the cues, find posts' channels)", diff --git a/common/bin/reports-export.ts b/common/bin/reports-export.ts @@ -0,0 +1,38 @@ +// `archilyzer reports export <siteId> [--report <id>] [--formats html,pdf,md,zip]` +// — write a site's published reports as files (report.html, report.pdf, +// report.md, evidence-pack.zip) into +// `.export-index/sites/<siteId>/report-exports/<reportId>/`, where compose +// publishes them from. The work is publish/reportExports.ts's, the same the +// editor's `reports-export` job runs; this file prints its log and problems. +// +// Exit 0 when every asked-for export was written (a PDF skipped for want of a +// browser is a note, not a problem); 1 when any problem is listed; 2 for bad +// arguments or a site or report that does not exist. + +import { + exportSiteReports, + formatReportExportProblems, + type ExportSiteReportsOptions, +} from "../publish/reportExports"; + +type Out = { log: (s: string) => void; error: (s: string) => void }; + +export async function main( + opts: Omit<ExportSiteReportsOptions, "onLog">, + out: Out = console, +): Promise<number> { + let result; + try { + result = await exportSiteReports({ ...opts, onLog: out.log }); + } catch (err) { + out.error(`reports export: ${(err as Error).message}`); + return opts.signal?.aborted ? 1 : 2; + } + for (const r of result.exported) { + for (const note of r.manifest.notes) out.log(` note: ${r.reportId}: ${note}`); + } + if (result.problems.length === 0) return 0; + out.error(`reports export ${opts.siteId}: ${result.problems.length} problem(s):`); + for (const line of formatReportExportProblems(result.problems)) out.error(` ${line}`); + return 1; +} diff --git a/common/bin/reports-prepare.ts b/common/bin/reports-prepare.ts @@ -4,16 +4,25 @@ // editor's `reports-prepare` job runs; this file prints its log and its // problems. // -// Exit 0 when every cited moment has its media; 1 when any problem is listed -// (the manifest is written either way, problems included); 2 for a site that -// does not exist. +// When nothing is missing it then exports the reports as files +// (publish/reportExports.ts, `archilyzer reports export`). +// +// Exit 0 when every cited moment has its media and every export was written; +// 1 when any problem is listed (the manifest is written either way, problems +// included); 2 for a site that does not exist. import { formatReportMediaProblems, prepareReportMedia } from "../publish/reportMedia"; +import { exportAfterPrepare, formatReportExportProblems, type ExportSiteReportsOptions } from "../publish/reportExports"; type Out = { log: (s: string) => void; error: (s: string) => void }; export async function main( - opts: { siteId: string; signal?: AbortSignal }, + opts: { + siteId: string; + signal?: AbortSignal; + // Passed to the export at the end (tests inject the PDF printer and zip). + exportOptions?: Partial<Pick<ExportSiteReportsOptions, "openPdfPrinter" | "zipBin" | "paths" | "settings">>; + }, out: Out = console, ): Promise<number> { let index; @@ -23,8 +32,21 @@ export async function main( out.error(`reports prepare: ${(err as Error).message}`); return opts.signal?.aborted ? 1 : 2; } - if (index.problems.length === 0) return 0; - out.error(`reports prepare ${opts.siteId}: ${index.problems.length} problem(s):`); - for (const line of formatReportMediaProblems(index.problems)) out.error(` ${line}`); + if (index.problems.length > 0) { + out.error(`reports prepare ${opts.siteId}: ${index.problems.length} problem(s):`); + for (const line of formatReportMediaProblems(index.problems)) out.error(` ${line}`); + out.error("reports prepare: the reports were not exported (the evidence media is not complete)."); + return 1; + } + let exported; + try { + exported = await exportAfterPrepare(index, { siteId: opts.siteId, signal: opts.signal, onLog: out.log, ...opts.exportOptions }); + } catch (err) { + out.error(`reports prepare: export: ${(err as Error).message}`); + return 1; + } + if (!exported || exported.problems.length === 0) return 0; + out.error(`reports prepare ${opts.siteId}: export: ${exported.problems.length} problem(s):`); + for (const line of formatReportExportProblems(exported.problems)) out.error(` ${line}`); return 1; } diff --git a/common/lib/builtExport.ts b/common/lib/builtExport.ts @@ -223,6 +223,21 @@ export const CITED_MEDIA_ALLOWED_DIRS: readonly string[] = ["clips", "posts"]; export const PAGES_MAX_FILE_BYTES = 25 * 1024 * 1024; export const PAGES_MAX_FILES = 20_000; +// What a step that PUTS a file on a site holds it to: 24 MiB, a mebibyte inside +// Pages' own limit (the source mirror's packs, an evidence clip, a report's +// exports). One number, so every step refuses the same file. +export const PUBLISH_MAX_FILE_BYTES = 24 * 1024 * 1024; + +/** + * Why a file of `bytes` may not be published, as one sentence naming `rel` — or + * null when it fits under PUBLISH_MAX_FILE_BYTES. + */ +export function publishFileSizeProblem(rel: string, bytes: number): string | null { + if (bytes <= PUBLISH_MAX_FILE_BYTES) return null; + const mib = (n: number) => (n / (1024 * 1024)).toFixed(1); + return `${rel} is ${mib(bytes)} MiB, over the publish limit of ${mib(PUBLISH_MAX_FILE_BYTES)} MiB (Pages allows 25 MiB per file)`; +} + // corpus.json's `site.scope`, or null. function builtScopeIn(outDir: string): string | null { try { diff --git a/common/lib/evidenceClip-server.ts b/common/lib/evidenceClip-server.ts @@ -46,6 +46,7 @@ import { createReadStream } from "node:fs"; import { mkdir, readFile, rename, rm, stat } from "node:fs/promises"; import path from "node:path"; import { execa } from "execa"; +import { PUBLISH_MAX_FILE_BYTES } from "./builtExport"; import { tmpPathFor, writeJsonAtomic } from "./jsonFile-server"; import { roundMomentSeconds } from "./citations/moments"; import type { CitationPad } from "./citations/schema"; @@ -66,7 +67,7 @@ export const EVIDENCE_CRF = 23; export const EVIDENCE_AUDIO_BITRATE = "128k"; // 24 MiB: a Pages file limit is 25 MiB, and a clip is published as one file. -export const EVIDENCE_MAX_BYTES = 24 * 1024 * 1024; +export const EVIDENCE_MAX_BYTES = PUBLISH_MAX_FILE_BYTES; // A cut of a cached window is seconds of work; a whole saved container on a // platter seeks once. Generous, so only a wedged ffmpeg reaches it. diff --git a/common/lib/report/exportHtml.ts b/common/lib/report/exportHtml.ts @@ -0,0 +1,624 @@ +// A REPORT AS ONE HTML FILE — the export a reader saves and hosts again +// (publish/reportExports.ts writes it as `report.html`, prints it to +// `report.pdf`, and packs a variant of it with its media as the evidence +// pack). Built from the report's page view (./views.ts), the same view the +// export site renders, so the file says what the page says. +// +// ONE FILE, NO NETWORK FOR WHAT IT SHOWS: its own small stylesheet inline, no +// script, no font or image fetched. Every image (a source's sentence as a +// still, a post's screenshot) is whatever the caller's `image` resolver +// answers — a data: URI for the one-file export, a relative path in the pack. +// Clips are LINKED, never inlined (the pack plays them from its media/). +// +// What it holds: the series on its own line in the accent, the title with the +// reviewed document's author and publisher inline after it, what the report +// is and which site publishes it, the dates, the subtitle, a fact-check's verdict tally, the summary, +// the sections and their claims — each its verdict, the document's own +// sentence, the findings with numbered citation markers, the evidence — the +// documents quoted with their archive links, and the numbered reference list: +// each quote with its speaker, date, record, the original at its time, and the +// moment page and clip on the site when the site has a public URL. Then the +// footer naming exactly which document this is (ReportExportFooter). +// +// PURE: no I/O, deterministic for its input. Imports only pure helpers. + +import { citationAnchor, CITE_SCHEME } from "../citations/inline"; +import type { SourceArchive } from "../citations/schema"; +import { + CITATION_KIND_LABELS, + orderedCitations, + reportFullTitle, + reportPagePath, + sourceAnchor, + spanLabel, + verdictTally, + type CitationView, + type ClaimView, + type ReportPageView, + type SourceView, + type SpanCitationView, +} from "./views"; + +// ─── The footer: which document this is ─── + +// What the footer names. `reportSha256` is the sha256 of the report.json the +// export was made from (hex). `revision` and `commit` are the report's +// revision and the commit of the corpus it was published from — filled by the +// revision history (slice RH); until then absent, and an absent part is left +// out of the line, never shown as unknown. +export type ReportExportFooter = { + revision?: number; + // The revision's date (`YYYY-MM-DD` or an ISO date-time). + date?: string; + reportSha256: string; + commit?: string; +}; + +// `Revision N · <date> · report sha256 <first 12> · commit <short>`, the +// unknown parts left out. +export function reportExportFooterLine(f: ReportExportFooter): string { + return [ + f.revision !== undefined ? `Revision ${f.revision}` : null, + dateLabel(f.date) ?? null, + `report sha256 ${f.reportSha256.slice(0, 12)}`, + f.commit ? `commit ${f.commit.slice(0, 12)}` : null, + ] + .filter(Boolean) + .join(" · "); +} + +// ─── Options ─── + +export type ExportClip = { + href: string; + kind: "video" | "audio"; + // Play it in place (the evidence pack, whose media/ holds it); else a link. + play?: boolean; +}; + +export type ReportExportOptions = { + // The site's public URL (site.json `siteUrl`). Absent: no link to the site, + // its moment pages or its clips (a private site has nowhere to link). + siteUrl?: string; + // The site's title: "Fact-check by <site title>" under the title. + siteTitle?: string; + footer: ReportExportFooter; +}; + +export type ReportExportHtmlOptions = ReportExportOptions & { + // A site-root image path the view names (a still, a post's shot) → the + // <img src>: a data: URI, or a path relative to the file. Undefined: the + // image is left out (a source sentence then shows its quote as text). + image: (sitePath: string) => string | undefined; + // A span citation's clip, or undefined for none. + clip?: (c: SpanCitationView) => ExportClip | undefined; +}; + +// ─── Small helpers ─── + +export function escapeHtml(s: string): string { + return s.replace(/&/g, "&amp;").replace(/</g, "&lt;").replace(/>/g, "&gt;").replace(/"/g, "&quot;").replace(/'/g, "&#39;"); +} + +// `2026-10-04T12:00:00Z` → `2026-10-04`; a partial date as given. +export function dateLabel(v: string | undefined): string | undefined { + if (!v) return undefined; + return /^\d{4}-\d{2}-\d{2}T/.test(v) ? v.slice(0, 10) : v; +} + +// A site-root path on the site's public URL, or undefined without one. +export function siteLink(siteUrl: string | undefined, sitePath: string): string | undefined { + if (!siteUrl) return undefined; + return `${siteUrl.replace(/\/+$/, "")}${sitePath.startsWith("/") ? "" : "/"}${sitePath}`; +} + +// A link a reader may follow out of a saved file: http(s) or mailto. A +// fragment stays in the file; a site-root path goes to the site, when known. +function safeHref(href: string, siteUrl: string | undefined): string | undefined { + if (/^(https?:|mailto:)/i.test(href)) return href; + if (href.startsWith("#")) return href; + if (href.startsWith("/")) return siteLink(siteUrl, href); + return undefined; +} + +const plural = (n: number, one: string, many = `${one}s`) => `${n} ${n === 1 ? one : many}`; + +// ─── Markdown, the small subset a report writes ─── + +export type CiteRef = { number?: number; anchor: string }; + +// A citation marker: `[n]` linking to its reference. +function citeMarker(ref: CiteRef | undefined, id: string): string { + const n = ref?.number !== undefined ? String(ref.number) : "?"; + return `<sup class="cite"><a href="#${escapeHtml(ref?.anchor ?? citationAnchor(id))}">[${n}]</a></sup>`; +} + +const PH = "\u0000"; + +// Emphasis over already-escaped text. +function emphasis(escaped: string): string { + return escaped + .replace(/\*\*(?=\S)([\s\S]*?\S)\*\*/g, "<strong>$1</strong>") + .replace(/(^|[^\w])__(?=\S)([\s\S]*?\S)__(?!\w)/g, "$1<strong>$2</strong>") + .replace(/\*(?=\S)([^*]*?\S)\*/g, "<em>$1</em>") + .replace(/(^|[^\w])_(?=\S)([^_]*?\S)_(?!\w)/g, "$1<em>$2</em>"); +} + +// One paragraph's inline markdown: code spans, links (a `cite:` link is the +// label and its number), autolinks, emphasis. Raw HTML is escaped. +function inlineMd(text: string, cite: (id: string) => CiteRef | undefined, siteUrl: string | undefined): string { + const held: string[] = []; + const hold = (html: string) => `${PH}${held.push(html) - 1}${PH}`; + let s = text.replace(/(`+)([^`]|[^`][\s\S]*?[^`])\1(?!`)/g, (_m, _t, code: string) => + hold(`<code>${escapeHtml(code.trim())}</code>`), + ); + s = s.replace(/\[([^\]]*)\]\(\s*([^)\s]+)(?:\s+"[^"]*")?\s*\)/g, (_m, label: string, href: string) => { + const shown = emphasis(escapeHtml(label)); + if (href.startsWith(CITE_SCHEME)) { + const id = href.slice(CITE_SCHEME.length).trim(); + return hold(`${shown}${citeMarker(cite(id), id)}`); + } + const safe = safeHref(href, siteUrl); + return hold(safe ? `<a href="${escapeHtml(safe)}">${shown}</a>` : shown); + }); + s = s.replace(/<(https?:\/\/[^\s<>]+)>/g, (_m, url: string) => hold(`<a href="${escapeHtml(url)}">${escapeHtml(url)}</a>`)); + s = emphasis(escapeHtml(s)) + .replace(/ {2,}\n/g, "<br>") + .replace(/\n/g, " "); + return s.replace(new RegExp(`${PH}(\\d+)${PH}`, "g"), (_m, i: string) => held[Number(i)]); +} + +const LIST_RE = /^ {0,3}([-*+]|\d{1,9}[.)])\s+(.*)$/; +const FENCE_RE = /^ {0,3}(`{3,}|~{3,})/; +const HEADING_RE = /^ {0,3}(#{1,6})\s+(.*?)\s*#*\s*$/; +const HR_RE = /^ {0,3}([-*_])(\s*\1){2,}\s*$/; +const QUOTE_RE = /^ {0,3}>/; + +const isBlank = (l: string) => /^\s*$/.test(l); +const startsBlock = (l: string) => FENCE_RE.test(l) || HEADING_RE.test(l) || HR_RE.test(l) || QUOTE_RE.test(l) || LIST_RE.test(l); + +// A report's markdown (a summary, a section's body, a claim's findings) as +// HTML: paragraphs, headings (shifted under the page's own), lists, quotes, +// code, rules. `headingBase` is the level a `#` becomes. +export function markdownToHtml( + md: string, + cite: (id: string) => CiteRef | undefined, + opts: { siteUrl?: string; headingBase?: number } = {}, +): string { + const lines = md.replace(/\r\n?/g, "\n").split("\n"); + const base = opts.headingBase ?? 3; + const inline = (t: string) => inlineMd(t, cite, opts.siteUrl); + const out: string[] = []; + let i = 0; + while (i < lines.length) { + const line = lines[i]; + if (isBlank(line)) { + i++; + continue; + } + const fence = FENCE_RE.exec(line); + if (fence) { + const close = new RegExp(`^ {0,3}${fence[1][0] === "`" ? "`" : "~"}{${fence[1].length},}\\s*$`); + const body: string[] = []; + i++; + while (i < lines.length && !close.test(lines[i])) body.push(lines[i++]); + i++; + out.push(`<pre><code>${escapeHtml(body.join("\n"))}</code></pre>`); + continue; + } + const h = HEADING_RE.exec(line); + if (h) { + const level = Math.min(6, base - 1 + h[1].length); + out.push(`<h${level}>${inline(h[2])}</h${level}>`); + i++; + continue; + } + if (HR_RE.test(line)) { + out.push("<hr>"); + i++; + continue; + } + if (QUOTE_RE.test(line)) { + const inner: string[] = []; + while (i < lines.length && QUOTE_RE.test(lines[i])) inner.push(lines[i++].replace(/^ {0,3}> ?/, "")); + out.push(`<blockquote>${markdownToHtml(inner.join("\n"), cite, opts)}</blockquote>`); + continue; + } + const first = LIST_RE.exec(line); + if (first) { + const ordered = /\d/.test(first[1]); + const items: string[] = []; + while (i < lines.length) { + const m = LIST_RE.exec(lines[i]); + if (m && /\d/.test(m[1]) === ordered) { + items.push(m[2]); + i++; + } else if (items.length > 0 && !isBlank(lines[i]) && /^\s{2,}\S/.test(lines[i])) { + items[items.length - 1] += `\n${lines[i++].trim()}`; + } else break; + } + const start = ordered ? Number.parseInt(first[1], 10) : 1; + const tag = ordered ? "ol" : "ul"; + const attr = ordered && start !== 1 ? ` start="${start}"` : ""; + out.push(`<${tag}${attr}>${items.map((it) => `<li>${inline(it)}</li>`).join("")}</${tag}>`); + continue; + } + const para: string[] = [line]; + i++; + while (i < lines.length && !isBlank(lines[i]) && !startsBlock(lines[i])) para.push(lines[i++]); + out.push(`<p>${inline(para.join("\n").trim())}</p>`); + } + return out.join("\n"); +} + +// ─── The stylesheet ─── + +// Small and self-contained: system fonts, a light page that prints as it +// reads. The verdict chip takes its colour from `--v` (the view's verdict +// styles, report overrides applied), as the site's VerdictChip does. +export const REPORT_EXPORT_CSS = ` +:root{--fg:#1b1b1b;--muted:#5d5d5d;--border:#d8d8d8;--surface:#f6f6f4;--accent:#1f4e8c;color-scheme:light} +*{box-sizing:border-box} +html{-webkit-text-size-adjust:100%} +body{margin:0;background:#fff;color:var(--fg);font:16px/1.55 -apple-system,BlinkMacSystemFont,"Segoe UI",Roboto,"Helvetica Neue",Arial,sans-serif} +main{max-width:46rem;margin:0 auto;padding:2rem 1rem 3rem} +a{color:var(--accent);text-decoration:underline;text-decoration-thickness:1px;text-underline-offset:2px;overflow-wrap:anywhere} +h1,h2,h3,h4{line-height:1.25;margin:0} +h1{font-size:2rem;font-weight:600;letter-spacing:-.01em} +h1 .series{display:block;color:var(--accent);font-weight:600;font-size:.8em;margin-bottom:.15rem} +h1 .series+.title{display:block;font-weight:400} +h2{font-size:1.5rem;font-weight:600;margin-top:2.5rem;padding-top:1rem;border-top:1px solid var(--border)} +h3{font-size:1.15rem;font-weight:600} +h4,h5,h6{font-size:1rem;font-weight:600;margin:1rem 0 .25rem} +p{margin:.6rem 0} +q{quotes:"\\201C" "\\201D"} +blockquote{margin:.75rem 0;padding-left:.9rem;border-left:3px solid var(--border);color:var(--muted)} +code{font-family:ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;font-size:.88em;background:var(--surface);padding:.05em .3em;border-radius:3px} +pre{background:var(--surface);padding:.75rem;border-radius:4px;overflow-x:auto} +pre code{background:none;padding:0} +img{max-width:100%;height:auto} +video,audio{display:block;width:100%;max-width:36rem;margin:.5rem 0} +.subtitle{font-size:1.15rem;color:var(--muted);margin:.5rem 0 0} +.site-line,.dates{margin:.35rem 0 0} +h1 .by{font-size:.55em;font-weight:400;color:var(--muted);letter-spacing:0} +.site-line,.dates,.label,.meta,.export-footer{font-size:.85rem;color:var(--muted)} +.label{font:600 .7rem/1.2 ui-monospace,SFMono-Regular,Menlo,monospace;text-transform:uppercase;letter-spacing:.14em;margin:1.25rem 0 .4rem} +header.report-head{padding-bottom:1.25rem;border-bottom:1px solid var(--border)} +.under-review{margin-top:1rem;padding:.75rem 1rem;border:1px solid var(--border);border-radius:6px;background:var(--surface);font-size:.95rem} +.tally{display:flex;flex-wrap:wrap;gap:.4rem;list-style:none;padding:0;margin:.4rem 0 0} +.verdict{display:inline-flex;align-items:center;gap:.4rem;border:1px solid var(--v);border-radius:999px;padding:.1rem .65rem;font-size:.8rem;font-weight:600;background:color-mix(in srgb,var(--v) 14%,#fff);white-space:nowrap} +.verdict::before{content:"";width:.5rem;height:.5rem;border-radius:50%;background:var(--v)} +.verdict .n{font-family:ui-monospace,SFMono-Regular,Menlo,monospace;color:var(--muted);font-weight:400} +nav.toc ol{margin:.5rem 0;padding-left:1.4rem} +.claim{margin:1.25rem 0;padding:1rem 1.1rem;border:1px solid var(--border);border-radius:8px} +.claim-head{display:flex;flex-wrap:wrap;align-items:baseline;gap:.4rem .7rem} +.claim-head h3{flex:1 1 15rem} +.says{color:var(--muted);font-size:.95rem} +.says q{color:var(--fg)} +figure.sentence{margin:.75rem 0} +figure.sentence img{display:block;border:1px solid var(--border);border-radius:4px;background:#fff;break-inside:avoid} +figure.sentence figcaption{font-size:.85rem;color:var(--muted);margin-top:.3rem} +.evidence{margin:.4rem 0 0;padding-left:1.2rem;font-size:.92rem} +.evidence li{margin:.2rem 0} +sup.cite{font-size:.72em;line-height:0;margin-left:.1em} +sup.cite a{text-decoration:none;font-weight:600} +.archives{margin:.3rem 0 0;padding-left:1.2rem;font-size:.85rem} +.archives li{margin:.15rem 0} +.archives .ctx{color:var(--muted)} +.source{margin:1rem 0} +ol.references{padding-left:2.2rem} +ol.references>li{margin:0 0 1rem;padding-left:.2rem} +ol.references .quote{margin:0;font-size:1rem} +ol.references .meta,ol.references .links{margin:.15rem 0;font-size:.85rem} +ol.references img{display:block;max-width:24rem;border:1px solid var(--border);border-radius:4px;margin:.4rem 0;break-inside:avoid} +:target{background:#fff6d6} +.export-footer{margin-top:3rem;padding-top:1rem;border-top:1px solid var(--border)} +.export-footer p{margin:.25rem 0} +@media print{ + @page{size:A4;margin:16mm 14mm} + body{font-size:11pt} + main{max-width:none;padding:0} + a{color:var(--fg)} + h2{break-after:avoid} + .claim{break-inside:auto} + :target{background:none} +} +`.trim(); + +// ─── The parts ─── + +function verdictChip(view: ReportPageView, verdict: NonNullable<ClaimView["verdict"]>, count?: number): string { + const s = view.verdicts[verdict]; + const n = count !== undefined ? ` <span class="n">${count}</span>` : ""; + return `<span class="verdict" data-verdict="${escapeHtml(verdict)}" style="--v:${escapeHtml(s.color)}">${escapeHtml(s.label)}${n}</span>`; +} + +function archiveList(archives: readonly SourceArchive[]): string { + if (archives.length === 0) return ""; + return `<ul class="archives">${archives + .map( + (a) => + `<li><a href="${escapeHtml(a.url)}">${escapeHtml(a.label)}</a> <span class="ctx">${escapeHtml(a.url)}</span>` + + `${a.context ? ` <span class="ctx">— ${escapeHtml(a.context)}</span>` : ""}</li>`, + ) + .join("")}</ul>`; +} + +function sourceByline(s: SourceView): string { + return [s.publisher, s.author, dateLabel(s.date)].filter(Boolean).map((x) => escapeHtml(x!)).join(" · "); +} + +function sourceTitleHtml(s: SourceView): string { + return s.url ? `<a href="${escapeHtml(s.url)}">${escapeHtml(s.title)}</a>` : escapeHtml(s.title); +} + +// The citation's record or document, as one line: the title, then where. +function citationSourceLine(c: CitationView): string { + switch (c.kind) { + case "video": + case "audio": { + const r = c.record; + return [r.title ? `<cite>${escapeHtml(r.title)}</cite>` : null, r.channelTitle ? escapeHtml(r.channelTitle) : null, spanLabel(c.start, c.end)] + .filter(Boolean) + .join(" · "); + } + case "post": + return [c.author ? escapeHtml(c.author) : null, c.record.channelTitle ? escapeHtml(c.record.channelTitle) : null, "post"] + .filter(Boolean) + .join(" · "); + case "source": + return `from <a href="#${escapeHtml(sourceAnchor(c.sourceId))}"><cite>${escapeHtml(c.sourceTitle)}</cite></a>`; + case "page": + return c.title ? `<cite>${escapeHtml(c.title)}</cite>` : "web page"; + } +} + +function citationDate(c: CitationView): string | undefined { + return dateLabel(c.date ?? (c.kind === "video" || c.kind === "audio" || c.kind === "post" ? c.record.date : undefined)); +} + +function link(href: string, text?: string): string { + return `<a href="${escapeHtml(href)}">${escapeHtml(text ?? href)}</a>`; +} + +function reference(c: CitationView, opts: ReportExportHtmlOptions): string { + const meta = [ + `<span class="kind">${CITATION_KIND_LABELS[c.kind]}</span>`, + c.speaker ? escapeHtml(c.speaker) : null, + citationDate(c) ? escapeHtml(citationDate(c)!) : null, + citationSourceLine(c), + c.verification?.quoteScore !== undefined ? `quote match ${Math.round(c.verification.quoteScore * 100)}%` : null, + ].filter(Boolean); + const links: string[] = []; + let media = ""; + switch (c.kind) { + case "video": + case "audio": { + if (c.record.originalUrl) links.push(`Original at ${spanLabel(c.start, c.end).split("–")[0]}: ${link(c.record.originalUrl)}`); + const moment = siteLink(opts.siteUrl, c.href); + if (moment) links.push(`Moment page: ${link(moment)}`); + const clip = opts.clip?.(c); + if (clip) { + links.push(`Clip: ${link(clip.href)}`); + if (clip.play) { + media = + clip.kind === "audio" + ? `<audio controls preload="none" src="${escapeHtml(clip.href)}"></audio>` + : `<video controls preload="none" src="${escapeHtml(clip.href)}"></video>`; + } + } + break; + } + case "post": { + if (c.record.originalUrl) links.push(`Original: ${link(c.record.originalUrl)}`); + const moment = siteLink(opts.siteUrl, c.href); + if (moment) links.push(`Moment page: ${link(moment)}`); + const shot = c.shot ? opts.image(c.shot) : undefined; + if (shot) media = `<img src="${escapeHtml(shot)}" alt="${escapeHtml(`Screenshot of the post: ${c.quote}`)}">`; + break; + } + case "source": + if (c.href) links.push(`Original: ${link(c.href)}`); + break; + case "page": + links.push(`Original: ${link(c.href)}`); + if (c.archiveUrl) links.push(`Archived: ${link(c.archiveUrl)}`); + break; + } + const extra = [c.label, c.note].filter(Boolean).map((x) => `<p class="meta">${escapeHtml(x!)}</p>`).join(""); + return ( + `<li id="${escapeHtml(citationAnchor(c.id))}" value="${c.number ?? ""}" data-reference="${escapeHtml(c.id)}">` + + `<p class="quote"><q>${escapeHtml(c.quote)}</q></p>` + + `<p class="meta">${meta.join(" · ")}</p>` + + extra + + (links.length > 0 ? `<p class="links">${links.join("<br>")}</p>` : "") + + media + + `</li>` + ); +} + +function claimHtml( + claim: ClaimView, + view: ReportPageView, + subjectLabel: string, + cite: (id: string) => CiteRef | undefined, + opts: ReportExportHtmlOptions, +): string { + const parts: string[] = []; + parts.push( + `<div class="claim-head">${claim.verdict ? verdictChip(view, claim.verdict) : ""}` + + `<h3>${escapeHtml(claim.title ?? claim.text)}</h3></div>`, + ); + if (claim.title) parts.push(`<p class="says">${escapeHtml(subjectLabel)}: <q>${escapeHtml(claim.text)}</q></p>`); + const sentence = claim.sourceQuote ? view.citations[claim.sourceQuote] : undefined; + if (sentence?.kind === "source") { + const src = sentence.image ? opts.image(sentence.image) : undefined; + const shown = src + ? `<img src="${escapeHtml(src)}" alt="${escapeHtml(sentence.quote)}">` + : `<blockquote><q>${escapeHtml(sentence.quote)}</q></blockquote>`; + parts.push( + `<figure class="sentence" data-source-sentence="${escapeHtml(sentence.id)}">${shown}` + + `<figcaption>from <a href="#${escapeHtml(sourceAnchor(sentence.sourceId))}">${escapeHtml(sentence.sourceTitle)}</a>` + + `${citeMarker(cite(sentence.id), sentence.id)}${claim.archives ? archiveList(claim.archives) : ""}</figcaption></figure>`, + ); + } + if (claim.findings) parts.push(`<div class="findings">${markdownToHtml(claim.findings, cite, { siteUrl: opts.siteUrl, headingBase: 4 })}</div>`); + const evidence = claim.citations + .filter((id) => id !== claim.sourceQuote) + .map((id) => view.citations[id]) + .filter((c): c is CitationView => !!c); + if (evidence.length > 0) { + parts.push( + `<p class="label">Evidence</p><ul class="evidence">${evidence + .map( + (c) => + `<li><a href="#${escapeHtml(citationAnchor(c.id))}">[${c.number ?? "?"}]</a> ` + + `${CITATION_KIND_LABELS[c.kind]} · ${citationSourceLine(c)} — <q>${escapeHtml(c.quote)}</q></li>`, + ) + .join("")}</ul>`, + ); + } + return `<article class="claim" id="${escapeHtml(claim.id)}" data-claim="${escapeHtml(claim.id)}">${parts.join("\n")}</article>`; +} + +// ─── The document ─── + +// "Fact-check by <site title>" ("Report by …"), or the kind alone with no +// site title: our line under the title. +export function reportKindLine(view: Pick<ReportPageView, "kind">, siteTitle?: string): string { + const kind = view.kind === "factcheck" ? "Fact-check" : "Report"; + return siteTitle ? `${kind} by ${siteTitle}` : kind; +} + +// `Published <date>`, `Updated <date>` (when it differs). +export function reportDates(view: Pick<ReportPageView, "published" | "updated">): string[] { + return [ + view.published ? `Published ${dateLabel(view.published)}` : null, + view.updated && view.updated !== view.published ? `Updated ${dateLabel(view.updated)}` : null, + ].filter((x): x is string => !!x); +} + +// The byline after the title: the author and publisher of the document under +// review (the subject source), the publisher linked to the document — or, with +// no publisher, the author. Neither: none. +export function subjectByline(view: Pick<ReportPageView, "subject" | "sources">): { author?: string; publisher?: string; url?: string } | undefined { + const s = view.subject ? view.sources[view.subject] : undefined; + if (!s || (!s.author && !s.publisher)) return undefined; + return { + ...(s.author ? { author: s.author } : {}), + ...(s.publisher ? { publisher: s.publisher } : {}), + ...(s.url ? { url: s.url } : {}), + }; +} + +export function reportExportHtml(view: ReportPageView, opts: ReportExportHtmlOptions): string { + const isFactcheck = view.kind === "factcheck"; + const cite = (id: string): CiteRef | undefined => { + const c = view.citations[id]; + return c ? { number: c.number, anchor: citationAnchor(id) } : undefined; + }; + const md = (text: string, headingBase = 3) => markdownToHtml(text, cite, { siteUrl: opts.siteUrl, headingBase }); + const subject = view.subject ? view.sources[view.subject] : undefined; + const subjectLabel = subject?.kind === "article" ? "The article says" : "The source says"; + const tally = isFactcheck ? verdictTally(view) : []; + const claimCount = view.sections.reduce((n, s) => n + s.claims.length, 0); + const pageUrl = siteLink(opts.siteUrl, reportPagePath(view.id)); + + // The heading, in the site's order: the series (its own line, the accent), + // the title with the reviewed document's byline inline, then ours — what + // the report is and who publishes it, its dates — then the subtitle. + const head: string[] = []; + const by = subjectByline(view); + const byHtml = by + ? ` <span class="by" data-report-byline="">by ${[ + by.author ? (by.url && !by.publisher ? link(by.url, by.author) : escapeHtml(by.author)) : null, + by.publisher ? (by.url ? link(by.url, by.publisher) : escapeHtml(by.publisher)) : null, + ] + .filter(Boolean) + .join(" · ")}</span>` + : ""; + head.push( + `<h1>${view.series ? `<span class="series" data-report-series="">${escapeHtml(view.series)}</span>` : ""}` + + `<span class="title">${escapeHtml(view.title)}${byHtml}</span></h1>`, + ); + head.push(`<p class="site-line" data-report-kind="">${escapeHtml(reportKindLine(view, opts.siteTitle))}</p>`); + const dates = reportDates(view); + if (dates.length > 0) head.push(`<p class="dates">${dates.map(escapeHtml).join(" · ")}</p>`); + if (view.subtitle) head.push(`<p class="subtitle">${escapeHtml(view.subtitle)}</p>`); + if (subject) { + const by = sourceByline(subject); + head.push( + `<div class="under-review"><p class="label" style="margin-top:0">Under review</p>` + + `<p style="margin:0">${sourceTitleHtml(subject)}</p>${by ? `<p class="meta" style="margin:.2rem 0 0">${by}</p>` : ""}` + + `${subject.archives.length > 0 ? `<p class="meta" style="margin:.2rem 0 0"><a href="#${escapeHtml(sourceAnchor(subject.id))}">${plural(subject.archives.length, "archive link")}</a></p>` : ""}</div>`, + ); + } + if (tally.length > 0) { + head.push( + `<p class="label">${plural(claimCount, "claim")} checked</p>` + + `<ul class="tally" data-verdict-tally="">${tally.map((t) => `<li>${verdictChip(view, t.verdict, t.count)}</li>`).join("")}</ul>`, + ); + } + + const body: string[] = [`<header class="report-head">${head.join("\n")}</header>`]; + if (view.summary) body.push(`<section class="summary">${md(view.summary)}</section>`); + if (view.sections.length > 1) { + body.push( + `<nav class="toc" aria-label="Sections"><ol>${view.sections + .map( + (s) => + `<li><a href="#${escapeHtml(s.id)}">${escapeHtml(s.title)}</a>` + + `${s.claims.length > 0 ? ` <span class="meta">${plural(s.claims.length, "claim")}</span>` : ""}</li>`, + ) + .join("")}</ol></nav>`, + ); + } + for (const s of view.sections) { + body.push( + `<section id="${escapeHtml(s.id)}" data-section="${escapeHtml(s.id)}"><h2>${escapeHtml(s.title)}</h2>` + + `${s.body ? md(s.body) : ""}${s.claims.map((c) => claimHtml(c, view, subjectLabel, cite, opts)).join("\n")}</section>`, + ); + } + // Every document the report quotes, the subject first, each once with its + // archive links in context. + const sources = Object.values(view.sources); + if (sources.length > 0) { + body.push( + `<section id="sources"><h2>Sources</h2>${sources + .map((s) => { + const by = sourceByline(s); + return ( + `<div class="source" id="${escapeHtml(sourceAnchor(s.id))}"><p style="margin:0"><strong>${sourceTitleHtml(s)}</strong>` + + `${s.id === view.subject ? ` <span class="meta">(under review)</span>` : ""}</p>` + + `${s.url ? `<p class="meta" style="margin:.1rem 0">${escapeHtml(s.url)}</p>` : ""}` + + `${by ? `<p class="meta" style="margin:.1rem 0">${by}</p>` : ""}` + + `${s.note ? `<p class="meta" style="margin:.1rem 0">${escapeHtml(s.note)}</p>` : ""}` + + `${s.archives.length > 0 ? `<p class="label">${plural(s.archives.length, "archive link")} in context</p>${archiveList(s.archives)}` : ""}</div>` + ); + }) + .join("\n")}</section>`, + ); + } + const refs = orderedCitations(view); + if (refs.length > 0) { + body.push( + `<section id="references"><h2>References</h2><ol class="references" data-reference-list="">${refs + .map((c) => reference(c, opts)) + .join("\n")}</ol></section>`, + ); + } + body.push( + `<footer class="export-footer" data-export-footer=""><p>${escapeHtml(reportExportFooterLine(opts.footer))}</p>` + + `${pageUrl ? `<p>Published at ${link(pageUrl)}</p>` : ""}</footer>`, + ); + + return ( + `<!doctype html>\n<html lang="en">\n<head>\n<meta charset="utf-8">\n` + + `<meta name="viewport" content="width=device-width, initial-scale=1">\n` + + `<title>${escapeHtml(reportFullTitle(view))}</title>\n` + + `${view.subtitle ? `<meta name="description" content="${escapeHtml(view.subtitle)}">\n` : ""}` + + `<meta name="generator" content="Archilyzer">\n` + + `<style>\n${REPORT_EXPORT_CSS}\n</style>\n</head>\n<body>\n<main data-report="${escapeHtml(view.id)}">\n` + + `${body.join("\n")}\n</main>\n</body>\n</html>\n` + ); +} diff --git a/common/lib/report/exportMarkdown.ts b/common/lib/report/exportMarkdown.ts @@ -0,0 +1,214 @@ +// A REPORT AS PLAIN MARKDOWN — `report.md`, the export that reads anywhere +// text does (publish/reportExports.ts writes it). Built from the report's page +// view (./views.ts), like the HTML export (./exportHtml.ts), with the same +// footer. +// +// The report's own markdown (summary, section bodies, findings) passes through +// as written, each inline citation `[label](cite:<id>)` becoming `label [n]` — +// the number of its entry in the numbered reference list at the end. Plain +// text fields (titles, quotes) are escaped so a `*` or `[` in a quote stays a +// character. No images: a still or a screenshot is in the HTML export and the +// evidence pack; here the quote is the text. +// +// PURE: no I/O, deterministic for its input. + +import { CITE_SCHEME } from "../citations/inline"; +import type { SourceArchive } from "../citations/schema"; +import { + dateLabel, + reportDates, + reportExportFooterLine, + reportKindLine, + siteLink, + subjectByline, + type ReportExportOptions, +} from "./exportHtml"; +import { + CITATION_KIND_LABELS, + orderedCitations, + reportPagePath, + spanLabel, + verdictTally, + type CitationView, + type ClaimView, + type ReportPageView, +} from "./views"; + +// Plain text, safe in a Markdown paragraph: the characters Markdown reads as +// syntax are escaped, and a line can never open a heading, a list or a quote. +export function mdText(s: string): string { + return s + .replace(/[\\`*_[\]<>|]/g, (c) => `\\${c}`) + .replace(/\s*\n\s*/g, " ") + .replace(/^([#>+-]|\d+[.)])/, "\\$1"); +} + +// A URL as a Markdown link target: spaces and parentheses escaped. +function mdUrl(u: string): string { + return u.replace(/ /g, "%20").replace(/\(/g, "%28").replace(/\)/g, "%29"); +} + +const mdLink = (text: string, url: string) => `[${mdText(text)}](${mdUrl(url)})`; + +// The report's markdown with each inline citation as `label [n]`. A site-root +// link goes to the site when its URL is known, else stays its label. +export function citedMarkdownToPlain(md: string, view: Pick<ReportPageView, "citations">, siteUrl?: string): string { + return md.replace(/\[([^\]]*)\]\(\s*([^)\s]+)(\s+"[^"]*")?\s*\)/g, (m, label: string, href: string, title?: string) => { + if (href.startsWith(CITE_SCHEME)) { + const c = view.citations[href.slice(CITE_SCHEME.length).trim()]; + return `${label} [${c?.number ?? "?"}]`; + } + if (href.startsWith("/")) { + const abs = siteLink(siteUrl, href); + return abs ? `[${label}](${mdUrl(abs)}${title ?? ""})` : label; + } + return m; + }); +} + +function archiveLines(archives: readonly SourceArchive[]): string[] { + return archives.map((a) => `- ${mdLink(a.label, a.url)} <${a.url}>${a.context ? ` — ${mdText(a.context)}` : ""}`); +} + +function citationWhere(c: CitationView): string { + switch (c.kind) { + case "video": + case "audio": + return [c.record.title ? `*${mdText(c.record.title)}*` : null, c.record.channelTitle ? mdText(c.record.channelTitle) : null, spanLabel(c.start, c.end)] + .filter(Boolean) + .join(", "); + case "post": + return [c.author ? mdText(c.author) : null, c.record.channelTitle ? mdText(c.record.channelTitle) : null, "post"].filter(Boolean).join(", "); + case "source": + return `from *${mdText(c.sourceTitle)}*`; + case "page": + return c.title ? `*${mdText(c.title)}*` : "web page"; + } +} + +function citationDate(c: CitationView): string | undefined { + return dateLabel(c.date ?? (c.kind === "video" || c.kind === "audio" || c.kind === "post" ? c.record.date : undefined)); +} + +function referenceLines(c: CitationView, siteUrl: string | undefined): string[] { + const meta = [ + CITATION_KIND_LABELS[c.kind], + c.speaker ? mdText(c.speaker) : null, + citationDate(c) ?? null, + citationWhere(c), + c.verification?.quoteScore !== undefined ? `quote match ${Math.round(c.verification.quoteScore * 100)}%` : null, + ].filter(Boolean); + const out = [`${c.number ?? "-"}. “${mdText(c.quote)}”`, ` ${meta.join(" · ")}`]; + for (const x of [c.label, c.note]) if (x) out.push(` ${mdText(x)}`); + const link = (label: string, url: string) => out.push(` ${label}: <${url}>`); + switch (c.kind) { + case "video": + case "audio": { + if (c.record.originalUrl) link(`Original at ${spanLabel(c.start, c.end).split("–")[0]}`, c.record.originalUrl); + const moment = siteLink(siteUrl, c.href); + if (moment) link("Moment page", moment); + break; + } + case "post": { + if (c.record.originalUrl) link("Original", c.record.originalUrl); + const moment = siteLink(siteUrl, c.href); + if (moment) link("Moment page", moment); + break; + } + case "source": + if (c.href) link("Original", c.href); + break; + case "page": + link("Original", c.href); + if (c.archiveUrl) link("Archived", c.archiveUrl); + break; + } + // One list item: its lines indented under the number, each ending in a + // hard break but the last. + return out.map((l, i, a) => (i < a.length - 1 ? `${l} ` : l)); +} + +function claimLines(claim: ClaimView, view: ReportPageView, subjectLabel: string, siteUrl: string | undefined): string[] { + const out: string[] = [`### ${mdText(claim.title ?? claim.text)}`, ""]; + if (claim.verdict) out.push(`**Verdict: ${mdText(view.verdicts[claim.verdict].label)}**`, ""); + if (claim.title) out.push(`${subjectLabel}: “${mdText(claim.text)}”`, ""); + const sentence = claim.sourceQuote ? view.citations[claim.sourceQuote] : undefined; + if (sentence?.kind === "source") { + out.push(`> “${mdText(sentence.quote)}” [${sentence.number ?? "?"}]`, ">", `> — from *${mdText(sentence.sourceTitle)}*`, ""); + if (claim.archives?.length) out.push(...archiveLines(claim.archives), ""); + } + if (claim.findings) out.push(citedMarkdownToPlain(claim.findings, view, siteUrl).trim(), ""); + const evidence = claim.citations + .filter((id) => id !== claim.sourceQuote) + .map((id) => view.citations[id]) + .filter((c): c is CitationView => !!c); + if (evidence.length > 0) { + out.push("Evidence:", ""); + for (const c of evidence) out.push(`- [${c.number ?? "?"}] ${CITATION_KIND_LABELS[c.kind]}, ${citationWhere(c)}: “${mdText(c.quote)}”`); + out.push(""); + } + return out; +} + +export function reportExportMarkdown(view: ReportPageView, opts: ReportExportOptions): string { + const out: string[] = []; + const subject = view.subject ? view.sources[view.subject] : undefined; + const subjectLabel = subject?.kind === "article" ? "The article says" : "The source says"; + const pageUrl = siteLink(opts.siteUrl, reportPagePath(view.id)); + const isFactcheck = view.kind === "factcheck"; + + // The series on its own line, the title, the reviewed document's byline; + // then ours: what the report is, who publishes it, its dates. + if (view.series) out.push(`**${mdText(view.series)}**`, ""); + out.push(`# ${mdText(view.title)}`, ""); + const by = subjectByline(view); + if (by) { + const author = by.author ? (by.url && !by.publisher ? mdLink(by.author, by.url) : mdText(by.author)) : null; + const publisher = by.publisher ? (by.url ? mdLink(by.publisher, by.url) : mdText(by.publisher)) : null; + out.push(`by ${[author, publisher].filter(Boolean).join(" · ")}`, ""); + } + out.push([mdText(reportKindLine(view, opts.siteTitle)), ...reportDates(view)].join(" · "), ""); + if (view.subtitle) out.push(`*${mdText(view.subtitle)}*`, ""); + if (subject) { + const byline = [subject.publisher, subject.author, dateLabel(subject.date)].filter(Boolean).map((x) => mdText(x!)); + out.push(`Under review: ${subject.url ? mdLink(subject.title, subject.url) : mdText(subject.title)}${byline.length ? ` — ${byline.join(" · ")}` : ""}`, ""); + } + const tally = isFactcheck ? verdictTally(view) : []; + if (tally.length > 0) { + const claims = view.sections.reduce((n, s) => n + s.claims.length, 0); + out.push( + `${claims} claim${claims === 1 ? "" : "s"} checked: ${tally.map((t) => `${mdText(view.verdicts[t.verdict].label)} ${t.count}`).join(" · ")}`, + "", + ); + } + if (view.summary) out.push(citedMarkdownToPlain(view.summary, view, opts.siteUrl).trim(), ""); + + for (const s of view.sections) { + out.push(`## ${mdText(s.title)}`, ""); + if (s.body) out.push(citedMarkdownToPlain(s.body, view, opts.siteUrl).trim(), ""); + for (const c of s.claims) out.push(...claimLines(c, view, subjectLabel, opts.siteUrl)); + } + + const sources = Object.values(view.sources); + if (sources.length > 0) { + out.push("## Sources", ""); + for (const s of sources) { + out.push(`### ${mdText(s.title)}${s.id === view.subject ? " (under review)" : ""}`, ""); + if (s.url) out.push(`<${s.url}>`, ""); + const byline = [s.publisher, s.author, dateLabel(s.date)].filter(Boolean).map((x) => mdText(x!)); + if (byline.length > 0) out.push(byline.join(" · "), ""); + if (s.note) out.push(mdText(s.note), ""); + if (s.archives.length > 0) out.push(`${s.archives.length} archive link${s.archives.length === 1 ? "" : "s"} in context:`, "", ...archiveLines(s.archives), ""); + } + } + + const refs = orderedCitations(view); + if (refs.length > 0) { + out.push("## References", ""); + for (const c of refs) out.push(...referenceLines(c, opts.siteUrl), ""); + } + + out.push("---", "", reportExportFooterLine(opts.footer)); + if (pageUrl) out.push("", `Published at <${pageUrl}>`); + return `${out.join("\n").replace(/\n{3,}/g, "\n\n").trimEnd()}\n`; +} diff --git a/common/lib/report/views.ts b/common/lib/report/views.ts @@ -87,6 +87,22 @@ export function reportCitationsDownloadPath(reportId: string, format: "json" | " return `/reports/${reportId}/citations.${format}`; } +// A report's exports (publish/reportExports.ts): the whole report as files a +// reader can save and host again. Their names, and where compose publishes +// them beside the report's page. +export const REPORT_EXPORT_FORMATS = ["html", "pdf", "md", "zip"] as const; +export type ReportExportFormat = (typeof REPORT_EXPORT_FORMATS)[number]; +export const REPORT_EXPORT_FILENAMES: Readonly<Record<ReportExportFormat, string>> = { + html: "report.html", + pdf: "report.pdf", + md: "report.md", + zip: "evidence-pack.zip", +}; + +export function reportExportDownloadPath(reportId: string, format: ReportExportFormat): string { + return `/reports/${reportId}/${REPORT_EXPORT_FILENAMES[format]}`; +} + // A file the report names relative to its own directory (a still, // `stills/a01.png` — validated by lib/report/validate.ts never to leave it), // as published. @@ -245,10 +261,16 @@ export type ReportPageView = { // is left out. citations: Record<string, CitationView>; sections: SectionView[]; - // The report's citations as files, when compose wrote them. - downloads?: { json?: string; csv?: string }; + // The report as files, and its citations as data, when compose published + // them (each a site-root path). + downloads?: ReportDownloads; }; +// What a report page offers to download: its exports (html, pdf, md, the +// evidence pack as zip) and its citations (json, csv). Each key is present +// only when the file is published. +export type ReportDownloads = Partial<Record<ReportExportFormat | "json" | "csv", string>>; + export type VerdictCount = { verdict: Verdict; count: number }; export type ReportIndexEntry = { @@ -346,7 +368,7 @@ export type ReportViewResolver = { record: (c: SpanCitation | PostCitation) => RecordView; poster?: (c: SpanCitation) => string | undefined; post?: (c: PostCitation) => { author?: string; text?: string; shot?: string } | undefined; - downloads?: { json?: string; csv?: string }; + downloads?: ReportDownloads; }; function sourceView(id: string, s: Source): SourceView { diff --git a/common/publish/composeReports.ts b/common/publish/composeReports.ts @@ -33,6 +33,11 @@ // an `archilyzer-citations` set; // never a source's `saved` copy) // reports/<id>/<still> each cited source still +// reports/<id>/report.{html,pdf,md}, the report's exports, when +// evidence-pack.zip `archilyzer reports export` made +// them from the report as it is now +// and each is within the publish +// limit (./reportExportFiles.ts) // m/index.json, m/<key>/moment.json one moment view per cited moment, // "cited in" across every report // media/clips/<channel>/<id>/<s>-<e>.mp4 a span's prepared clip (.m4a for @@ -86,8 +91,10 @@ import { momentViewPath, orderedCitations, reportCitationsDownloadPath, + reportExportDownloadPath, reportIndexEntry, reportViewPath, + REPORT_EXPORT_FORMATS, type CitationView, type CueLineView, type MomentPageView, @@ -95,6 +102,7 @@ import { type RecordView, type ReportIndexEntry, type ReportIndexView, + type ReportDownloads, type ReportPageView, } from "../lib/report/views"; import { evidenceSpan, isAudioOnlyPlatform, type EvidenceSpan } from "../lib/evidenceClip-server"; @@ -106,6 +114,7 @@ import { siteReportDir, type ReportMediaEntry, } from "./reportMedia"; +import { publishableReportExports, type PublishableReportExports } from "./reportExportFiles"; // The public dir's entries this stage owns. Every compose removes them first. export const REPORT_PUBLIC_ENTRIES: readonly string[] = ["reports", "m", "media"]; @@ -382,25 +391,35 @@ const isFile = async (p: string) => (await stat(p).catch(() => null))?.isFile() const sameSpan = (a: EvidenceSpan, b: EvidenceSpan) => Math.abs(a.from - b.from) < 0.001 && Math.abs(a.to - b.to) < 0.001; -// ─── The stage ─── +// ─── Resolving: the reports as views, nothing written ─── -export async function composeReports(opts: ComposeReportsOptions): Promise<ComposedReports> { +export type ResolveSiteReportsOptions = Omit<ComposeReportsOptions, "publicDir"> & { + // Each report's downloads, as its view carries them. + downloads?: (reportId: string) => ReportDownloads | undefined; +}; + +// The site's reports resolved against the corpus — verified, their views and +// moment pages built, their prepared media matched — and nothing written. +// compose writes what this answers; publish/reportExports.ts renders the same +// views as files. Throws ComposeReportsError with every problem. +export type ResolvedSiteReports = { + // The reports (verification computed) and their views, in the site's order. + reports: Report[]; + views: ReportPageView[]; + index: ReportIndexView; + moments: MomentPageView[]; + // moment key → its prepared media, in `cacheDir`. + mediaOf: Map<string, ReportMediaEntry>; + cacheDir: string; + allowed: ComposeReportsProblem[]; +}; + +export async function resolveSiteReports(opts: ResolveSiteReportsOptions): Promise<ResolvedSiteReports> { const { paths, site } = opts; - const publicDir = opts.publicDir ?? paths.exportPublicDir; const log = opts.log ?? (() => {}); const now = (opts.now?.() ?? new Date()).toISOString(); const cited = isCitedSite(site); - // Whatever an earlier compose left. rm removes a link, never its target (a - // worktree's public/ entries may be links into the primary checkout). - for (const entry of REPORT_PUBLIC_ENTRIES) { - await rm(path.join(publicDir, entry), { recursive: true, force: true }); - } - if ((site.reports ?? []).length === 0) { - log("[reports] none published."); - return { reports: [], moments: [], allowed: [] }; - } - const problems: ComposeReportsProblem[] = []; const loaded = await loadSiteReports(paths, site); for (const p of loaded.problems) { @@ -627,10 +646,7 @@ export async function composeReports(opts: ComposeReportsOptions): Promise<Compo const post = postsByChannel.get(c.channel)?.get(c.id); return post ? { author: postAuthor(post), text: post.text, shot: postShot(c) } : undefined; }, - downloads: { - json: reportCitationsDownloadPath(report.id, "json"), - csv: reportCitationsDownloadPath(report.id, "csv"), - }, + downloads: opts.downloads?.(report.id), }), ); @@ -708,6 +724,44 @@ export async function composeReports(opts: ComposeReportsOptions): Promise<Compo }); } moments.sort((a, b) => (a.key < b.key ? -1 : a.key > b.key ? 1 : 0)); + return { reports, views, index, moments, mediaOf, cacheDir, allowed }; +} + +// ─── The stage ─── + +export async function composeReports(opts: ComposeReportsOptions): Promise<ComposedReports> { + const { paths, site } = opts; + const publicDir = opts.publicDir ?? paths.exportPublicDir; + const log = opts.log ?? (() => {}); + + // Whatever an earlier compose left. rm removes a link, never its target (a + // worktree's public/ entries may be links into the primary checkout). + for (const entry of REPORT_PUBLIC_ENTRIES) { + await rm(path.join(publicDir, entry), { recursive: true, force: true }); + } + if ((site.reports ?? []).length === 0) { + log("[reports] none published."); + return { reports: [], moments: [], allowed: [] }; + } + + // Each report's exports (publish/reportExports.ts), made from the report + // as it is now and small enough to publish; the citations as files always. + const exportsOf = new Map<string, PublishableReportExports>(); + for (const id of site.reports ?? []) { + const found = await publishableReportExports(paths, site.siteId, id); + exportsOf.set(id, found); + for (const note of found.notes) log(`[reports] ${id}: ${note}`); + } + const { reports, views, index, moments, mediaOf, cacheDir, allowed } = await resolveSiteReports({ + ...opts, + downloads: (id) => ({ + ...Object.fromEntries( + REPORT_EXPORT_FORMATS.filter((f) => exportsOf.get(id)?.files[f]).map((f) => [f, reportExportDownloadPath(id, f)]), + ), + json: reportCitationsDownloadPath(id, "json"), + csv: reportCitationsDownloadPath(id, "csv"), + }), + }); // ─── Writing ─── @@ -723,6 +777,11 @@ export async function composeReports(opts: ComposeReportsOptions): Promise<Compo const rel = (report.citations![c.id] as { image: string }).image; await copyOut(publicDir, path.join(siteReportDir(paths, site.siteId, report.id), rel), c.image); } + const exported = exportsOf.get(report.id)?.files ?? {}; + for (const f of REPORT_EXPORT_FORMATS) { + const src = exported[f]; + if (src) await copyOut(publicDir, src, reportExportDownloadPath(report.id, f)); + } } await writeOut( publicDir, @@ -753,7 +812,7 @@ export async function composeReports(opts: ComposeReportsOptions): Promise<Compo // A clip's published path: the moment's (lib/report/views.ts), `.m4a` for a // clip cut as sound. -function clipPath(m: SpanMoment, kind: "video" | "audio"): string { +export function clipPath(m: SpanMoment, kind: "video" | "audio"): string { const p = evidenceClipPath(m); return kind === "audio" ? p.replace(/\.mp4$/, ".m4a") : p; } diff --git a/common/publish/reportExportFiles.ts b/common/publish/reportExportFiles.ts @@ -0,0 +1,134 @@ +// WHERE A REPORT'S EXPORTS LIVE, and which of them may be published. +// +// `archilyzer reports export` (./reportExports.ts) writes a report's exports +// to `.export-index/sites/<siteId>/report-exports/<reportId>/` — staging, not +// served — beside a manifest, `export.json`, naming each file with its size +// and checksum and the sha256 of the report.json it was made from. Compose +// (./composeReports.ts) asks `publishableReportExports` which to copy into the +// site: a file is published only when +// - the manifest's report sha256 is the report.json's NOW (an export of an +// earlier version of the report is never shipped beside the new one), and +// - the file is within the publish limit (lib/builtExport.ts, +// PUBLISH_MAX_FILE_BYTES) — an evidence pack over it stays local. +// +// Kept apart from ./reportExports.ts, which renders the exports from the +// resolved reports and so imports compose: compose imports this. + +import { createHash } from "node:crypto"; +import { readFile, stat } from "node:fs/promises"; +import path from "node:path"; +import { publishFileSizeProblem } from "../lib/builtExport"; +import { readJsonFile } from "../lib/jsonFile-server"; +import type { Paths } from "../lib/paths"; +import type { ReportExportFooter } from "../lib/report/exportHtml"; +import { REPORT_EXPORT_FILENAMES, REPORT_EXPORT_FORMATS, type ReportExportFormat } from "../lib/report/views"; +import { siteIndexDir } from "../lib/site"; +import { siteReportFile } from "./reportMedia"; + +export const REPORT_EXPORTS_DIRNAME = "report-exports"; +export const REPORT_EXPORT_MANIFEST_FILENAME = "export.json"; +export const REPORT_EXPORT_MANIFEST_FORMAT = "archilyzer-report-export"; +export const REPORT_EXPORT_MANIFEST_VERSION = 1; + +// `.export-index/sites/<siteId>/report-exports/`. +export function reportExportsDir(paths: Paths, siteId: string): string { + return path.join(siteIndexDir(paths, siteId), REPORT_EXPORTS_DIRNAME); +} + +// `.export-index/sites/<siteId>/report-exports/<reportId>/`. +export function reportExportDir(paths: Paths, siteId: string, reportId: string): string { + return path.join(reportExportsDir(paths, siteId), reportId); +} + +export type ReportExportFileEntry = { + // The file's name in the report's export dir (REPORT_EXPORT_FILENAMES). + file: string; + bytes: number; + sha256: string; + // Why it may not be published (over the limit), or absent when it may. + localOnly?: string; +}; + +export type ReportExportManifest = { + format: typeof REPORT_EXPORT_MANIFEST_FORMAT; + version: typeof REPORT_EXPORT_MANIFEST_VERSION; + siteId: string; + reportId: string; + // The sha256 of the report.json the exports were made from. + reportSha256: string; + exportedAt: string; + footer: ReportExportFooter; + files: Partial<Record<ReportExportFormat, ReportExportFileEntry>>; + // What was skipped, and why (a PDF with no browser, a pack with no `zip`). + notes: string[]; +}; + +export function sha256Hex(data: string | Uint8Array): string { + return createHash("sha256").update(data).digest("hex"); +} + +// The sha256 of a report.json as it is on disk, or null when it is unreadable. +export async function reportFileSha256(paths: Paths, siteId: string, reportId: string): Promise<string | null> { + try { + return sha256Hex(await readFile(siteReportFile(paths, siteId, reportId))); + } catch { + return null; + } +} + +export async function readReportExportManifest( + paths: Paths, + siteId: string, + reportId: string, +): Promise<ReportExportManifest | null> { + const read = await readJsonFile(path.join(reportExportDir(paths, siteId, reportId), REPORT_EXPORT_MANIFEST_FILENAME)); + if (!read.ok) return null; + const v = read.value as Partial<ReportExportManifest> | null; + if (!v || v.format !== REPORT_EXPORT_MANIFEST_FORMAT || v.version !== REPORT_EXPORT_MANIFEST_VERSION) return null; + return v as ReportExportManifest; +} + +export type PublishableReportExports = { + // format → the local file to publish. + files: Partial<Record<ReportExportFormat, string>>; + // Why an export is not published, one sentence each. + notes: string[]; +}; + +// The exports of a report compose may publish: made from the report.json as +// it is now, each file present and within the publish limit. A report never +// exported has none, and no note. +export async function publishableReportExports( + paths: Paths, + siteId: string, + reportId: string, +): Promise<PublishableReportExports> { + const manifest = await readReportExportManifest(paths, siteId, reportId); + if (!manifest) return { files: {}, notes: [] }; + const current = await reportFileSha256(paths, siteId, reportId); + if (current !== manifest.reportSha256) { + return { + files: {}, + notes: ["its exports were made from another version of report.json — export again; none are published"], + }; + } + const dir = reportExportDir(paths, siteId, reportId); + const files: PublishableReportExports["files"] = {}; + const notes: string[] = []; + for (const f of REPORT_EXPORT_FORMATS) { + if (!manifest.files[f]) continue; + const file = path.join(dir, REPORT_EXPORT_FILENAMES[f]); + const st = await stat(file).catch(() => null); + if (!st?.isFile()) { + notes.push(`${REPORT_EXPORT_FILENAMES[f]} is named in export.json but missing — export again`); + continue; + } + const tooBig = publishFileSizeProblem(REPORT_EXPORT_FILENAMES[f], st.size); + if (tooBig) { + notes.push(`${tooBig}; it stays local`); + continue; + } + files[f] = file; + } + return { files, notes }; +} diff --git a/common/publish/reportExports.ts b/common/publish/reportExports.ts @@ -0,0 +1,592 @@ +// EXPORT A SITE'S REPORTS AS FILES — so a report survives a takedown as files +// anyone can save and host again (plans/report-sites.md, "Exports"). +// `archilyzer reports export <siteId>` and the editor's `reports-export` job +// run `exportSiteReports`; `reports prepare` runs it at its end. +// +// What it reads: the site's published reports, resolved exactly as compose +// resolves them (./composeReports.ts resolveSiteReports — verified against the +// corpus, the prepared media matched), and the media `reports prepare` left in +// the site's report-media cache (./reportMedia.ts) with each report's stills. +// +// What it writes, per report, to +// `.export-index/sites/<siteId>/report-exports/<reportId>/` (staging, never +// served; ./reportExportFiles.ts names it): +// +// report.html ONE self-contained file (lib/report/exportHtml.ts): +// inline CSS, no script, every still and post +// screenshot a data: URI, recompressed here to WebP (or +// JPEG) at most EXPORT_IMAGE_MAX_WIDTH wide; clips are +// linked on the site, never inlined +// report.pdf that HTML printed by headless Chromium (A4). Where +// Playwright or its browser is missing the PDF is skipped +// with a note — never a failure +// report.md plain Markdown, numbered references +// (lib/report/exportMarkdown.ts) +// evidence-pack.zip `<reportId>/report.html` pointing at its own media/ +// (the clips play offline, the stills and screenshots +// as files), report.md, citations.json and .csv. Packed +// by the system `zip`, deterministically (sorted names, +// fixed times and modes, no extra attributes); a host +// without `zip` fails the format, naming it +// export.json the manifest: each file's size and sha256, the sha256 +// of the report.json it was made from, the footer, notes +// +// Each run replaces the report's export dir whole: a format not asked for this +// time is gone, never left from an older version of the report. Compose +// publishes from here only what was made from the report.json as it is now, +// each file within the publish limit (an evidence pack over 24 MiB stays +// local). +// +// THE FOOTER (lib/report/exportHtml.ts ReportExportFooter) names which document +// this is: today the report's date and the sha256 of its report.json; +// `exportFooterFor` below is where the revision history (slice RH) fills in +// the revision number and the commit. + +import { chmod, mkdir, mkdtemp, readdir, readFile, rm, utimes, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { execa } from "execa"; +import { publishFileSizeProblem } from "../lib/builtExport"; +import { writeFileAtomic, writeJsonAtomic } from "../lib/jsonFile-server"; +import { getPaths, type Paths } from "../lib/paths"; +import { getSite, listSiteIds, type Site } from "../lib/site"; +import { parseMomentKey, type SpanMoment } from "../lib/citations/moments"; +import type { Report } from "../lib/report/schema"; +import { reportExportHtml, type ExportClip, type ReportExportFooter } from "../lib/report/exportHtml"; +import { reportExportMarkdown } from "../lib/report/exportMarkdown"; +import { + REPORT_EXPORT_FILENAMES, + REPORT_EXPORT_FORMATS, + type ReportExportFormat, + type ReportPageView, + type SpanCitationView, +} from "../lib/report/views"; +import { importPlaywright } from "../social/playwrightRuntime"; +import { + citationSet, + citationsCsv, + clipPath, + ComposeReportsError, + formatComposeReportsProblems, + resolveSiteReports, + type ResolvedSiteReports, +} from "./composeReports"; +import { + REPORT_EXPORT_MANIFEST_FILENAME, + REPORT_EXPORT_MANIFEST_FORMAT, + REPORT_EXPORT_MANIFEST_VERSION, + reportExportDir, + reportExportsDir, + reportFileSha256, + sha256Hex, + type ReportExportFileEntry, + type ReportExportManifest, +} from "./reportExportFiles"; +import { siteReportDir, type ReportMediaIndex } from "./reportMedia"; + +// A still or a screenshot in report.html: at most this wide, recompressed. +export const EXPORT_IMAGE_MAX_WIDTH = 1200; +export const EXPORT_WEBP_QUALITY = 80; +// The evidence pack's file times: fixed, so the same files pack the same. +export const EVIDENCE_PACK_MTIME = new Date("2000-01-01T00:00:00Z"); + +export type ReportExportProblem = { + // The report, absent for a problem of the whole run. + report?: string; + format?: ReportExportFormat; + message: string; +}; + +// Prints HTML to PDF. One is opened per run and closed at its end. +export type PdfPrinter = { + print: (html: string) => Promise<Uint8Array>; + close: () => Promise<void>; +}; + +// A printer, or why there is none (the PDF is then skipped with that note). +export type OpenPdfPrinter = () => Promise<PdfPrinter | { missing: string }>; + +export type ExportSiteReportsOptions = { + siteId: string; + paths?: Paths; + // One published report; default every one. + reportId?: string; + // Default: all of REPORT_EXPORT_FORMATS. + formats?: readonly ReportExportFormat[]; + // Export even where a citation's media was not prepared (its clip is then + // neither linked nor packed), as compose's `--allow-missing-media`. + allowMissingMedia?: boolean; + onLog?: (line: string) => void; + signal?: AbortSignal; + // `social.x.visibility` and friends; default the live settings. + settings?: { social?: { x?: { visibility?: unknown } } }; + now?: () => Date; + // Default: headless Chromium through Playwright (openPlaywrightPdfPrinter). + openPdfPrinter?: OpenPdfPrinter; + // Default: `zip` on PATH. + zipBin?: string; +}; + +export type ReportExportResult = { + reportId: string; + dir: string; + manifest: ReportExportManifest; +}; + +export type ExportSiteReportsResult = { + exported: ReportExportResult[]; + problems: ReportExportProblem[]; +}; + +const mib = (n: number) => `${(n / 1024 / 1024).toFixed(1)} MiB`; +const firstLine = (e: unknown) => String((e as Error)?.message ?? e).split("\n")[0].slice(0, 200); + +export function parseReportExportFormats(v: string): ReportExportFormat[] | null { + const parts = v.split(",").map((s) => s.trim()).filter(Boolean); + if (parts.length === 0) return null; + for (const p of parts) if (!(REPORT_EXPORT_FORMATS as readonly string[]).includes(p)) return null; + return REPORT_EXPORT_FORMATS.filter((f) => parts.includes(f)); +} + +// THE FOOTER HOOK. Today: the report's own date and the sha256 of its +// report.json. The revision history (slice RH) adds `revision` and `commit` +// here — nothing else changes: the HTML, PDF and Markdown all print +// reportExportFooterLine of what this returns, leaving out what is absent. +export function exportFooterFor(view: Pick<ReportPageView, "published" | "updated">, reportSha256: string): ReportExportFooter { + const date = view.updated ?? view.published; + return { ...(date ? { date } : {}), reportSha256 }; +} + +// ─── PDF ─── + +export const openPlaywrightPdfPrinter: OpenPdfPrinter = async () => { + let chromium; + try { + ({ chromium } = await importPlaywright()); + } catch { + return { missing: "Playwright is not available on this host" }; + } + let browser; + try { + browser = await chromium.launch({ headless: true }); + } catch (e) { + return { missing: `headless Chromium did not start (${firstLine(e)})` }; + } + const b = browser; + return { + print: async (html) => { + const context = await b.newContext({}); + try { + const page = await context.newPage(); + if (!page.setContent || !page.pdf) throw new Error("this Playwright cannot print a page"); + await page.setContent(html, { waitUntil: "load" }); + return await page.pdf({ format: "A4", printBackground: true }); + } finally { + await context.close(); + } + }, + close: () => b.close(), + }; +}; + +// ─── Images ─── + +const IMAGE_MIME: Record<string, string> = { + ".png": "image/png", + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".gif": "image/gif", + ".webp": "image/webp", + ".avif": "image/avif", +}; + +async function ffmpegImage(ffmpegBin: string, file: string, codec: "webp" | "jpeg", signal?: AbortSignal): Promise<Uint8Array | null> { + const scale = `scale='min(${EXPORT_IMAGE_MAX_WIDTH},iw)':-2`; + const tail = + codec === "webp" + ? ["-c:v", "libwebp", "-quality", String(EXPORT_WEBP_QUALITY), "-compression_level", "6", "-f", "webp"] + : ["-pix_fmt", "yuvj420p", "-c:v", "mjpeg", "-q:v", "4", "-f", "image2pipe"]; + const r = await execa( + ffmpegBin, + ["-hide_banner", "-loglevel", "error", "-i", file, "-frames:v", "1", "-vf", scale, "-map_metadata", "-1", + "-fflags", "+bitexact", "-flags", "+bitexact", ...tail, "pipe:1"], + { reject: false, encoding: "buffer", timeout: 60_000, cancelSignal: signal }, + ).catch(() => null); + if (!r || r.exitCode !== 0 || !(r.stdout instanceof Uint8Array) || r.stdout.length === 0) return null; + return r.stdout; +} + +// An image as a data: URI for report.html: recompressed to WebP (else JPEG) at +// most EXPORT_IMAGE_MAX_WIDTH wide when that is smaller than the file, else +// the file as it is. Null when the file cannot be read. +export async function exportImageDataUri(file: string, ffmpegBin: string, signal?: AbortSignal): Promise<string | null> { + let raw: Buffer; + try { + raw = await readFile(file); + } catch { + return null; + } + let best: { mime: string; data: Uint8Array } = { + mime: IMAGE_MIME[path.extname(file).toLowerCase()] ?? "application/octet-stream", + data: raw, + }; + for (const codec of ["webp", "jpeg"] as const) { + const out = await ffmpegImage(ffmpegBin, file, codec, signal); + if (out && out.length < best.data.length) best = { mime: codec === "webp" ? "image/webp" : "image/jpeg", data: out }; + if (out) break; + } + return `data:${best.mime};base64,${Buffer.from(best.data).toString("base64")}`; +} + +// The image paths a view names (a source sentence's still, a post's shot). +function viewImagePaths(view: ReportPageView): string[] { + const out = new Set<string>(); + for (const c of Object.values(view.citations)) { + if (c.kind === "source" && c.image) out.add(c.image); + if (c.kind === "post" && c.shot) out.add(c.shot); + } + return [...out].sort(); +} + +// ─── Where a view's files are on this host ─── + +type ReportFiles = { + // A site-root path the view names → the file on disk. + local: (sitePath: string) => string | undefined; + // A span citation's prepared clip: its site-root path and file. + clip: (c: SpanCitationView) => { sitePath: string; file: string; kind: "video" | "audio" } | undefined; +}; + +function within(base: string, rel: string): string | undefined { + const b = path.resolve(base); + const f = path.resolve(b, rel); + return f.startsWith(b + path.sep) ? f : undefined; +} + +function reportFiles(paths: Paths, siteId: string, view: ReportPageView, resolved: ResolvedSiteReports): ReportFiles { + const own = `/reports/${view.id}/`; + return { + local: (p) => { + if (p.startsWith(own)) return within(siteReportDir(paths, siteId, view.id), p.slice(own.length)); + if (p.startsWith("/media/")) return within(resolved.cacheDir, p.slice("/media/".length)); + return undefined; + }, + clip: (c) => { + const entry = resolved.mediaOf.get(c.moment); + if (!entry || entry.kind === "post") return undefined; + const file = within(resolved.cacheDir, entry.file); + if (!file) return undefined; + return { sitePath: clipPath(parseMomentKey(c.moment) as SpanMoment, entry.kind), file, kind: entry.kind }; + }, + }; +} + +// A site-root path's place in the evidence pack: the site's media under +// media/, the report's own files (its stills) under media/report/. +function packPathOf(view: ReportPageView, sitePath: string): string | undefined { + const own = `/reports/${view.id}/`; + if (sitePath.startsWith(own)) return `media/report/${sitePath.slice(own.length)}`; + if (sitePath.startsWith("/media/")) return sitePath.slice(1); + return undefined; +} + +// ─── The evidence pack ─── + +async function zipAvailable(zipBin: string): Promise<boolean> { + const r = await execa(zipBin, ["-h"], { reject: false }).catch(() => null); + return !!r && r.exitCode === 0; +} + +// `names` (relative to `cwd`) zipped into `outFile`, the same bytes for the +// same files: names sorted, every time and mode fixed, no extra attributes, +// times read in UTC. Media is stored, text deflated. +export async function zipDeterministic(cwd: string, names: readonly string[], outFile: string, zipBin = "zip"): Promise<void> { + const sorted = [...names].sort(); + for (const n of sorted) { + await chmod(path.join(cwd, n), 0o644); + await utimes(path.join(cwd, n), EVIDENCE_PACK_MTIME, EVIDENCE_PACK_MTIME); + } + await rm(outFile, { force: true }); + const r = await execa( + zipBin, + ["-X", "-D", "-q", "-n", ".mp4:.m4a:.png:.jpg:.jpeg:.webp:.gif:.avif:.pdf:.zip", outFile, "-@"], + { cwd, input: `${sorted.join("\n")}\n`, env: { TZ: "UTC" }, reject: false }, + ); + if (r.exitCode !== 0) throw new Error(`zip failed (exit ${r.exitCode}): ${String(r.stderr ?? "").trim().slice(0, 300)}`); +} + +async function buildEvidencePack(o: { + report: Report; + view: ReportPageView; + files: ReportFiles; + site: Site; + footer: ReportExportFooter; + markdown: string; + outFile: string; + zipBin: string; +}): Promise<void> { + const tmp = await mkdtemp(path.join(os.tmpdir(), "report-evidence-pack-")); + try { + const top = o.view.id; + const names: string[] = []; + const put = async (rel: string, data: string | Uint8Array) => { + const f = path.join(tmp, top, rel); + await mkdir(path.dirname(f), { recursive: true }); + await writeFile(f, data); + names.push(`${top}/${rel}`); + }; + const images = new Map<string, string>(); + for (const p of viewImagePaths(o.view)) { + const local = o.files.local(p); + const rel = packPathOf(o.view, p); + if (!local || !rel) continue; + const data = await readFile(local).catch(() => null); + if (!data) continue; + await put(rel, data); + images.set(p, rel); + } + const clips = new Map<string, ExportClip>(); + for (const c of Object.values(o.view.citations)) { + if (c.kind !== "video" && c.kind !== "audio") continue; + const clip = o.files.clip(c); + if (!clip || clips.has(c.moment)) continue; + const rel = packPathOf(o.view, clip.sitePath); + const data = rel ? await readFile(clip.file).catch(() => null) : null; + if (!rel || !data) continue; + await put(rel, data); + clips.set(c.moment, { href: rel, kind: clip.kind, play: true }); + } + const html = reportExportHtml(o.view, { + siteUrl: o.site.siteUrl || undefined, + siteTitle: o.site.siteTitle || undefined, + footer: o.footer, + image: (p) => images.get(p), + clip: (c) => clips.get(c.moment), + }); + await put(REPORT_EXPORT_FILENAMES.html, html); + await put(REPORT_EXPORT_FILENAMES.md, o.markdown); + await put("citations.json", `${JSON.stringify(citationSet(o.report, o.view), null, 2)}\n`); + await put("citations.csv", citationsCsv(o.view)); + const tmpZip = path.join(tmp, "pack.zip"); + await zipDeterministic(tmp, names, tmpZip, o.zipBin); + await writeFileAtomic(o.outFile, await readFile(tmpZip)); + } finally { + await rm(tmp, { recursive: true, force: true }); + } +} + +// ─── One report ─── + +async function exportOneReport(o: { + paths: Paths; + site: Site; + report: Report; + view: ReportPageView; + resolved: ResolvedSiteReports; + formats: readonly ReportExportFormat[]; + printer: () => Promise<PdfPrinter | { missing: string }>; + zipBin: string; + now: Date; + log: (line: string) => void; + signal?: AbortSignal; +}): Promise<{ result?: ReportExportResult; problems: ReportExportProblem[] }> { + const { paths, site, report, view } = o; + const problems: ReportExportProblem[] = []; + const sha = await reportFileSha256(paths, site.siteId, report.id); + if (!sha) return { problems: [{ report: report.id, message: "its report.json cannot be read" }] }; + const footer = exportFooterFor(view, sha); + const files = reportFiles(paths, site.siteId, view, o.resolved); + const siteUrl = site.siteUrl || undefined; + const siteTitle = site.siteTitle || undefined; + + // The one-file export's images, as data: URIs. + const dataUris = new Map<string, string>(); + for (const p of viewImagePaths(view)) { + const local = files.local(p); + const uri = local ? await exportImageDataUri(local, paths.ffmpegBin, o.signal) : null; + if (uri) dataUris.set(p, uri); + else problems.push({ report: report.id, format: "html", message: `the image ${p} cannot be read` }); + } + const html = reportExportHtml(view, { + siteUrl, + siteTitle, + footer, + image: (p) => dataUris.get(p), + clip: (c) => { + const clip = files.clip(c); + const href = clip && siteUrl ? `${siteUrl.replace(/\/+$/, "")}${clip.sitePath}` : undefined; + return clip && href ? { href, kind: clip.kind } : undefined; + }, + }); + const markdown = reportExportMarkdown(view, { siteUrl, siteTitle, footer }); + + const dir = reportExportDir(paths, site.siteId, report.id); + await rm(dir, { recursive: true, force: true }); + await mkdir(dir, { recursive: true }); + const entries: ReportExportManifest["files"] = {}; + const notes: string[] = []; + const record = async (f: ReportExportFormat) => { + const name = REPORT_EXPORT_FILENAMES[f]; + const data = await readFile(path.join(dir, name)); + const localOnly = publishFileSizeProblem(name, data.length); + const entry: ReportExportFileEntry = { file: name, bytes: data.length, sha256: sha256Hex(data), ...(localOnly ? { localOnly } : {}) }; + entries[f] = entry; + o.log(` + ${report.id}/${name} (${mib(entry.bytes)})${localOnly ? ` — local only: ${localOnly}` : ""}`); + }; + + for (const f of o.formats) { + if (o.signal?.aborted) throw new Error("export cancelled"); + const out = path.join(dir, REPORT_EXPORT_FILENAMES[f]); + switch (f) { + case "html": + await writeFileAtomic(out, html); + await record(f); + break; + case "md": + await writeFileAtomic(out, markdown); + await record(f); + break; + case "pdf": { + const printer = await o.printer(); + if ("missing" in printer) { + notes.push(`report.pdf skipped: ${printer.missing}`); + o.log(` - ${report.id}/report.pdf skipped: ${printer.missing}`); + break; + } + try { + await writeFileAtomic(out, Buffer.from(await printer.print(html))); + await record(f); + } catch (e) { + notes.push(`report.pdf skipped: printing failed (${firstLine(e)})`); + o.log(` - ${report.id}/report.pdf skipped: printing failed (${firstLine(e)})`); + } + break; + } + case "zip": + if (!(await zipAvailable(o.zipBin))) { + problems.push({ + report: report.id, + format: "zip", + message: `the evidence pack needs the \`zip\` program (Info-ZIP), and "${o.zipBin}" cannot be run on this host — install zip, or export without zip (--formats html,pdf,md)`, + }); + break; + } + try { + await buildEvidencePack({ report, view, files, site, footer, markdown, outFile: out, zipBin: o.zipBin }); + await record(f); + } catch (e) { + problems.push({ report: report.id, format: "zip", message: `the evidence pack could not be packed: ${firstLine(e)}` }); + } + break; + } + } + + const manifest: ReportExportManifest = { + format: REPORT_EXPORT_MANIFEST_FORMAT, + version: REPORT_EXPORT_MANIFEST_VERSION, + siteId: site.siteId, + reportId: report.id, + reportSha256: sha, + exportedAt: o.now.toISOString(), + footer, + files: entries, + notes, + }; + await writeJsonAtomic(path.join(dir, REPORT_EXPORT_MANIFEST_FILENAME), manifest); + return { result: { reportId: report.id, dir, manifest }, problems }; +} + +// ─── The run ─── + +export async function exportSiteReports(opts: ExportSiteReportsOptions): Promise<ExportSiteReportsResult> { + const paths = opts.paths ?? getPaths(); + const log = opts.onLog ?? (() => {}); + const { siteId } = opts; + if (!listSiteIds(paths).includes(siteId)) throw new Error(`no site "${siteId}" (no sites/${siteId}/site.json)`); + const site = getSite(siteId, paths); + const published = site.reports ?? []; + if (opts.reportId && !published.includes(opts.reportId)) { + throw new Error(`site "${siteId}" does not publish a report "${opts.reportId}" (only a published report is exported)`); + } + const formats = opts.formats ?? REPORT_EXPORT_FORMATS; + + let resolved: ResolvedSiteReports; + try { + resolved = await resolveSiteReports({ + paths, + site, + allowMissingMedia: opts.allowMissingMedia, + settings: opts.settings, + now: opts.now, + log, + }); + } catch (e) { + if (e instanceof ComposeReportsError) { + return { + exported: [], + problems: formatComposeReportsProblems(e.problems).map((message) => ({ message: `${message} (as compose would refuse it)` })), + }; + } + throw e; + } + + // A report no longer published keeps no exports. + if (!opts.reportId) { + for (const name of await readdir(reportExportsDir(paths, siteId)).catch(() => [] as string[])) { + if (!published.includes(name)) await rm(path.join(reportExportsDir(paths, siteId), name), { recursive: true, force: true }); + } + } + + // Opened on the first PDF, once for the run. + const opened: { printer?: Promise<PdfPrinter | { missing: string }> } = {}; + const openPrinter = () => (opened.printer ??= (opts.openPdfPrinter ?? openPlaywrightPdfPrinter)()); + const now = opts.now?.() ?? new Date(); + const exported: ReportExportResult[] = []; + const problems: ReportExportProblem[] = []; + log(`${siteId}: exporting ${opts.reportId ?? `${resolved.reports.length} report(s)`} as ${formats.join(", ")}.`); + try { + for (let i = 0; i < resolved.reports.length; i++) { + const report = resolved.reports[i]; + if (opts.reportId && report.id !== opts.reportId) continue; + if (opts.signal?.aborted) throw new Error("export cancelled"); + const r = await exportOneReport({ + paths, + site, + report, + view: resolved.views[i], + resolved, + formats, + printer: openPrinter, + zipBin: opts.zipBin ?? "zip", + now, + log, + signal: opts.signal, + }); + if (r.result) exported.push(r.result); + problems.push(...r.problems); + } + } finally { + const p = await opened.printer; + if (p && !("missing" in p)) await p.close().catch(() => {}); + } + log(`${exported.length} report(s) exported; ${problems.length} problem(s).`); + return { exported, problems }; +} + +export function formatReportExportProblems(problems: readonly ReportExportProblem[]): string[] { + return problems.map((p) => `${p.report ?? "site"}${p.format ? ` (${p.format})` : ""}: ${p.message}`); +} + +// The end of `reports prepare`: export the reports when their media is +// complete; a prepare with problems leaves the exports as they were (compose +// would refuse the site anyway). +export async function exportAfterPrepare( + index: ReportMediaIndex, + opts: ExportSiteReportsOptions, +): Promise<ExportSiteReportsResult | null> { + if (index.problems.length > 0) { + opts.onLog?.("exports: not made — the evidence media is not complete."); + return null; + } + return exportSiteReports(opts); +} diff --git a/common/publish/reportMedia.test.ts b/common/publish/reportMedia.test.ts @@ -225,5 +225,31 @@ test("a clean site: no problems, and a clip no longer cited leaves the cache", a const files = readdirSync(reportMediaDir(paths, SITE)).sort(); const clip = index.moments[VIDEO_KEY].file; assert.deepEqual(files, [clip.replace(/\.mp4$/, ".json"), clip, "index.json"].sort()); - assert.equal(await prepareMain({ siteId: SITE }, { log: () => {}, error: () => {} }), 0); +}); + +test("a clean prepare ends by exporting the reports; a host with no browser skips the PDF with a note", async () => { + // The record behind r2's citation, so the export can resolve it as compose does. + writeJson(path.join(videoDir(CH, "abc123"), "metadata.info.json"), { + id: "abc123", + title: "Demo stream", + upload_date: "20260110", + webpage_url: "https://www.youtube.com/watch?v=abc123", + extractor_key: "Youtube", + }); + writeFileSync( + path.join(videoDir(CH, "abc123"), "transcript.en.vtt"), + "WEBVTT\n\n00:00:00.000 --> 00:00:06.000 align:start position:0%\none<00:00:00.000><c></c>\n", + ); + const out: string[] = []; + const err: string[] = []; + const code = await prepareMain( + { siteId: SITE, exportOptions: { openPdfPrinter: async () => ({ missing: "no browser here" }) } }, + { log: (l) => out.push(l), error: (l) => err.push(l) }, + ); + assert.equal(code, 0, err.join("\n")); + const dir = path.join(paths.exportSitesIndexDir, SITE, "report-exports", "r2"); + assert.deepEqual(readdirSync(dir).sort(), ["evidence-pack.zip", "export.json", "report.html", "report.md"]); + const manifest = JSON.parse(readFileSync(path.join(dir, "export.json"), "utf8")); + assert.deepEqual(manifest.notes, ["report.pdf skipped: no browser here"]); + assert.ok(out.some((l) => l.includes("report.pdf skipped: no browser here"))); }); diff --git a/common/publish/source.ts b/common/publish/source.ts @@ -42,6 +42,7 @@ import { runChildIntoLog } from "../jobs/runChild"; import { copyPublicFile, ownDir, writePublicFile } from "../bin/_publicFile"; import { getPaths, type Paths } from "../lib/paths"; import { PROJECT_NAME, PROJECT_URL } from "../lib/project"; +import { PUBLISH_MAX_FILE_BYTES } from "../lib/builtExport"; import { CLONE_URL, HISTORY_DIR, @@ -118,7 +119,7 @@ export const HOME_REPLACEMENT = "/home/user"; // Cloudflare Pages allows 20,000 files per deployment and 25 MiB per file; // the step refuses well inside both, leaving the rest of the site its room. export const MAX_FILES = 15_000; -export const MAX_FILE_BYTES = 24 * 1024 * 1024; +export const MAX_FILE_BYTES = PUBLISH_MAX_FILE_BYTES; // Packs are split at this size (under the per-file cap, with room to grow). const PACK_SIZE = "20m"; diff --git a/common/social/playwrightRuntime.ts b/common/social/playwrightRuntime.ts @@ -1,5 +1,6 @@ // Locating Playwright at runtime, for the modules that need a real browser -// (the Nitter fetcher, the X fallback fetcher and the X session broker). +// (the Nitter fetcher, the X fallback fetcher, the X session broker, and a +// report's PDF export). // // Two constraints pull against each other: // - `common` must NOT depend on Playwright. It is imported by the Docker @@ -48,6 +49,14 @@ export type PageLike = { fullPage?: boolean; clip?: { x: number; y: number; width: number; height: number }; }) => Promise<Uint8Array>; + // A report's PDF export (publish/reportExports.ts): the export's HTML set as + // the page, printed. Optional: the fetchers' fakes do not print. + setContent?: (html: string, opts?: { waitUntil?: "load" | "domcontentloaded" | "networkidle" }) => Promise<void>; + pdf?: (opts?: { + format?: string; + printBackground?: boolean; + margin?: { top?: string; right?: string; bottom?: string; left?: string }; + }) => Promise<Uint8Array>; // The page's own request context (the profile's cookies): an X Article's // inline images are fetched through it (xArticleCapture.ts). request?: {