commit c407383f03aa83336de341c60ccc399bc5a4ed57
parent 31d0aaca17f251a423207cff6596b3cf7b596518
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 15:22:22 -0400
Merge report/exports (per-report HTML, PDF, Markdown and evidence-pack exports; reports-export job; shared publish size limit)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
# Conflicts:
# editor/CHANGELOG.md
Diffstat:
42 files changed, 2999 insertions(+), 77 deletions(-)
diff --git a/PUBLISH.md b/PUBLISH.md
@@ -53,7 +53,8 @@ entry points in `common/publish/build.ts`.
| The hub | /sites → Hub → **Build hub** / **Deploy hub** | `build-hub`, `deploy-hub` | `build hub`, `deploy hub [--preview <branch>]` |
| The homepage | /sites → Homepage → **Build homepage** (tick *Deploy after build*) / **Deploy homepage**, with an optional preview branch | `build-homepage` (`{"deploy":true}` to deploy after), `deploy-homepage` (`{"preview":"<branch>"}`) | `build homepage [--no-source]`, `deploy homepage [--preview <branch>]` |
| The source mirror alone | — (every homepage build runs it) | — | `source publish [--force] [--check] [--keep-scratch]`, `source audit [<git dir>]` |
-| A site's report evidence media | — (a Reports tab is to come) | `reports-prepare` (`{"siteId"}`) | `reports prepare <id>` |
+| A site's report evidence media (then its exports) | a site's **Reports** tab → *Prepare evidence media* | `reports-prepare` (`{"siteId"}`) | `reports prepare <id>` |
+| A site's report exports (HTML, PDF, Markdown, evidence pack) | a site's **Reports** tab → *Export reports* | `reports-export` (`{"siteId", "reportId"?, "formats"?}`) | `reports export <id> [--report <rid>] [--formats html,pdf,md,zip]` |
| A report from a /sweep report, an /ask answer or a report-to-video manifest, and a starter manifest from a report | — | — | `reports convert <sweep\|ask\|manifest> <in> --out <report.json> [--channels-dir <dir>]`, `reports to-manifest <report.json> --out <manifest.json>` |
`pnpm archilyzer <command>` is the short form of
@@ -74,9 +75,25 @@ is not on disk, a clip over 24 MiB or an invalid report is listed and fails the
(exit 1, or a failed job) — fetch the window or persist the video, capture the
post, and run it again; what is already cut is reused.
+When nothing is missing, prepare ends by **exporting** the reports (`reports export`
+runs the same step alone): each published report is written, as compose would
+resolve it, to `.export-index/sites/<id>/report-exports/<report>/` as `report.html`
+(one self-contained file — inline style, no script, its stills and post screenshots
+inlined; clips linked on the site), `report.pdf` (that page printed by headless
+Chromium; skipped with a note on a host without Playwright's browser), `report.md`
+(plain Markdown with numbered references) and `evidence-pack.zip` (the page with its
+clips, stills and screenshots as files, so the clips play offline, plus the Markdown
+and the citations; packed by the system `zip`, which a host must have), with an
+`export.json` naming each file's size and checksum and the hash of the report.json it
+was made from. Every export ends with a footer naming the report's date and the first
+12 characters of that hash.
+
The build's compose then writes the reports from what prepare left
(`common/publish/composeReports.ts`): each report's page, its citations as
-`citations.json` and `citations.csv`, its cited stills (never a saved source copy),
+`citations.json` and `citations.csv`, its exports (`report.html`, `report.pdf`,
+`report.md`, `evidence-pack.zip` — only those made from the report.json as it is now,
+and only a file of at most 24 MiB: a larger pack stays on the host; the page's
+download line lists what was published), its cited stills (never a saved source copy),
one page per cited moment with the transcript lines around it, and the prepared clips
and captures — only those the published reports cite. Every quote is checked as it is
composed: a span's against its cues within 5 s either side (a record whose `en` track
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -159,7 +159,7 @@ export const COMMANDS: Command[] = [
{
path: ["reports", "prepare"],
usage:
- "<id> cut every clip and copy every post capture the site's published reports cite into its report-media cache, before its build (exit 1 when a citation lacks media; default id: SITE_ID)",
+ "<id> cut every clip and copy every post capture the site's published reports cite into its report-media cache, before its build, then export the reports as files (reports export) when nothing is missing (exit 1 when a citation lacks media or an export fails; default id: SITE_ID)",
maxPositionals: 1,
run: async ({ positionals, env }) => {
const siteId = siteIdFrom(positionals, env, "reports prepare");
@@ -168,6 +168,30 @@ export const COMMANDS: Command[] = [
},
},
{
+ path: ["reports", "export"],
+ usage:
+ "<id> [--report <reportId>] [--formats html,pdf,md,zip] [--allow-missing-media] write the site's published reports as files (report.html, report.pdf, report.md, evidence-pack.zip) into its report-exports staging, where compose publishes them from (exit 1 on a problem; a PDF skipped for want of a browser is a note; default id: SITE_ID)",
+ flags: { report: "string", formats: "string", "allow-missing-media": "boolean" },
+ maxPositionals: 1,
+ run: async ({ positionals, flags, env }) => {
+ const siteId = siteIdFrom(positionals, env, "reports export");
+ if (!siteId) return 2;
+ const { parseReportExportFormats } = await import("../publish/reportExports");
+ const formats = typeof flags.formats === "string" ? parseReportExportFormats(flags.formats) : undefined;
+ if (formats === null) {
+ console.error("reports export: --formats is a comma-separated list of html, pdf, md, zip");
+ return 2;
+ }
+ return (await import("./reports-export")).main({
+ siteId,
+ signal: interrupted(),
+ ...(typeof flags.report === "string" ? { reportId: flags.report } : {}),
+ ...(formats ? { formats } : {}),
+ allowMissingMedia: flags["allow-missing-media"] === true,
+ });
+ },
+ },
+ {
path: ["reports", "convert"],
usage:
"<sweep|ask|manifest> <in> --out <report.json> [--channels-dir <dir>] [--id <id>] [--title <title>] a /sweep report (markdown), an /ask answer or a report-to-video manifest as a report.json, written only when it validates (--channels-dir: widen spans from the cues, find posts' channels)",
diff --git a/common/bin/reports-export.ts b/common/bin/reports-export.ts
@@ -0,0 +1,38 @@
+// `archilyzer reports export <siteId> [--report <id>] [--formats html,pdf,md,zip]`
+// — write a site's published reports as files (report.html, report.pdf,
+// report.md, evidence-pack.zip) into
+// `.export-index/sites/<siteId>/report-exports/<reportId>/`, where compose
+// publishes them from. The work is publish/reportExports.ts's, the same the
+// editor's `reports-export` job runs; this file prints its log and problems.
+//
+// Exit 0 when every asked-for export was written (a PDF skipped for want of a
+// browser is a note, not a problem); 1 when any problem is listed; 2 for bad
+// arguments or a site or report that does not exist.
+
+import {
+ exportSiteReports,
+ formatReportExportProblems,
+ type ExportSiteReportsOptions,
+} from "../publish/reportExports";
+
+type Out = { log: (s: string) => void; error: (s: string) => void };
+
+export async function main(
+ opts: Omit<ExportSiteReportsOptions, "onLog">,
+ out: Out = console,
+): Promise<number> {
+ let result;
+ try {
+ result = await exportSiteReports({ ...opts, onLog: out.log });
+ } catch (err) {
+ out.error(`reports export: ${(err as Error).message}`);
+ return opts.signal?.aborted ? 1 : 2;
+ }
+ for (const r of result.exported) {
+ for (const note of r.manifest.notes) out.log(` note: ${r.reportId}: ${note}`);
+ }
+ if (result.problems.length === 0) return 0;
+ out.error(`reports export ${opts.siteId}: ${result.problems.length} problem(s):`);
+ for (const line of formatReportExportProblems(result.problems)) out.error(` ${line}`);
+ return 1;
+}
diff --git a/common/bin/reports-prepare.ts b/common/bin/reports-prepare.ts
@@ -4,16 +4,25 @@
// editor's `reports-prepare` job runs; this file prints its log and its
// problems.
//
-// Exit 0 when every cited moment has its media; 1 when any problem is listed
-// (the manifest is written either way, problems included); 2 for a site that
-// does not exist.
+// When nothing is missing it then exports the reports as files
+// (publish/reportExports.ts, `archilyzer reports export`).
+//
+// Exit 0 when every cited moment has its media and every export was written;
+// 1 when any problem is listed (the manifest is written either way, problems
+// included); 2 for a site that does not exist.
import { formatReportMediaProblems, prepareReportMedia } from "../publish/reportMedia";
+import { exportAfterPrepare, formatReportExportProblems, type ExportSiteReportsOptions } from "../publish/reportExports";
type Out = { log: (s: string) => void; error: (s: string) => void };
export async function main(
- opts: { siteId: string; signal?: AbortSignal },
+ opts: {
+ siteId: string;
+ signal?: AbortSignal;
+ // Passed to the export at the end (tests inject the PDF printer and zip).
+ exportOptions?: Partial<Pick<ExportSiteReportsOptions, "openPdfPrinter" | "zipBin" | "paths" | "settings">>;
+ },
out: Out = console,
): Promise<number> {
let index;
@@ -23,8 +32,21 @@ export async function main(
out.error(`reports prepare: ${(err as Error).message}`);
return opts.signal?.aborted ? 1 : 2;
}
- if (index.problems.length === 0) return 0;
- out.error(`reports prepare ${opts.siteId}: ${index.problems.length} problem(s):`);
- for (const line of formatReportMediaProblems(index.problems)) out.error(` ${line}`);
+ if (index.problems.length > 0) {
+ out.error(`reports prepare ${opts.siteId}: ${index.problems.length} problem(s):`);
+ for (const line of formatReportMediaProblems(index.problems)) out.error(` ${line}`);
+ out.error("reports prepare: the reports were not exported (the evidence media is not complete).");
+ return 1;
+ }
+ let exported;
+ try {
+ exported = await exportAfterPrepare(index, { siteId: opts.siteId, signal: opts.signal, onLog: out.log, ...opts.exportOptions });
+ } catch (err) {
+ out.error(`reports prepare: export: ${(err as Error).message}`);
+ return 1;
+ }
+ if (!exported || exported.problems.length === 0) return 0;
+ out.error(`reports prepare ${opts.siteId}: export: ${exported.problems.length} problem(s):`);
+ for (const line of formatReportExportProblems(exported.problems)) out.error(` ${line}`);
return 1;
}
diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts
@@ -86,6 +86,8 @@ const ADDED_KINDS: Record<string, { label: string; drainable: boolean }> = {
"persist-videos": { label: "Persist videos", drainable: true },
// A report site's evidence media: one pass, cancelled rather than drained.
"reports-prepare": { label: "Prepare report media", drainable: false },
+ // A report site's exports: one pass, cancelled rather than drained.
+ "reports-export": { label: "Export reports", drainable: false },
};
test("added kinds carry their pinned label and drainability", () => {
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -679,6 +679,23 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
queueKeyStrategy: "custom",
needsMedia: true,
},
+ // A REPORT SITE'S EXPORTS (publish/reportExports.ts): each published report
+ // as report.html, report.pdf, report.md and an evidence pack, into the
+ // site's report-exports staging, for compose to publish. It reads the
+ // reports' records (text) and the prepared media cache — never a channel's
+ // big files — so `needsText`; like prepare it spans channels with no
+ // `channelSlug`, and compose's own check reports an unreadable channel. On
+ // prepare's queue (`reports-prepare`): it reads the cache prepare writes and
+ // prunes. Replayable: the spec is the site (and the report and formats).
+ "reports-export": {
+ kind: "reports-export",
+ label: "Export reports",
+ drainable: false,
+ replayable: true,
+ queueKeyStrategy: "custom",
+ needsMedia: false,
+ needsText: true,
+ },
// THE PER-VIDEO WRITERS THAT WERE NOT IN THIS TABLE (release 16 slice RM).
// Each runs with a channelSlug and writes under `data/<id>/` — a single
// video's transcription (the video page's two Transcribe buttons, and its
diff --git a/common/lib/builtExport.test.ts b/common/lib/builtExport.test.ts
@@ -13,7 +13,10 @@ import {
builtSiteProblem,
citedBuildProblem,
deployAudienceProblem,
+ PAGES_MAX_FILE_BYTES,
PAGES_MAX_FILES,
+ PUBLISH_MAX_FILE_BYTES,
+ publishFileSizeProblem,
} from "./builtExport";
function tempOut(siteJson?: string): { dir: string; cleanup: () => void } {
@@ -390,3 +393,13 @@ test("a site configured cited with a full build is refused at deploy; a cited bu
cited.cleanup();
}
});
+
+test("the publish limit: 24 MiB, inside Pages' 25, one sentence naming the file over it", () => {
+ assert.equal(PUBLISH_MAX_FILE_BYTES, 24 * 1024 * 1024);
+ assert.ok(PUBLISH_MAX_FILE_BYTES < PAGES_MAX_FILE_BYTES);
+ assert.equal(publishFileSizeProblem("evidence-pack.zip", PUBLISH_MAX_FILE_BYTES), null);
+ assert.equal(
+ publishFileSizeProblem("evidence-pack.zip", PUBLISH_MAX_FILE_BYTES + 1),
+ "evidence-pack.zip is 24.0 MiB, over the publish limit of 24.0 MiB (Pages allows 25 MiB per file)",
+ );
+});
diff --git a/common/lib/builtExport.ts b/common/lib/builtExport.ts
@@ -223,6 +223,21 @@ export const CITED_MEDIA_ALLOWED_DIRS: readonly string[] = ["clips", "posts"];
export const PAGES_MAX_FILE_BYTES = 25 * 1024 * 1024;
export const PAGES_MAX_FILES = 20_000;
+// What a step that PUTS a file on a site holds it to: 24 MiB, a mebibyte inside
+// Pages' own limit (the source mirror's packs, an evidence clip, a report's
+// exports). One number, so every step refuses the same file.
+export const PUBLISH_MAX_FILE_BYTES = 24 * 1024 * 1024;
+
+/**
+ * Why a file of `bytes` may not be published, as one sentence naming `rel` — or
+ * null when it fits under PUBLISH_MAX_FILE_BYTES.
+ */
+export function publishFileSizeProblem(rel: string, bytes: number): string | null {
+ if (bytes <= PUBLISH_MAX_FILE_BYTES) return null;
+ const mib = (n: number) => (n / (1024 * 1024)).toFixed(1);
+ return `${rel} is ${mib(bytes)} MiB, over the publish limit of ${mib(PUBLISH_MAX_FILE_BYTES)} MiB (Pages allows 25 MiB per file)`;
+}
+
// corpus.json's `site.scope`, or null.
function builtScopeIn(outDir: string): string | null {
try {
diff --git a/common/lib/evidenceClip-server.ts b/common/lib/evidenceClip-server.ts
@@ -46,6 +46,7 @@ import { createReadStream } from "node:fs";
import { mkdir, readFile, rename, rm, stat } from "node:fs/promises";
import path from "node:path";
import { execa } from "execa";
+import { PUBLISH_MAX_FILE_BYTES } from "./builtExport";
import { tmpPathFor, writeJsonAtomic } from "./jsonFile-server";
import { roundMomentSeconds } from "./citations/moments";
import type { CitationPad } from "./citations/schema";
@@ -66,7 +67,7 @@ export const EVIDENCE_CRF = 23;
export const EVIDENCE_AUDIO_BITRATE = "128k";
// 24 MiB: a Pages file limit is 25 MiB, and a clip is published as one file.
-export const EVIDENCE_MAX_BYTES = 24 * 1024 * 1024;
+export const EVIDENCE_MAX_BYTES = PUBLISH_MAX_FILE_BYTES;
// A cut of a cached window is seconds of work; a whole saved container on a
// platter seeks once. Generous, so only a wedged ffmpeg reaches it.
diff --git a/common/lib/report/export.test.ts b/common/lib/report/export.test.ts
@@ -0,0 +1,236 @@
+// The report exports' renderers (exportHtml.ts, exportMarkdown.ts), from a
+// fixture view: deterministic, self-contained, numbered.
+//
+// Run with: node_modules/.bin/tsx --test lib/report/export.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import type { Report } from "./schema";
+import { buildReportPageView, type RecordView } from "./views";
+import { markdownToHtml, reportExportFooterLine, reportExportHtml, type ReportExportHtmlOptions } from "./exportHtml";
+import { citedMarkdownToPlain, mdText, reportExportMarkdown } from "./exportMarkdown";
+
+const report: Report = {
+ format: "archilyzer-report",
+ version: 1,
+ id: "demo",
+ kind: "factcheck",
+ series: "Demo Checks",
+ title: "A demo fact-check",
+ subtitle: "Two claims, tested",
+ summary: "It starts [here](cite:v1). <script>alert(1)</script> and [a link](javascript:alert(1)).",
+ method: "Each quote was checked against the recording.",
+ published: "2026-10-01",
+ updated: "2026-10-04T10:00:00Z",
+ subject: { source: "s0" },
+ verdicts: { PARTLY: { label: "Half" } },
+ sources: {
+ s0: {
+ kind: "article",
+ title: "An article",
+ url: "https://example.org/a",
+ author: "A. Writer",
+ publisher: "Example Gazette",
+ accent: "#aa3300",
+ archives: [{ label: "archive", url: "https://archive.example.org/a", context: "the first edition" }],
+ saved: "sources/s0/page.html",
+ },
+ },
+ citations: {
+ v1: { kind: "video", channel: "demo-channel", id: "abc123", start: 3126, end: 3151, quote: "one * two", origin: "added", verification: { quoteScore: 0.97, quoteCheckedAt: "2026-10-04T12:00:00Z" } },
+ a1: { kind: "audio", channel: "demo-podcast", id: "ep-1", start: 10.5, end: 20, quote: "two", speaker: "Guest", origin: "subject" },
+ p1: { kind: "post", channel: "demo-social", id: "123", quote: "three" },
+ s1: { kind: "source", source: "s0", quote: "four", image: "stills/s1.png" },
+ w1: { kind: "page", url: "https://example.org/p", title: "A page", quote: "five", archiveUrl: "https://archive.example.org/p" },
+ },
+ sections: [
+ {
+ id: "one",
+ title: "One",
+ claims: [
+ {
+ id: "c1",
+ title: "One",
+ text: "Claim one.",
+ verdict: "CONTRADICTED",
+ gist: "The recording says otherwise.",
+ flag: "No source given",
+ sourceQuote: { citation: "s1" },
+ findings: "See [this](cite:w1), **bold**, and `[code](cite:never)`.\n\n- a list item\n- another",
+ citations: ["v1", "a1"],
+ },
+ ],
+ },
+ {
+ id: "two",
+ title: "Two",
+ body: "A body citing [a post](cite:p1).",
+ claims: [{ id: "c2", text: "Claim two.", verdict: "PARTLY", citations: ["a1"] }],
+ },
+ ],
+};
+
+const RECORDS: Record<string, RecordView> = {
+ "demo-channel/abc123": { channel: "demo-channel", channelTitle: "Demo Channel", id: "abc123", title: "Demo stream", date: "2026-01-10", originalUrl: "https://media.example.org/abc123?t=3126" },
+ "demo-podcast/ep-1": { channel: "demo-podcast", channelTitle: "Demo Podcast", id: "ep-1", title: "Episode 1", date: "2026-02-02", originalUrl: "https://media.example.org/ep-1" },
+ "demo-social/123": { channel: "demo-social", channelTitle: "Demo Social", id: "123", date: "2025-03-14", originalUrl: "https://social.example.org/123" },
+};
+
+const view = buildReportPageView(report, {
+ record: (c) => RECORDS[`${c.channel}/${c.id}`],
+ post: () => ({ author: "@demo", text: "three", shot: "/media/posts/demo-social/123/shot.png" }),
+});
+
+const PNG = "data:image/png;base64,iVBORw0KGgo=";
+const opts: ReportExportHtmlOptions = {
+ siteUrl: "https://reports.example.org/",
+ siteTitle: "Demo Site",
+ footer: { date: "2026-10-04", reportSha256: "0123456789abcdef".repeat(4) },
+ image: () => PNG,
+ clip: (c) => ({ href: `https://reports.example.org/media/clips/${c.moment}.mp4`, kind: c.kind }),
+};
+
+test("the HTML export is deterministic, holds no script, and every image is a data: URI", () => {
+ const html = reportExportHtml(view, opts);
+ assert.equal(reportExportHtml(view, opts), html);
+ assert.doesNotMatch(html, /<script/i, "a report's raw HTML is escaped, and the export has no script");
+ assert.match(html, /<script>alert\(1\)<\/script>/);
+ assert.doesNotMatch(html, /javascript:/, "a link no reader should follow is left as its label");
+ const srcs = [...html.matchAll(/<img [^>]*src="([^"]+)"/g)].map((m) => m[1]);
+ assert.equal(srcs.length, 2, "the claim's still and the post's screenshot");
+ for (const s of srcs) assert.ok(s.startsWith("data:image/"), s);
+ // No stylesheet, font or script is fetched.
+ assert.doesNotMatch(html, /<link |@import|url\(/);
+});
+
+test("the heading: the series on its own line, the title with the document's byline, then ours", () => {
+ const html = reportExportHtml(view, opts);
+ assert.match(
+ html,
+ /<h1><span class="series" data-report-series="">Demo Checks<\/span><span class="title">A demo fact-check <span class="by" data-report-byline="">by A\. Writer · <a href="https:\/\/example\.org\/a">Example Gazette<\/a><\/span><\/span><\/h1>/,
+ );
+ assert.match(html, /<p class="site-line" data-report-attribution="">Fact-check by Demo Site<\/p>\n<p class="dates">Published 2026-10-01 · Updated 2026-10-04<\/p>\n<p class="subtitle">Two claims, tested<\/p>/);
+ assert.match(html, /<title>Demo Checks: A demo fact-check<\/title>/);
+ assert.doesNotMatch(html, /class="eyebrow"/);
+});
+
+test("the byline: the author alone takes the link when there is no publisher", () => {
+ const html = reportExportHtml({ ...view, sources: { s0: { ...view.sources.s0, publisher: undefined } } }, opts);
+ assert.match(html, /by <a href="https:\/\/example\.org\/a">A\. Writer<\/a>/);
+ assert.doesNotMatch(reportExportHtml({ ...view, subject: undefined }, opts), /data-report-byline/);
+});
+
+test("the three tiers: the quick take, what the check found, every claim from how it was checked", () => {
+ const html = reportExportHtml(view, opts);
+ const order = ['data-report-tier="1"', 'data-verdict-tally', 'data-report-jumps', 'id="found" data-report-tier="2"', 'id="claims" data-report-tier="3"', "data-report-method", 'id="references"'];
+ const at = order.map((m) => html.indexOf(m));
+ assert.ok(at.every((i) => i >= 0), JSON.stringify(at));
+ assert.deepEqual([...at].sort((a, b) => a - b), at, "in the page's order");
+ assert.match(html, /<div class="tier" data-tier-marker="1"><span class="dots" aria-hidden="true"><i class="on"><\/i><i class=""><\/i><i class=""><\/i><\/span><span class="min">1 min<\/span>/);
+ assert.match(html, /<a href="#found">What the check found<\/a> · <a href="#claims">Every claim<\/a> · <a href="#references">References<\/a>/);
+ // Grouped in FOUND_VERDICT_ORDER, each row its title, gist and flag, to the claim.
+ const groups = [...html.matchAll(/data-found-group="([A-Z_]+)"/g)].map((m) => m[1]);
+ assert.deepEqual(groups, ["CONTRADICTED", "PARTLY"]);
+ assert.match(html, /<li data-found-row="c1"><a href="#c1">One<\/a> <span class="gist">— The recording says otherwise\.<\/span> <span class="flag" data-claim-flag=""><svg class="mark"/);
+ assert.match(html, /<h3 class="label">How it was checked<\/h3><p>Each quote was checked against the recording\.<\/p>/);
+ // A sweep has tiers 1 and 3.
+ const sweep = reportExportHtml({ ...view, kind: "sweep" }, opts);
+ assert.doesNotMatch(sweep, /data-report-tier="2"/);
+ assert.match(sweep, /data-report-tier="3"/);
+});
+
+test("a claim: its flag, the subject's sentence on the accent rail with no link, evidence by origin", () => {
+ const html = reportExportHtml(view, opts);
+ const claim = html.slice(html.indexOf('<article class="claim" id="c1"'), html.indexOf("</article>", html.indexOf('id="c1"')));
+ assert.match(claim, /data-verdict="CONTRADICTED"[^]*<span class="flag" data-claim-flag=""><svg class="mark"[^>]*>[^]*<\/svg>No source given<\/span><h3>One<\/h3>/);
+ assert.match(claim, /<figure class="sentence" data-source-sentence="s1" style="border-left-color:#aa3300">/);
+ assert.doesNotMatch(claim, /from <a href="#source-s0">/, "the rail says whose sentence it is");
+ assert.match(html, /<div class="under-review" style="border-left-color:#aa3300">/);
+ // Added first, marked; the subject's own folded under "In the article".
+ assert.match(claim, /<li data-evidence="v1"><a href="#c-v1">\[1\]<\/a> <span class="added" data-citation-added=""><svg class="mark"[^]*?<\/svg>Not in the article<\/span> Video/);
+ assert.match(claim, /<details class="given" data-subject-evidence=""><summary>In the article \(1\)<\/summary><ul class="evidence"><li data-evidence="a1">/);
+ assert.match(reportExportHtml(view, { ...opts, print: true }), /<details class="given" data-subject-evidence="" open>/, "printed open");
+ // The reference list marks added evidence too.
+ assert.match(html, /<li id="c-v1"[^>]*><p class="quote"><q>one \* two<\/q><\/p><p class="meta"><span class="added"/);
+});
+
+test("verdicts, inline markers [n] to the reference list, and references with every link", () => {
+ const html = reportExportHtml(view, opts);
+ assert.match(html, /data-verdict="CONTRADICTED" style="--v:#[0-9a-f]{6}">/i);
+ assert.match(html, /data-verdict="PARTLY"[^>]*>Half/, "the report's override");
+ assert.match(html, /here<sup class="cite"><a href="#c-v1">\[1\]<\/a><\/sup>/);
+ assert.match(html, /<code>\[code\]\(cite:never\)<\/code>/, "code is not citing");
+ const refs = [...html.matchAll(/<li id="c-([a-z0-9]+)" value="(\d+)"/g)].map((m) => `${m[2]}:${m[1]}`);
+ assert.deepEqual(refs, ["1:v1", "2:s1", "3:w1", "4:a1", "5:p1"]);
+ // A span: the original at its time, the moment page and the clip on the site.
+ assert.match(html, /Original at 52:06: <a href="https:\/\/media\.example\.org\/abc123\?t=3126">/);
+ assert.match(html, /Moment page: <a href="https:\/\/reports\.example\.org\/m\/demo-channel\/abc123\/3126\.00-3151\.00\/">/);
+ assert.match(html, /Clip: <a href="https:\/\/reports\.example\.org\/media\/clips\//);
+ assert.doesNotMatch(html, /<video/, "a clip is linked, not played, in the one-file export");
+ assert.match(html, /Archived: <a href="https:\/\/archive\.example\.org\/p">/);
+ assert.match(html, /quote match 97%/);
+ assert.match(html, /<footer class="export-footer" data-export-footer=""><p>2026-10-04 · report sha256 0123456789ab<\/p><p>Published at <a href="https:\/\/reports\.example\.org\/reports\/demo\/">/);
+});
+
+test("without a site URL nothing links to the site; the pack's clip plays in place", () => {
+ const html = reportExportHtml(view, { ...opts, siteUrl: undefined, siteTitle: undefined, clip: undefined });
+ assert.doesNotMatch(html, /reports\.example\.org/);
+ assert.match(html, /<p class="site-line" data-report-attribution="">Fact-check<\/p>/);
+ const pack = reportExportHtml(view, { ...opts, image: () => "media/x.png", clip: () => ({ href: "media/clips/a.mp4", kind: "video", play: true }) });
+ assert.match(pack, /<video controls preload="none" src="media\/clips\/a\.mp4"><\/video>/);
+ assert.match(pack, /<img src="media\/x\.png"/);
+});
+
+test("the footer line names what is known and leaves the rest out", () => {
+ const sha = "f".repeat(64);
+ assert.equal(reportExportFooterLine({ reportSha256: sha }), "report sha256 ffffffffffff");
+ assert.equal(
+ reportExportFooterLine({ revision: 3, date: "2026-10-05T09:00:00Z", reportSha256: sha, commit: "abcdef1234567890" }),
+ "Revision 3 · 2026-10-05 · report sha256 ffffffffffff · commit abcdef123456",
+ );
+});
+
+test("markdownToHtml: paragraphs, lists, quotes, code, headings under the page's own", () => {
+ const html = markdownToHtml("# Head\n\nA *word* and __strong__.\n\n1. one\n2. two\n\n> quoted\n\n```\n<b>\n```", () => undefined);
+ assert.equal(
+ html,
+ "<h3>Head</h3>\n<p>A <em>word</em> and <strong>strong</strong>.</p>\n<ol><li>one</li><li>two</li></ol>\n" +
+ "<blockquote><p>quoted</p></blockquote>\n<pre><code><b></code></pre>",
+ );
+ assert.equal(markdownToHtml("[site](/reports/x/)", () => undefined, { siteUrl: "https://s.example" }), '<p><a href="https://s.example/reports/x/">site</a></p>');
+ assert.equal(markdownToHtml("[site](/reports/x/)", () => undefined), "<p>site</p>");
+});
+
+test("the Markdown export: the heading lines, `label [n]` citations, numbered references, the footer", () => {
+ const md = reportExportMarkdown(view, opts);
+ assert.equal(reportExportMarkdown(view, opts), md);
+ assert.ok(
+ md.startsWith(
+ "**Demo Checks**\n\n# A demo fact-check\n\nby A. Writer · [Example Gazette](https://example.org/a)\n\n" +
+ "Fact-check by Demo Site · Published 2026-10-01 · Updated 2026-10-04\n\n*Two claims, tested*\n",
+ ),
+ md.slice(0, 300),
+ );
+ assert.match(md, /It starts here \[1\]\./);
+ assert.match(md, /See this \[3\], \*\*bold\*\*/);
+ assert.match(md, /^### One$/m);
+ assert.match(md, /^\*\*Verdict: Contradicted\*\* \*\[No source given\]\*$/m);
+ // The tiers as plain lines; what the check found by verdict; how it was checked.
+ assert.deepEqual([...md.matchAll(/^— (\d+) min —$/gm)].map((m) => m[1]), ["1", "1", "1"]);
+ assert.match(md, /## What the check found\n\n### Contradicted \(1\)\n\n- One — The recording says otherwise\. \*\[No source given\]\*\n/);
+ assert.match(md, /## Every claim, with its evidence\n\n\*\*How it was checked\*\*\n\nEach quote was checked against the recording\.\n/);
+ // Evidence by origin: the added marked in words, the subject's under its own line.
+ assert.match(md, /Evidence:\n\n- \[1\] Not in the article: Video, /);
+ assert.match(md, /In the article:\n\n- \[4\] Audio, /);
+ assert.match(md, /^1\. “one \\\* two” \n Not in the article \n/m);
+ assert.doesNotMatch(reportExportMarkdown({ ...view, kind: "sweep" }, opts), /What the check found/);
+ const refs = [...md.matchAll(/^(\d+)\. “(.*)”/gm)].map((m) => `${m[1]}:${m[2]}`);
+ assert.deepEqual(refs, ["1:one \\* two", "2:four", "3:five", "4:two", "5:three"]);
+ assert.match(md, /Moment page: <https:\/\/reports\.example\.org\/m\/demo-channel\/abc123\/3126\.00-3151\.00\/>/);
+ assert.match(md, /---\n\n2026-10-04 · report sha256 0123456789ab\n\nPublished at <https:\/\/reports\.example\.org\/reports\/demo\/>\n$/);
+});
+
+test("mdText escapes what Markdown would read; citedMarkdownToPlain numbers citations", () => {
+ assert.equal(mdText("# a *b* [c]"), "\\# a \\*b\\* \\[c\\]");
+ assert.equal(citedMarkdownToPlain("x [y](cite:p1) [z](https://e.example)", view), "x y [5] [z](https://e.example)");
+});
diff --git a/common/lib/report/exportHtml.ts b/common/lib/report/exportHtml.ts
@@ -0,0 +1,743 @@
+// A REPORT AS ONE HTML FILE — the export a reader saves and hosts again
+// (publish/reportExports.ts writes it as `report.html`, prints it to
+// `report.pdf`, and packs a variant of it with its media as the evidence
+// pack). Built from the report's page view (./views.ts), the same view the
+// export site renders, so the file says what the page says.
+//
+// ONE FILE, NO NETWORK FOR WHAT IT SHOWS: its own small stylesheet inline, no
+// script, no font or image fetched. Every image (a source's sentence as a
+// still, a post's screenshot) is whatever the caller's `image` resolver
+// answers — a data: URI for the one-file export, a relative path in the pack.
+// Clips are LINKED, never inlined (the pack plays them from its media/).
+//
+// What it holds, as the site's report page has it: the heading (the series
+// on its own line in the accent, the title with the reviewed document's
+// byline inline, "Fact-check by <site>", the dates, the subtitle, the
+// document under review on its colour's edge), then the three tiers, each
+// opened by a marker (depth dots, minutes to read): the quick take (the
+// tally, the summary, jump links); what the check found (a fact-check's
+// claims by verdict, a line each with its gist and flag); and every claim
+// with its evidence, opening with how it was checked — each claim its
+// verdict and flag pill, the document's sentence on its rail, the findings
+// with numbered markers, and the evidence: what the report added first
+// (the project's mark, "Not in the article"), then the rest, then what the
+// document gave itself under "In the article (n)" (a <details>, printed
+// open). Then the documents quoted with their archive links, the numbered
+// references (quote, speaker, date, record, the original at its time, the
+// moment page and clip on the site when it has a public URL), and the
+// footer naming exactly which document this is (ReportExportFooter).
+//
+// PURE: no I/O, deterministic for its input. Imports only pure helpers.
+
+import { citationAnchor, CITE_SCHEME } from "../citations/inline";
+import type { SourceArchive } from "../citations/schema";
+import { ICON_PALETTES, markSvg } from "../brand";
+import {
+ CITATION_KIND_LABELS,
+ foundGroups,
+ orderedCitations,
+ reportAttribution,
+ reportByline,
+ reportFullTitle,
+ reportPagePath,
+ reportTierMinutes,
+ sourceAnchor,
+ spanLabel,
+ verdictTally,
+ type BylinePart,
+ type CitationView,
+ type ClaimView,
+ type ReportPageView,
+ type SourceView,
+ type SpanCitationView,
+} from "./views";
+
+// ─── The footer: which document this is ───
+
+// What the footer names. `reportSha256` is the sha256 of the report.json the
+// export was made from (hex). `revision` and `commit` are the report's
+// revision and the commit of the corpus it was published from — filled by the
+// revision history (slice RH); until then absent, and an absent part is left
+// out of the line, never shown as unknown.
+export type ReportExportFooter = {
+ revision?: number;
+ // The revision's date (`YYYY-MM-DD` or an ISO date-time).
+ date?: string;
+ reportSha256: string;
+ commit?: string;
+};
+
+// `Revision N · <date> · report sha256 <first 12> · commit <short>`, the
+// unknown parts left out.
+export function reportExportFooterLine(f: ReportExportFooter): string {
+ return [
+ f.revision !== undefined ? `Revision ${f.revision}` : null,
+ dateLabel(f.date) ?? null,
+ `report sha256 ${f.reportSha256.slice(0, 12)}`,
+ f.commit ? `commit ${f.commit.slice(0, 12)}` : null,
+ ]
+ .filter(Boolean)
+ .join(" · ");
+}
+
+// ─── Options ───
+
+export type ExportClip = {
+ href: string;
+ kind: "video" | "audio";
+ // Play it in place (the evidence pack, whose media/ holds it); else a link.
+ play?: boolean;
+};
+
+export type ReportExportOptions = {
+ // The site's public URL (site.json `siteUrl`). Absent: no link to the site,
+ // its moment pages or its clips (a private site has nowhere to link).
+ siteUrl?: string;
+ // The site's title: "Fact-check by <site title>" under the title.
+ siteTitle?: string;
+ footer: ReportExportFooter;
+};
+
+export type ReportExportHtmlOptions = ReportExportOptions & {
+ // A site-root image path the view names (a still, a post's shot) → the
+ // <img src>: a data: URI, or a path relative to the file. Undefined: the
+ // image is left out (a source sentence then shows its quote as text).
+ image: (sitePath: string) => string | undefined;
+ // A span citation's clip, or undefined for none.
+ clip?: (c: SpanCitationView) => ExportClip | undefined;
+ // For print (the PDF): what the page folds away — the evidence the
+ // document under review gave itself — is printed open.
+ print?: boolean;
+};
+
+// ─── Small helpers ───
+
+export function escapeHtml(s: string): string {
+ return s.replace(/&/g, "&").replace(/</g, "<").replace(/>/g, ">").replace(/"/g, """).replace(/'/g, "'");
+}
+
+// `2026-10-04T12:00:00Z` → `2026-10-04`; a partial date as given.
+export function dateLabel(v: string | undefined): string | undefined {
+ if (!v) return undefined;
+ return /^\d{4}-\d{2}-\d{2}T/.test(v) ? v.slice(0, 10) : v;
+}
+
+// A site-root path on the site's public URL, or undefined without one.
+export function siteLink(siteUrl: string | undefined, sitePath: string): string | undefined {
+ if (!siteUrl) return undefined;
+ return `${siteUrl.replace(/\/+$/, "")}${sitePath.startsWith("/") ? "" : "/"}${sitePath}`;
+}
+
+// A link a reader may follow out of a saved file: http(s) or mailto. A
+// fragment stays in the file; a site-root path goes to the site, when known.
+function safeHref(href: string, siteUrl: string | undefined): string | undefined {
+ if (/^(https?:|mailto:)/i.test(href)) return href;
+ if (href.startsWith("#")) return href;
+ if (href.startsWith("/")) return siteLink(siteUrl, href);
+ return undefined;
+}
+
+const plural = (n: number, one: string, many = `${one}s`) => `${n} ${n === 1 ? one : many}`;
+
+// ─── Markdown, the small subset a report writes ───
+
+export type CiteRef = { number?: number; anchor: string };
+
+// A citation marker: `[n]` linking to its reference.
+function citeMarker(ref: CiteRef | undefined, id: string): string {
+ const n = ref?.number !== undefined ? String(ref.number) : "?";
+ return `<sup class="cite"><a href="#${escapeHtml(ref?.anchor ?? citationAnchor(id))}">[${n}]</a></sup>`;
+}
+
+const PH = "\u0000";
+
+// Emphasis over already-escaped text.
+function emphasis(escaped: string): string {
+ return escaped
+ .replace(/\*\*(?=\S)([\s\S]*?\S)\*\*/g, "<strong>$1</strong>")
+ .replace(/(^|[^\w])__(?=\S)([\s\S]*?\S)__(?!\w)/g, "$1<strong>$2</strong>")
+ .replace(/\*(?=\S)([^*]*?\S)\*/g, "<em>$1</em>")
+ .replace(/(^|[^\w])_(?=\S)([^_]*?\S)_(?!\w)/g, "$1<em>$2</em>");
+}
+
+// One paragraph's inline markdown: code spans, links (a `cite:` link is the
+// label and its number), autolinks, emphasis. Raw HTML is escaped.
+function inlineMd(text: string, cite: (id: string) => CiteRef | undefined, siteUrl: string | undefined): string {
+ const held: string[] = [];
+ const hold = (html: string) => `${PH}${held.push(html) - 1}${PH}`;
+ let s = text.replace(/(`+)([^`]|[^`][\s\S]*?[^`])\1(?!`)/g, (_m, _t, code: string) =>
+ hold(`<code>${escapeHtml(code.trim())}</code>`),
+ );
+ s = s.replace(/\[([^\]]*)\]\(\s*([^)\s]+)(?:\s+"[^"]*")?\s*\)/g, (_m, label: string, href: string) => {
+ const shown = emphasis(escapeHtml(label));
+ if (href.startsWith(CITE_SCHEME)) {
+ const id = href.slice(CITE_SCHEME.length).trim();
+ return hold(`${shown}${citeMarker(cite(id), id)}`);
+ }
+ const safe = safeHref(href, siteUrl);
+ return hold(safe ? `<a href="${escapeHtml(safe)}">${shown}</a>` : shown);
+ });
+ s = s.replace(/<(https?:\/\/[^\s<>]+)>/g, (_m, url: string) => hold(`<a href="${escapeHtml(url)}">${escapeHtml(url)}</a>`));
+ s = emphasis(escapeHtml(s))
+ .replace(/ {2,}\n/g, "<br>")
+ .replace(/\n/g, " ");
+ return s.replace(new RegExp(`${PH}(\\d+)${PH}`, "g"), (_m, i: string) => held[Number(i)]);
+}
+
+const LIST_RE = /^ {0,3}([-*+]|\d{1,9}[.)])\s+(.*)$/;
+const FENCE_RE = /^ {0,3}(`{3,}|~{3,})/;
+const HEADING_RE = /^ {0,3}(#{1,6})\s+(.*?)\s*#*\s*$/;
+const HR_RE = /^ {0,3}([-*_])(\s*\1){2,}\s*$/;
+const QUOTE_RE = /^ {0,3}>/;
+
+const isBlank = (l: string) => /^\s*$/.test(l);
+const startsBlock = (l: string) => FENCE_RE.test(l) || HEADING_RE.test(l) || HR_RE.test(l) || QUOTE_RE.test(l) || LIST_RE.test(l);
+
+// A report's markdown (a summary, a section's body, a claim's findings) as
+// HTML: paragraphs, headings (shifted under the page's own), lists, quotes,
+// code, rules. `headingBase` is the level a `#` becomes.
+export function markdownToHtml(
+ md: string,
+ cite: (id: string) => CiteRef | undefined,
+ opts: { siteUrl?: string; headingBase?: number } = {},
+): string {
+ const lines = md.replace(/\r\n?/g, "\n").split("\n");
+ const base = opts.headingBase ?? 3;
+ const inline = (t: string) => inlineMd(t, cite, opts.siteUrl);
+ const out: string[] = [];
+ let i = 0;
+ while (i < lines.length) {
+ const line = lines[i];
+ if (isBlank(line)) {
+ i++;
+ continue;
+ }
+ const fence = FENCE_RE.exec(line);
+ if (fence) {
+ const close = new RegExp(`^ {0,3}${fence[1][0] === "`" ? "`" : "~"}{${fence[1].length},}\\s*$`);
+ const body: string[] = [];
+ i++;
+ while (i < lines.length && !close.test(lines[i])) body.push(lines[i++]);
+ i++;
+ out.push(`<pre><code>${escapeHtml(body.join("\n"))}</code></pre>`);
+ continue;
+ }
+ const h = HEADING_RE.exec(line);
+ if (h) {
+ const level = Math.min(6, base - 1 + h[1].length);
+ out.push(`<h${level}>${inline(h[2])}</h${level}>`);
+ i++;
+ continue;
+ }
+ if (HR_RE.test(line)) {
+ out.push("<hr>");
+ i++;
+ continue;
+ }
+ if (QUOTE_RE.test(line)) {
+ const inner: string[] = [];
+ while (i < lines.length && QUOTE_RE.test(lines[i])) inner.push(lines[i++].replace(/^ {0,3}> ?/, ""));
+ out.push(`<blockquote>${markdownToHtml(inner.join("\n"), cite, opts)}</blockquote>`);
+ continue;
+ }
+ const first = LIST_RE.exec(line);
+ if (first) {
+ const ordered = /\d/.test(first[1]);
+ const items: string[] = [];
+ while (i < lines.length) {
+ const m = LIST_RE.exec(lines[i]);
+ if (m && /\d/.test(m[1]) === ordered) {
+ items.push(m[2]);
+ i++;
+ } else if (items.length > 0 && !isBlank(lines[i]) && /^\s{2,}\S/.test(lines[i])) {
+ items[items.length - 1] += `\n${lines[i++].trim()}`;
+ } else break;
+ }
+ const start = ordered ? Number.parseInt(first[1], 10) : 1;
+ const tag = ordered ? "ol" : "ul";
+ const attr = ordered && start !== 1 ? ` start="${start}"` : "";
+ out.push(`<${tag}${attr}>${items.map((it) => `<li>${inline(it)}</li>`).join("")}</${tag}>`);
+ continue;
+ }
+ const para: string[] = [line];
+ i++;
+ while (i < lines.length && !isBlank(lines[i]) && !startsBlock(lines[i])) para.push(lines[i++]);
+ out.push(`<p>${inline(para.join("\n").trim())}</p>`);
+ }
+ return out.join("\n");
+}
+
+// ─── The stylesheet ───
+
+// Small and self-contained: system fonts, a light page that prints as it
+// reads. The verdict chip takes its colour from `--v` (the view's verdict
+// styles, report overrides applied), as the site's VerdictChip does.
+export const REPORT_EXPORT_CSS = `
+:root{--fg:#1b1b1b;--muted:#5d5d5d;--border:#d8d8d8;--surface:#f6f6f4;--accent:#1f4e8c;color-scheme:light}
+*{box-sizing:border-box}
+html{-webkit-text-size-adjust:100%}
+body{margin:0;background:#fff;color:var(--fg);font:16px/1.55 -apple-system,BlinkMacSystemFont,"Segoe UI",Roboto,"Helvetica Neue",Arial,sans-serif}
+main{max-width:46rem;margin:0 auto;padding:2rem 1rem 3rem}
+a{color:var(--accent);text-decoration:underline;text-decoration-thickness:1px;text-underline-offset:2px;overflow-wrap:anywhere}
+h1,h2,h3,h4{line-height:1.25;margin:0}
+h1{font-size:2rem;font-weight:600;letter-spacing:-.01em}
+h1 .series{display:block;color:var(--accent);font-weight:600;font-size:.8em;margin-bottom:.15rem}
+h1 .series+.title{display:block;font-weight:400}
+h2{font-size:1.5rem;font-weight:600;margin-top:2.5rem;padding-top:1rem;border-top:1px solid var(--border)}
+h3{font-size:1.15rem;font-weight:600}
+h4,h5,h6{font-size:1rem;font-weight:600;margin:1rem 0 .25rem}
+p{margin:.6rem 0}
+q{quotes:"\\201C" "\\201D"}
+blockquote{margin:.75rem 0;padding-left:.9rem;border-left:3px solid var(--border);color:var(--muted)}
+code{font-family:ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;font-size:.88em;background:var(--surface);padding:.05em .3em;border-radius:3px}
+pre{background:var(--surface);padding:.75rem;border-radius:4px;overflow-x:auto}
+pre code{background:none;padding:0}
+img{max-width:100%;height:auto}
+video,audio{display:block;width:100%;max-width:36rem;margin:.5rem 0}
+.subtitle{font-size:1.15rem;color:var(--muted);margin:.5rem 0 0}
+.site-line,.dates{margin:.35rem 0 0}
+h1 .by{font-size:.55em;font-weight:400;color:var(--muted);letter-spacing:0}
+.site-line,.dates,.label,.meta,.export-footer{font-size:.85rem;color:var(--muted)}
+.label{font:600 .7rem/1.2 ui-monospace,SFMono-Regular,Menlo,monospace;text-transform:uppercase;letter-spacing:.14em;margin:1.25rem 0 .4rem}
+header.report-head{padding-bottom:1.25rem;border-bottom:1px solid var(--border)}
+.under-review{margin-top:1rem;padding:.75rem 1rem;border:1px solid var(--border);border-left-width:4px;border-radius:6px;background:var(--surface);font-size:.95rem}
+.tier{display:flex;align-items:center;gap:.6rem;margin:2.25rem 0 .5rem}
+.tier .dots{display:inline-flex;gap:.25rem}
+.tier .dots i{display:inline-block;width:.4rem;height:.4rem;border-radius:50%;border:1px solid var(--muted)}
+.tier .dots i.on{background:var(--accent);border-color:var(--accent)}
+.tier .min{font:.75rem/1 ui-monospace,SFMono-Regular,Menlo,monospace;color:var(--muted)}
+.tier .rule{flex:1;height:1px;background:var(--border)}
+.tier+h2{margin-top:.5rem;padding-top:0;border-top:0}
+h2.section{font-size:1.3rem}
+.jumps{font-size:.92rem}
+.found h3{margin:1rem 0 .4rem}
+.found ul{margin:0;padding-left:1.1rem;font-size:.95rem}
+.found li{margin:.3rem 0}
+.found li>a{color:var(--fg);font-weight:600;text-decoration:none}
+.gist{color:var(--muted)}
+.method{margin:.75rem 0}
+.flag,.added{display:inline-flex;align-items:center;gap:.35rem;white-space:nowrap}
+.flag{border:1px solid color-mix(in srgb,var(--accent) 40%,#fff);background:color-mix(in srgb,var(--accent) 8%,#fff);color:var(--accent);border-radius:999px;padding:.05rem .6rem;font-size:.78rem;font-weight:500}
+.added{font:.7rem/1.2 ui-monospace,SFMono-Regular,Menlo,monospace;text-transform:uppercase;letter-spacing:.12em;color:var(--muted)}
+svg.mark{flex:none;border-radius:22%}
+details.given{margin:.4rem 0 0}
+details.given summary{cursor:pointer;font:.8rem/1.4 ui-monospace,SFMono-Regular,Menlo,monospace;color:var(--muted)}
+.source{padding-left:.75rem;border-left:4px solid var(--border)}
+.tally{display:flex;flex-wrap:wrap;gap:.4rem;list-style:none;padding:0;margin:.4rem 0 0}
+.verdict{display:inline-flex;align-items:center;gap:.4rem;border:1px solid var(--v);border-radius:999px;padding:.1rem .65rem;font-size:.8rem;font-weight:600;background:color-mix(in srgb,var(--v) 14%,#fff);white-space:nowrap}
+.verdict::before{content:"";width:.5rem;height:.5rem;border-radius:50%;background:var(--v)}
+.verdict .n{font-family:ui-monospace,SFMono-Regular,Menlo,monospace;color:var(--muted);font-weight:400}
+nav.toc ol{margin:.5rem 0;padding-left:1.4rem}
+.claim{margin:1.25rem 0;padding:1rem 1.1rem;border:1px solid var(--border);border-radius:8px}
+.claim-head{display:flex;flex-wrap:wrap;align-items:baseline;gap:.4rem .7rem}
+.claim-head h3{flex:1 1 15rem}
+.says{color:var(--muted);font-size:.95rem}
+.says q{color:var(--fg)}
+figure.sentence{margin:.75rem 0;padding-left:.75rem;border-left:4px solid var(--border)}
+figure.sentence img{display:block;border:1px solid var(--border);border-radius:4px;background:#fff;break-inside:avoid}
+figure.sentence figcaption{font-size:.85rem;color:var(--muted);margin-top:.3rem}
+.evidence{margin:.4rem 0 0;padding-left:1.2rem;font-size:.92rem}
+.evidence li{margin:.2rem 0}
+sup.cite{font-size:.72em;line-height:0;margin-left:.1em}
+sup.cite a{text-decoration:none;font-weight:600}
+.archives{margin:.3rem 0 0;padding-left:1.2rem;font-size:.85rem}
+.archives li{margin:.15rem 0}
+.archives .ctx{color:var(--muted)}
+.source{margin:1rem 0}
+ol.references{padding-left:2.2rem}
+ol.references>li{margin:0 0 1rem;padding-left:.2rem}
+ol.references .quote{margin:0;font-size:1rem}
+ol.references .meta,ol.references .links{margin:.15rem 0;font-size:.85rem}
+ol.references img{display:block;max-width:24rem;border:1px solid var(--border);border-radius:4px;margin:.4rem 0;break-inside:avoid}
+:target{background:#fff6d6}
+.export-footer{margin-top:3rem;padding-top:1rem;border-top:1px solid var(--border)}
+.export-footer p{margin:.25rem 0}
+@media print{
+ @page{size:A4;margin:16mm 14mm}
+ body{font-size:11pt}
+ main{max-width:none;padding:0}
+ a{color:var(--fg)}
+ h2{break-after:avoid}
+ .claim{break-inside:auto}
+ :target{background:none}
+}
+`.trim();
+
+// ─── The parts ───
+
+function verdictChip(view: ReportPageView, verdict: NonNullable<ClaimView["verdict"]>, count?: number): string {
+ const s = view.verdicts[verdict];
+ const n = count !== undefined ? ` <span class="n">${count}</span>` : "";
+ return `<span class="verdict" data-verdict="${escapeHtml(verdict)}" style="--v:${escapeHtml(s.color)}">${escapeHtml(s.label)}${n}</span>`;
+}
+
+function archiveList(archives: readonly SourceArchive[]): string {
+ if (archives.length === 0) return "";
+ return `<ul class="archives">${archives
+ .map(
+ (a) =>
+ `<li><a href="${escapeHtml(a.url)}">${escapeHtml(a.label)}</a> <span class="ctx">${escapeHtml(a.url)}</span>` +
+ `${a.context ? ` <span class="ctx">— ${escapeHtml(a.context)}</span>` : ""}</li>`,
+ )
+ .join("")}</ul>`;
+}
+
+function sourceByline(s: SourceView): string {
+ return [s.publisher, s.author, dateLabel(s.date)].filter(Boolean).map((x) => escapeHtml(x!)).join(" · ");
+}
+
+function sourceTitleHtml(s: SourceView): string {
+ return s.url ? `<a href="${escapeHtml(s.url)}">${escapeHtml(s.title)}</a>` : escapeHtml(s.title);
+}
+
+// The citation's record or document, as one line: the title, then where.
+function citationSourceLine(c: CitationView): string {
+ switch (c.kind) {
+ case "video":
+ case "audio": {
+ const r = c.record;
+ return [r.title ? `<cite>${escapeHtml(r.title)}</cite>` : null, r.channelTitle ? escapeHtml(r.channelTitle) : null, spanLabel(c.start, c.end)]
+ .filter(Boolean)
+ .join(" · ");
+ }
+ case "post":
+ return [c.author ? escapeHtml(c.author) : null, c.record.channelTitle ? escapeHtml(c.record.channelTitle) : null, "post"]
+ .filter(Boolean)
+ .join(" · ");
+ case "source":
+ return `from <a href="#${escapeHtml(sourceAnchor(c.sourceId))}"><cite>${escapeHtml(c.sourceTitle)}</cite></a>`;
+ case "page":
+ return c.title ? `<cite>${escapeHtml(c.title)}</cite>` : "web page";
+ }
+}
+
+function citationDate(c: CitationView): string | undefined {
+ return dateLabel(c.date ?? (c.kind === "video" || c.kind === "audio" || c.kind === "post" ? c.record.date : undefined));
+}
+
+function link(href: string, text?: string): string {
+ return `<a href="${escapeHtml(href)}">${escapeHtml(text ?? href)}</a>`;
+}
+
+// Evidence the report's author found (`origin: "added"`): the project's mark
+// (the site's AddedMark, as a static SVG) and the words.
+const ADDED_MARK = markSvg(ICON_PALETTES.archilyzer).replace(
+ "<svg ",
+ '<svg class="mark" width="12" height="12" aria-hidden="true" focusable="false" ',
+);
+
+function addedTag(label: string): string {
+ return `<span class="added" data-citation-added="">${ADDED_MARK}${escapeHtml(label)}</span>`;
+}
+
+function flagPill(flag: string): string {
+ return `<span class="flag" data-claim-flag="">${ADDED_MARK}${escapeHtml(flag)}</span>`;
+}
+
+function reference(c: CitationView, opts: ReportExportHtmlOptions, addedLabel: string): string {
+ const meta = [
+ `<span class="kind">${CITATION_KIND_LABELS[c.kind]}</span>`,
+ c.speaker ? escapeHtml(c.speaker) : null,
+ citationDate(c) ? escapeHtml(citationDate(c)!) : null,
+ citationSourceLine(c),
+ c.verification?.quoteScore !== undefined ? `quote match ${Math.round(c.verification.quoteScore * 100)}%` : null,
+ ].filter(Boolean);
+ const links: string[] = [];
+ let media = "";
+ switch (c.kind) {
+ case "video":
+ case "audio": {
+ if (c.record.originalUrl) links.push(`Original at ${spanLabel(c.start, c.end).split("–")[0]}: ${link(c.record.originalUrl)}`);
+ const moment = siteLink(opts.siteUrl, c.href);
+ if (moment) links.push(`Moment page: ${link(moment)}`);
+ const clip = opts.clip?.(c);
+ if (clip) {
+ links.push(`Clip: ${link(clip.href)}`);
+ if (clip.play) {
+ media =
+ clip.kind === "audio"
+ ? `<audio controls preload="none" src="${escapeHtml(clip.href)}"></audio>`
+ : `<video controls preload="none" src="${escapeHtml(clip.href)}"></video>`;
+ }
+ }
+ break;
+ }
+ case "post": {
+ if (c.record.originalUrl) links.push(`Original: ${link(c.record.originalUrl)}`);
+ const moment = siteLink(opts.siteUrl, c.href);
+ if (moment) links.push(`Moment page: ${link(moment)}`);
+ const shot = c.shot ? opts.image(c.shot) : undefined;
+ if (shot) media = `<img src="${escapeHtml(shot)}" alt="${escapeHtml(`Screenshot of the post: ${c.quote}`)}">`;
+ break;
+ }
+ case "source":
+ if (c.href) links.push(`Original: ${link(c.href)}`);
+ break;
+ case "page":
+ links.push(`Original: ${link(c.href)}`);
+ if (c.archiveUrl) links.push(`Archived: ${link(c.archiveUrl)}`);
+ break;
+ }
+ const extra = [c.label, c.note].filter(Boolean).map((x) => `<p class="meta">${escapeHtml(x!)}</p>`).join("");
+ return (
+ `<li id="${escapeHtml(citationAnchor(c.id))}" value="${c.number ?? ""}" data-reference="${escapeHtml(c.id)}">` +
+ `<p class="quote"><q>${escapeHtml(c.quote)}</q></p>` +
+ (c.origin === "added" ? `<p class="meta">${addedTag(addedLabel)}</p>` : "") +
+ `<p class="meta">${meta.join(" · ")}</p>` +
+ extra +
+ (links.length > 0 ? `<p class="links">${links.join("<br>")}</p>` : "") +
+ media +
+ `</li>`
+ );
+}
+
+// One line of a claim's evidence, numbered to its reference.
+function evidenceItem(c: CitationView, addedLabel: string): string {
+ return (
+ `<li data-evidence="${escapeHtml(c.id)}"><a href="#${escapeHtml(citationAnchor(c.id))}">[${c.number ?? "?"}]</a> ` +
+ `${c.origin === "added" ? `${addedTag(addedLabel)} ` : ""}` +
+ `${CITATION_KIND_LABELS[c.kind]} · ${citationSourceLine(c)} — <q>${escapeHtml(c.quote)}</q></li>`
+ );
+}
+
+type ClaimContext = {
+ view: ReportPageView;
+ subjectLabel: string;
+ subjectNoun: string;
+ cite: (id: string) => CiteRef | undefined;
+ opts: ReportExportHtmlOptions;
+};
+
+function claimHtml(claim: ClaimView, x: ClaimContext): string {
+ const { view, opts } = x;
+ const addedLabel = `Not in the ${x.subjectNoun}`;
+ const parts: string[] = [];
+ parts.push(
+ `<div class="claim-head">${claim.verdict ? verdictChip(view, claim.verdict) : ""}` +
+ `${claim.flag ? flagPill(claim.flag) : ""}<h3>${escapeHtml(claim.title ?? claim.text)}</h3></div>`,
+ );
+ if (claim.title) parts.push(`<p class="says">${escapeHtml(x.subjectLabel)}: <q>${escapeHtml(claim.text)}</q></p>`);
+ const sentence = claim.sourceQuote ? view.citations[claim.sourceQuote] : undefined;
+ if (sentence?.kind === "source") {
+ // On a rail in the document's colour; a sentence of the document under
+ // review needs no link to it (the rail says whose it is).
+ const isSubject = sentence.sourceId === view.subject;
+ const src = sentence.image ? opts.image(sentence.image) : undefined;
+ const shown = src
+ ? `<img src="${escapeHtml(src)}" alt="${escapeHtml(sentence.quote)}">`
+ : `<blockquote><q>${escapeHtml(sentence.quote)}</q></blockquote>`;
+ const caption = [
+ isSubject
+ ? citeMarker(x.cite(sentence.id), sentence.id)
+ : `from <a href="#${escapeHtml(sourceAnchor(sentence.sourceId))}">${escapeHtml(sentence.sourceTitle)}</a>${citeMarker(x.cite(sentence.id), sentence.id)}`,
+ claim.archives ? archiveList(claim.archives) : "",
+ ].join("");
+ parts.push(
+ `<figure class="sentence" data-source-sentence="${escapeHtml(sentence.id)}"${railStyle(sentence.sourceAccent)}>${shown}` +
+ `<figcaption>${caption}</figcaption></figure>`,
+ );
+ }
+ if (claim.findings) parts.push(`<div class="findings">${markdownToHtml(claim.findings, x.cite, { siteUrl: opts.siteUrl, headingBase: 4 })}</div>`);
+ const evidence = claim.citations
+ .filter((id) => id !== claim.sourceQuote)
+ .map((id) => view.citations[id])
+ .filter((c): c is CitationView => !!c);
+ // What the report adds first, marked; then evidence of unknown origin; what
+ // the document under review gave itself folded away at the end (printed
+ // open).
+ const shown = [...evidence.filter((c) => c.origin === "added"), ...evidence.filter((c) => c.origin === undefined)];
+ const given = evidence.filter((c) => c.origin === "subject");
+ if (evidence.length > 0) {
+ parts.push(`<p class="label">Evidence</p>`);
+ if (shown.length > 0) parts.push(`<ul class="evidence">${shown.map((c) => evidenceItem(c, addedLabel)).join("")}</ul>`);
+ if (given.length > 0) {
+ parts.push(
+ `<details class="given" data-subject-evidence=""${opts.print ? " open" : ""}>` +
+ `<summary>${escapeHtml(`In the ${x.subjectNoun} (${given.length})`)}</summary>` +
+ `<ul class="evidence">${given.map((c) => evidenceItem(c, addedLabel)).join("")}</ul></details>`,
+ );
+ }
+ }
+ return `<article class="claim" id="${escapeHtml(claim.id)}" data-claim="${escapeHtml(claim.id)}">${parts.join("\n")}</article>`;
+}
+
+// A left rail in a document's colour (its `accent`, `#rrggbb` by validation),
+// else the border colour the stylesheet gives.
+function railStyle(accent: string | undefined): string {
+ return accent && /^#[0-9a-fA-F]{6}$/.test(accent) ? ` style="border-left-color:${accent}"` : "";
+}
+
+// Where a tier of the page begins: a hairline with, at its left, how deep it
+// goes (one, two or three dots) and how long it takes to read.
+function tierMarker(depth: 1 | 2 | 3, minutes: number): string {
+ const dots = [1, 2, 3].map((i) => `<i class="${i <= depth ? "on" : ""}"></i>`).join("");
+ return (
+ `<div class="tier" data-tier-marker="${depth}"><span class="dots" aria-hidden="true">${dots}</span>` +
+ `<span class="min">${Math.max(1, minutes)} min</span><span class="rule" aria-hidden="true"></span></div>`
+ );
+}
+
+// ─── The document ───
+
+// `Published <date>`, `Updated <date>` (when it differs).
+export function reportDates(view: Pick<ReportPageView, "published" | "updated">): string[] {
+ return [
+ view.published ? `Published ${dateLabel(view.published)}` : null,
+ view.updated && view.updated !== view.published ? `Updated ${dateLabel(view.updated)}` : null,
+ ].filter((x): x is string => !!x);
+}
+
+function bylinePart(p: BylinePart): string {
+ return p.href ? link(p.href, p.text) : escapeHtml(p.text);
+}
+
+export function reportExportHtml(view: ReportPageView, opts: ReportExportHtmlOptions): string {
+ const isFactcheck = view.kind === "factcheck";
+ const cite = (id: string): CiteRef | undefined => {
+ const c = view.citations[id];
+ return c ? { number: c.number, anchor: citationAnchor(id) } : undefined;
+ };
+ const md = (text: string, headingBase = 3) => markdownToHtml(text, cite, { siteUrl: opts.siteUrl, headingBase });
+ const subject = view.subject ? view.sources[view.subject] : undefined;
+ const subjectNoun = subject?.kind === "article" ? "article" : "source";
+ const subjectLabel = `The ${subjectNoun} says`;
+ const tally = isFactcheck ? verdictTally(view) : [];
+ const groups = isFactcheck ? foundGroups(view) : [];
+ const minutes = reportTierMinutes(view);
+ const claimCount = view.sections.reduce((n, s) => n + s.claims.length, 0);
+ const pageUrl = siteLink(opts.siteUrl, reportPagePath(view.id));
+ const ctx: ClaimContext = { view, subjectLabel, subjectNoun, cite, opts };
+
+ // The heading, in the site's order: the series (its own line, the accent),
+ // the title with the reviewed document's byline inline, then ours — what
+ // the report is and who publishes it, its dates — then the subtitle and the
+ // document under review.
+ const head: string[] = [];
+ const by = reportByline(view);
+ const byHtml = by
+ ? ` <span class="by" data-report-byline="">by ${[by.author ? bylinePart(by.author) : null, by.publisher ? bylinePart(by.publisher) : null]
+ .filter(Boolean)
+ .join(" · ")}</span>`
+ : "";
+ head.push(
+ `<h1>${view.series ? `<span class="series" data-report-series="">${escapeHtml(view.series)}</span>` : ""}` +
+ `<span class="title">${escapeHtml(view.title)}${byHtml}</span></h1>`,
+ );
+ head.push(`<p class="site-line" data-report-attribution="">${escapeHtml(reportAttribution(view.kind, opts.siteTitle))}</p>`);
+ const dates = reportDates(view);
+ if (dates.length > 0) head.push(`<p class="dates">${dates.map(escapeHtml).join(" · ")}</p>`);
+ if (view.subtitle) head.push(`<p class="subtitle">${escapeHtml(view.subtitle)}</p>`);
+ if (subject) {
+ const sb = sourceByline(subject);
+ head.push(
+ `<div class="under-review"${railStyle(subject.accent)}><p class="label" style="margin-top:0">Under review</p>` +
+ `<p style="margin:0">${sourceTitleHtml(subject)}</p>${sb ? `<p class="meta" style="margin:.2rem 0 0">${sb}</p>` : ""}` +
+ `${subject.archives.length > 0 ? `<p class="meta" style="margin:.2rem 0 0"><a href="#${escapeHtml(sourceAnchor(subject.id))}">${plural(subject.archives.length, "archive link")}</a></p>` : ""}</div>`,
+ );
+ }
+ const body: string[] = [`<header class="report-head">${head.join("\n")}</header>`];
+
+ // Tier 1, the quick take: the tally, the summary, where to go next.
+ const quick: string[] = [tierMarker(1, minutes.quick)];
+ if (tally.length > 0) {
+ quick.push(
+ `<p class="label">${plural(claimCount, "claim")} checked</p>` +
+ `<ul class="tally" data-verdict-tally="">${tally.map((t) => `<li>${verdictChip(view, t.verdict, t.count)}</li>`).join("")}</ul>`,
+ );
+ }
+ if (view.summary) quick.push(`<div class="summary">${md(view.summary)}</div>`);
+ const jumps = [
+ ...(groups.length > 0 ? [["#found", "What the check found"]] : []),
+ ["#claims", "Every claim"],
+ ...(orderedCitations(view).length > 0 ? [["#references", "References"]] : []),
+ ];
+ quick.push(`<p class="jumps" data-report-jumps="">${jumps.map(([h, l]) => `<a href="${h}">${l}</a>`).join(" · ")}</p>`);
+ body.push(`<section class="tier-1" data-report-tier="1" aria-label="In brief">${quick.join("\n")}</section>`);
+
+ // Tier 2, what the check found: every ruled claim by verdict, one line each.
+ if (groups.length > 0) {
+ body.push(
+ `<section id="found" data-report-tier="2">${tierMarker(2, minutes.found ?? 0)}<h2>What the check found</h2>` +
+ groups
+ .map(
+ (g) =>
+ `<div class="found" data-found-group="${escapeHtml(g.verdict)}"><h3>${verdictChip(view, g.verdict, g.claims.length)}</h3><ul>` +
+ g.claims
+ .map(
+ (c) =>
+ `<li data-found-row="${escapeHtml(c.id)}"><a href="#${escapeHtml(c.id)}">${escapeHtml(c.title ?? c.text)}</a>` +
+ `${c.gist ? ` <span class="gist">— ${escapeHtml(c.gist)}</span>` : ""}${c.flag ? ` ${flagPill(c.flag)}` : ""}</li>`,
+ )
+ .join("") +
+ `</ul></div>`,
+ )
+ .join("\n") +
+ `</section>`,
+ );
+ }
+
+ // Tier 3, every claim with its evidence, opening with how it was checked.
+ const full: string[] = [tierMarker(3, minutes.claims), `<h2>Every claim, with its evidence</h2>`];
+ if (view.method) full.push(`<div class="method" data-report-method=""><h3 class="label">How it was checked</h3>${md(view.method)}</div>`);
+ if (view.sections.length > 1) {
+ full.push(
+ `<nav class="toc" aria-label="Sections"><ol>${view.sections
+ .map(
+ (s) =>
+ `<li><a href="#${escapeHtml(s.id)}">${escapeHtml(s.title)}</a>` +
+ `${s.claims.length > 0 ? ` <span class="meta">${plural(s.claims.length, "claim")}</span>` : ""}</li>`,
+ )
+ .join("")}</ol></nav>`,
+ );
+ }
+ for (const s of view.sections) {
+ full.push(
+ `<section id="${escapeHtml(s.id)}" data-section="${escapeHtml(s.id)}"><h2 class="section">${escapeHtml(s.title)}</h2>` +
+ `${s.body ? md(s.body) : ""}${s.claims.map((c) => claimHtml(c, ctx)).join("\n")}</section>`,
+ );
+ }
+ body.push(`<div id="claims" data-report-tier="3">${full.join("\n")}</div>`);
+
+ // Every document the report quotes, the subject first, each once with its
+ // archive links in context.
+ const sources = Object.values(view.sources);
+ if (sources.length > 0) {
+ body.push(
+ `<section id="sources"><h2>Sources</h2>${sources
+ .map((s) => {
+ const sb = sourceByline(s);
+ return (
+ `<div class="source" id="${escapeHtml(sourceAnchor(s.id))}"${railStyle(s.accent)}><p style="margin:0"><strong>${sourceTitleHtml(s)}</strong>` +
+ `${s.id === view.subject ? ` <span class="meta">(under review)</span>` : ""}</p>` +
+ `${s.url ? `<p class="meta" style="margin:.1rem 0">${escapeHtml(s.url)}</p>` : ""}` +
+ `${sb ? `<p class="meta" style="margin:.1rem 0">${sb}</p>` : ""}` +
+ `${s.note ? `<p class="meta" style="margin:.1rem 0">${escapeHtml(s.note)}</p>` : ""}` +
+ `${s.archives.length > 0 ? `<p class="label">${plural(s.archives.length, "archive link")} in context</p>${archiveList(s.archives)}` : ""}</div>`
+ );
+ })
+ .join("\n")}</section>`,
+ );
+ }
+ const refs = orderedCitations(view);
+ if (refs.length > 0) {
+ body.push(
+ `<section id="references"><h2>References</h2><ol class="references" data-reference-list="">${refs
+ .map((c) => reference(c, opts, `Not in the ${subjectNoun}`))
+ .join("\n")}</ol></section>`,
+ );
+ }
+ body.push(
+ `<footer class="export-footer" data-export-footer=""><p>${escapeHtml(reportExportFooterLine(opts.footer))}</p>` +
+ `${pageUrl ? `<p>Published at ${link(pageUrl)}</p>` : ""}</footer>`,
+ );
+
+ return (
+ `<!doctype html>\n<html lang="en">\n<head>\n<meta charset="utf-8">\n` +
+ `<meta name="viewport" content="width=device-width, initial-scale=1">\n` +
+ `<title>${escapeHtml(reportFullTitle(view))}</title>\n` +
+ `${view.subtitle ? `<meta name="description" content="${escapeHtml(view.subtitle)}">\n` : ""}` +
+ `<meta name="generator" content="Archilyzer">\n` +
+ `<style>\n${REPORT_EXPORT_CSS}\n</style>\n</head>\n<body>\n<main data-report="${escapeHtml(view.id)}">\n` +
+ `${body.join("\n")}\n</main>\n</body>\n</html>\n`
+ );
+}
diff --git a/common/lib/report/exportMarkdown.ts b/common/lib/report/exportMarkdown.ts
@@ -0,0 +1,255 @@
+// A REPORT AS PLAIN MARKDOWN — `report.md`, the export that reads anywhere
+// text does (publish/reportExports.ts writes it). Built from the report's page
+// view (./views.ts), like the HTML export (./exportHtml.ts), with the same
+// footer.
+//
+// The report's own markdown (summary, section bodies, findings) passes through
+// as written, each inline citation `[label](cite:<id>)` becoming `label [n]` —
+// the number of its entry in the numbered reference list at the end. Plain
+// text fields (titles, quotes) are escaped so a `*` or `[` in a quote stays a
+// character. No images: a still or a screenshot is in the HTML export and the
+// evidence pack; here the quote is the text.
+//
+// PURE: no I/O, deterministic for its input.
+
+import { CITE_SCHEME } from "../citations/inline";
+import type { SourceArchive } from "../citations/schema";
+import { dateLabel, reportDates, reportExportFooterLine, siteLink, type ReportExportOptions } from "./exportHtml";
+import {
+ CITATION_KIND_LABELS,
+ foundGroups,
+ orderedCitations,
+ reportAttribution,
+ reportByline,
+ reportPagePath,
+ reportTierMinutes,
+ spanLabel,
+ verdictTally,
+ type BylinePart,
+ type CitationView,
+ type ClaimView,
+ type ReportPageView,
+} from "./views";
+
+// Plain text, safe in a Markdown paragraph: the characters Markdown reads as
+// syntax are escaped, and a line can never open a heading, a list or a quote.
+export function mdText(s: string): string {
+ return s
+ .replace(/[\\`*_[\]<>|]/g, (c) => `\\${c}`)
+ .replace(/\s*\n\s*/g, " ")
+ .replace(/^([#>+-]|\d+[.)])/, "\\$1");
+}
+
+// A URL as a Markdown link target: spaces and parentheses escaped.
+function mdUrl(u: string): string {
+ return u.replace(/ /g, "%20").replace(/\(/g, "%28").replace(/\)/g, "%29");
+}
+
+const mdLink = (text: string, url: string) => `[${mdText(text)}](${mdUrl(url)})`;
+
+// The report's markdown with each inline citation as `label [n]`. A site-root
+// link goes to the site when its URL is known, else stays its label.
+export function citedMarkdownToPlain(md: string, view: Pick<ReportPageView, "citations">, siteUrl?: string): string {
+ return md.replace(/\[([^\]]*)\]\(\s*([^)\s]+)(\s+"[^"]*")?\s*\)/g, (m, label: string, href: string, title?: string) => {
+ if (href.startsWith(CITE_SCHEME)) {
+ const c = view.citations[href.slice(CITE_SCHEME.length).trim()];
+ return `${label} [${c?.number ?? "?"}]`;
+ }
+ if (href.startsWith("/")) {
+ const abs = siteLink(siteUrl, href);
+ return abs ? `[${label}](${mdUrl(abs)}${title ?? ""})` : label;
+ }
+ return m;
+ });
+}
+
+function archiveLines(archives: readonly SourceArchive[]): string[] {
+ return archives.map((a) => `- ${mdLink(a.label, a.url)} <${a.url}>${a.context ? ` — ${mdText(a.context)}` : ""}`);
+}
+
+function citationWhere(c: CitationView): string {
+ switch (c.kind) {
+ case "video":
+ case "audio":
+ return [c.record.title ? `*${mdText(c.record.title)}*` : null, c.record.channelTitle ? mdText(c.record.channelTitle) : null, spanLabel(c.start, c.end)]
+ .filter(Boolean)
+ .join(", ");
+ case "post":
+ return [c.author ? mdText(c.author) : null, c.record.channelTitle ? mdText(c.record.channelTitle) : null, "post"].filter(Boolean).join(", ");
+ case "source":
+ return `from *${mdText(c.sourceTitle)}*`;
+ case "page":
+ return c.title ? `*${mdText(c.title)}*` : "web page";
+ }
+}
+
+function citationDate(c: CitationView): string | undefined {
+ return dateLabel(c.date ?? (c.kind === "video" || c.kind === "audio" || c.kind === "post" ? c.record.date : undefined));
+}
+
+function referenceLines(c: CitationView, siteUrl: string | undefined, addedLabel: string): string[] {
+ const meta = [
+ CITATION_KIND_LABELS[c.kind],
+ c.speaker ? mdText(c.speaker) : null,
+ citationDate(c) ?? null,
+ citationWhere(c),
+ c.verification?.quoteScore !== undefined ? `quote match ${Math.round(c.verification.quoteScore * 100)}%` : null,
+ ].filter(Boolean);
+ const out = [`${c.number ?? "-"}. “${mdText(c.quote)}”`];
+ if (c.origin === "added") out.push(` ${addedLabel}`);
+ out.push(` ${meta.join(" · ")}`);
+ for (const x of [c.label, c.note]) if (x) out.push(` ${mdText(x)}`);
+ const link = (label: string, url: string) => out.push(` ${label}: <${url}>`);
+ switch (c.kind) {
+ case "video":
+ case "audio": {
+ if (c.record.originalUrl) link(`Original at ${spanLabel(c.start, c.end).split("–")[0]}`, c.record.originalUrl);
+ const moment = siteLink(siteUrl, c.href);
+ if (moment) link("Moment page", moment);
+ break;
+ }
+ case "post": {
+ if (c.record.originalUrl) link("Original", c.record.originalUrl);
+ const moment = siteLink(siteUrl, c.href);
+ if (moment) link("Moment page", moment);
+ break;
+ }
+ case "source":
+ if (c.href) link("Original", c.href);
+ break;
+ case "page":
+ link("Original", c.href);
+ if (c.archiveUrl) link("Archived", c.archiveUrl);
+ break;
+ }
+ // One list item: its lines indented under the number, each ending in a
+ // hard break but the last.
+ return out.map((l, i, a) => (i < a.length - 1 ? `${l} ` : l));
+}
+
+// A tier's start, as a plain line: `— 6 min —`.
+function tierLine(minutes: number): string {
+ return `— ${Math.max(1, minutes)} min —`;
+}
+
+function evidenceLine(c: CitationView, addedLabel: string): string {
+ return `- [${c.number ?? "?"}] ${c.origin === "added" ? `${addedLabel}: ` : ""}${CITATION_KIND_LABELS[c.kind]}, ${citationWhere(c)}: “${mdText(c.quote)}”`;
+}
+
+function claimLines(
+ claim: ClaimView,
+ view: ReportPageView,
+ subjectLabel: string,
+ subjectNoun: string,
+ siteUrl: string | undefined,
+): string[] {
+ const addedLabel = `Not in the ${subjectNoun}`;
+ const out: string[] = [`### ${mdText(claim.title ?? claim.text)}`, ""];
+ const marks = [
+ claim.verdict ? `**Verdict: ${mdText(view.verdicts[claim.verdict].label)}**` : null,
+ claim.flag ? `*[${mdText(claim.flag)}]*` : null,
+ ].filter(Boolean);
+ if (marks.length > 0) out.push(marks.join(" "), "");
+ if (claim.title) out.push(`${subjectLabel}: “${mdText(claim.text)}”`, "");
+ const sentence = claim.sourceQuote ? view.citations[claim.sourceQuote] : undefined;
+ if (sentence?.kind === "source") {
+ out.push(`> “${mdText(sentence.quote)}” [${sentence.number ?? "?"}]`);
+ if (sentence.sourceId !== view.subject) out.push(">", `> — from *${mdText(sentence.sourceTitle)}*`);
+ out.push("");
+ if (claim.archives?.length) out.push(...archiveLines(claim.archives), "");
+ }
+ if (claim.findings) out.push(citedMarkdownToPlain(claim.findings, view, siteUrl).trim(), "");
+ const evidence = claim.citations
+ .filter((id) => id !== claim.sourceQuote)
+ .map((id) => view.citations[id])
+ .filter((c): c is CitationView => !!c);
+ // What the report adds first, then evidence of unknown origin, then what
+ // the document under review gave itself.
+ const shown = [...evidence.filter((c) => c.origin === "added"), ...evidence.filter((c) => c.origin === undefined)];
+ const given = evidence.filter((c) => c.origin === "subject");
+ if (shown.length > 0) out.push("Evidence:", "", ...shown.map((c) => evidenceLine(c, addedLabel)), "");
+ if (given.length > 0) out.push(`In the ${subjectNoun}:`, "", ...given.map((c) => evidenceLine(c, addedLabel)), "");
+ return out;
+}
+
+const bylinePart = (p: BylinePart) => (p.href ? mdLink(p.text, p.href) : mdText(p.text));
+
+export function reportExportMarkdown(view: ReportPageView, opts: ReportExportOptions): string {
+ const out: string[] = [];
+ const subject = view.subject ? view.sources[view.subject] : undefined;
+ const subjectNoun = subject?.kind === "article" ? "article" : "source";
+ const subjectLabel = `The ${subjectNoun} says`;
+ const pageUrl = siteLink(opts.siteUrl, reportPagePath(view.id));
+ const isFactcheck = view.kind === "factcheck";
+ const minutes = reportTierMinutes(view);
+
+ // The series on its own line, the title, the reviewed document's byline;
+ // then ours: what the report is, who publishes it, its dates.
+ if (view.series) out.push(`**${mdText(view.series)}**`, "");
+ out.push(`# ${mdText(view.title)}`, "");
+ const by = reportByline(view);
+ if (by) out.push(`by ${[by.author ? bylinePart(by.author) : null, by.publisher ? bylinePart(by.publisher) : null].filter(Boolean).join(" · ")}`, "");
+ out.push([mdText(reportAttribution(view.kind, opts.siteTitle)), ...reportDates(view)].join(" · "), "");
+ if (view.subtitle) out.push(`*${mdText(view.subtitle)}*`, "");
+ if (subject) {
+ const byline = [subject.publisher, subject.author, dateLabel(subject.date)].filter(Boolean).map((x) => mdText(x!));
+ out.push(`Under review: ${subject.url ? mdLink(subject.title, subject.url) : mdText(subject.title)}${byline.length ? ` — ${byline.join(" · ")}` : ""}`, "");
+ }
+
+ // Tier 1: the quick take.
+ out.push(tierLine(minutes.quick), "");
+ const tally = isFactcheck ? verdictTally(view) : [];
+ if (tally.length > 0) {
+ const claims = view.sections.reduce((n, s) => n + s.claims.length, 0);
+ out.push(
+ `${claims} claim${claims === 1 ? "" : "s"} checked: ${tally.map((t) => `${mdText(view.verdicts[t.verdict].label)} ${t.count}`).join(" · ")}`,
+ "",
+ );
+ }
+ if (view.summary) out.push(citedMarkdownToPlain(view.summary, view, opts.siteUrl).trim(), "");
+
+ // Tier 2: what the check found (a fact-check with verdicts).
+ const groups = isFactcheck ? foundGroups(view) : [];
+ if (groups.length > 0) {
+ out.push(tierLine(minutes.found ?? 0), "", "## What the check found", "");
+ for (const g of groups) {
+ out.push(`### ${mdText(view.verdicts[g.verdict].label)} (${g.claims.length})`, "");
+ for (const c of g.claims) {
+ out.push(`- ${mdText(c.title ?? c.text)}${c.gist ? ` — ${mdText(c.gist)}` : ""}${c.flag ? ` *[${mdText(c.flag)}]*` : ""}`);
+ }
+ out.push("");
+ }
+ }
+
+ // Tier 3: every claim, with its evidence.
+ out.push(tierLine(minutes.claims), "", "## Every claim, with its evidence", "");
+ if (view.method) out.push("**How it was checked**", "", citedMarkdownToPlain(view.method, view, opts.siteUrl).trim(), "");
+ for (const s of view.sections) {
+ out.push(`## ${mdText(s.title)}`, "");
+ if (s.body) out.push(citedMarkdownToPlain(s.body, view, opts.siteUrl).trim(), "");
+ for (const c of s.claims) out.push(...claimLines(c, view, subjectLabel, subjectNoun, opts.siteUrl));
+ }
+
+ const sources = Object.values(view.sources);
+ if (sources.length > 0) {
+ out.push("## Sources", "");
+ for (const s of sources) {
+ out.push(`### ${mdText(s.title)}${s.id === view.subject ? " (under review)" : ""}`, "");
+ if (s.url) out.push(`<${s.url}>`, "");
+ const byline = [s.publisher, s.author, dateLabel(s.date)].filter(Boolean).map((x) => mdText(x!));
+ if (byline.length > 0) out.push(byline.join(" · "), "");
+ if (s.note) out.push(mdText(s.note), "");
+ if (s.archives.length > 0) out.push(`${s.archives.length} archive link${s.archives.length === 1 ? "" : "s"} in context:`, "", ...archiveLines(s.archives), "");
+ }
+ }
+
+ const refs = orderedCitations(view);
+ if (refs.length > 0) {
+ out.push("## References", "");
+ for (const c of refs) out.push(...referenceLines(c, opts.siteUrl, `Not in the ${subjectNoun}`), "");
+ }
+
+ out.push("---", "", reportExportFooterLine(opts.footer));
+ if (pageUrl) out.push("", `Published at <${pageUrl}>`);
+ return `${out.join("\n").replace(/\n{3,}/g, "\n\n").trimEnd()}\n`;
+}
diff --git a/common/lib/report/views.ts b/common/lib/report/views.ts
@@ -88,6 +88,22 @@ export function reportCitationsDownloadPath(reportId: string, format: "json" | "
return `/reports/${reportId}/citations.${format}`;
}
+// A report's exports (publish/reportExports.ts): the whole report as files a
+// reader can save and host again. Their names, and where compose publishes
+// them beside the report's page.
+export const REPORT_EXPORT_FORMATS = ["html", "pdf", "md", "zip"] as const;
+export type ReportExportFormat = (typeof REPORT_EXPORT_FORMATS)[number];
+export const REPORT_EXPORT_FILENAMES: Readonly<Record<ReportExportFormat, string>> = {
+ html: "report.html",
+ pdf: "report.pdf",
+ md: "report.md",
+ zip: "evidence-pack.zip",
+};
+
+export function reportExportDownloadPath(reportId: string, format: ReportExportFormat): string {
+ return `/reports/${reportId}/${REPORT_EXPORT_FILENAMES[format]}`;
+}
+
// A file the report names relative to its own directory (a still,
// `stills/a01.png` — validated by lib/report/validate.ts never to leave it),
// as published.
@@ -261,10 +277,16 @@ export type ReportPageView = {
// is left out.
citations: Record<string, CitationView>;
sections: SectionView[];
- // The report's citations as files, when compose wrote them.
- downloads?: { json?: string; csv?: string };
+ // The report as files, and its citations as data, when compose published
+ // them (each a site-root path).
+ downloads?: ReportDownloads;
};
+// What a report page offers to download: its exports (html, pdf, md, the
+// evidence pack as zip) and its citations (json, csv). Each key is present
+// only when the file is published.
+export type ReportDownloads = Partial<Record<ReportExportFormat | "json" | "csv", string>>;
+
export type VerdictCount = { verdict: Verdict; count: number };
export type ReportIndexEntry = {
@@ -363,7 +385,7 @@ export type ReportViewResolver = {
record: (c: SpanCitation | PostCitation) => RecordView;
poster?: (c: SpanCitation) => string | undefined;
post?: (c: PostCitation) => { author?: string; text?: string; shot?: string } | undefined;
- downloads?: { json?: string; csv?: string };
+ downloads?: ReportDownloads;
};
function sourceView(id: string, s: Source): SourceView {
diff --git a/common/publish/composeReports.test.ts b/common/publish/composeReports.test.ts
@@ -551,3 +551,139 @@ test("a site with no reports ships none of the last site's", async () => {
const corpus = readJson<{ reports?: unknown }>(pub("corpus.json"));
assert.equal(corpus.reports, undefined);
});
+
+// ─── Exports (publish/reportExports.ts) ───
+
+const { exportSiteReports } = await import("./reportExports");
+const { reportExportDir } = await import("./reportExportFiles");
+const { execFileSync } = await import("node:child_process");
+const { createHash } = await import("node:crypto");
+const { truncateSync } = await import("node:fs");
+
+const sha = (file: string) => createHash("sha256").update(readFileSync(file)).digest("hex");
+const fakePrinter = (printed: string[]) => async () => ({
+ print: async (html: string) => {
+ printed.push(html);
+ return Buffer.from("%PDF-1.4 fake");
+ },
+ close: async () => {},
+});
+
+test("reports export: every format from the resolved view, a self-contained HTML, a deterministic pack", async () => {
+ const printed: string[] = [];
+ // `now` is the verification's time, which citations.json carries: the
+ // same report, checked at the same time, packs to the same bytes.
+ const now = () => new Date("2026-10-05T12:00:00Z");
+ const r = await exportSiteReports({ siteId: "cited", paths, now, openPdfPrinter: fakePrinter(printed), onLog: () => {} });
+ assert.deepEqual(r.problems, []);
+ const dir = reportExportDir(paths, "cited", REPORT);
+ assert.deepEqual(readdirSync(dir).sort(), ["evidence-pack.zip", "export.json", "report.html", "report.md", "report.pdf"]);
+ const html = readFileSync(path.join(dir, "report.html"), "utf8");
+ assert.deepEqual(printed, [html], "the PDF is the HTML, printed");
+ assert.doesNotMatch(html, /<script/i);
+ const imgs = [...html.matchAll(/<img [^>]*src="([^"]+)"/g)].map((m) => m[1]);
+ assert.equal(imgs.length, 2, "the claim's still and the post's screenshot");
+ for (const src of imgs) assert.match(src, /^data:image\//);
+ // Verified as compose verifies, the clip linked on the site, never inlined.
+ assert.match(html, /quote match 100%/);
+ assert.match(html, /https:\/\/cited\.example\.test\/media\/clips\/demo-channel\/abc123\/10\.00-20\.00\.mp4/);
+ assert.doesNotMatch(html, /<video/);
+ const md = readFileSync(path.join(dir, "report.md"), "utf8");
+ assert.match(md, /^1\. “He opened the bridge himself\.”/m);
+
+ const manifest = readJson<{ reportSha256: string; files: Record<string, { bytes: number; sha256: string }>; notes: string[] }>(
+ path.join(dir, "export.json"),
+ );
+ assert.equal(manifest.reportSha256, sha(path.join(paths.sitesDir, "cited", "reports", REPORT, "report.json")));
+ assert.deepEqual(Object.keys(manifest.files), ["html", "pdf", "md", "zip"]);
+ assert.equal(manifest.files.html.sha256, sha(path.join(dir, "report.html")));
+ assert.deepEqual(manifest.notes, []);
+ assert.match(html, new RegExp(`report sha256 ${manifest.reportSha256.slice(0, 12)}`));
+
+ // The pack: the HTML playing its own media, the Markdown, the citations.
+ const zip = path.join(dir, "evidence-pack.zip");
+ const names = execFileSync("unzip", ["-Z1", zip], { encoding: "utf8" }).trim().split("\n");
+ assert.deepEqual(names, [
+ `${REPORT}/citations.csv`,
+ `${REPORT}/citations.json`,
+ `${REPORT}/media/clips/${VIDEOS}/abc123/10.00-20.00.mp4`,
+ `${REPORT}/media/clips/${VIDEOS}/def456/5.00-9.00.m4a`,
+ `${REPORT}/media/posts/${SOCIAL}/3kabc/shot.png`,
+ `${REPORT}/media/report/stills/a01.png`,
+ `${REPORT}/report.html`,
+ `${REPORT}/report.md`,
+ ]);
+ const packed = execFileSync("unzip", ["-p", zip, `${REPORT}/report.html`], { encoding: "utf8" });
+ assert.match(packed, /<video controls preload="none" src="media\/clips\/demo-channel\/abc123\/10\.00-20\.00\.mp4">/);
+ assert.match(packed, /<img src="media\/report\/stills\/a01\.png"/);
+
+ // The same report exports to the same bytes.
+ const before = { html: sha(path.join(dir, "report.html")), zip: sha(zip) };
+ await exportSiteReports({ siteId: "cited", paths, now, openPdfPrinter: fakePrinter([]), onLog: () => {} });
+ assert.deepEqual({ html: sha(path.join(dir, "report.html")), zip: sha(zip) }, before);
+});
+
+test("compose publishes the exports beside the report and lists them as downloads", async () => {
+ await compose("cited");
+ for (const f of ["report.html", "report.pdf", "report.md", "evidence-pack.zip"]) {
+ assert.equal(sha(pub("reports", REPORT, f)), sha(path.join(reportExportDir(paths, "cited", REPORT), f)), f);
+ }
+ const view = readJson<{ downloads: Record<string, string> }>(pub("reports", REPORT, "page.json"));
+ assert.deepEqual(view.downloads, {
+ html: `/reports/${REPORT}/report.html`,
+ pdf: `/reports/${REPORT}/report.pdf`,
+ md: `/reports/${REPORT}/report.md`,
+ zip: `/reports/${REPORT}/evidence-pack.zip`,
+ json: `/reports/${REPORT}/citations.json`,
+ csv: `/reports/${REPORT}/citations.csv`,
+ });
+ assert.equal(citedBuildProblem(paths.exportPublicDir), null, "the exports are within reports/, which the audit allows");
+});
+
+test("an export of another version of the report is not published; a pack over the limit stays local", async () => {
+ const dir = reportExportDir(paths, "cited", REPORT);
+ const file = path.join(paths.sitesDir, "cited", "reports", REPORT, "report.json");
+ const saved = readFileSync(file, "utf8");
+ writeFileSync(file, `${saved}\n`);
+ try {
+ await compose("cited");
+ assert.ok(!existsSync(pub("reports", REPORT, "report.html")));
+ const view = readJson<{ downloads: Record<string, string> }>(pub("reports", REPORT, "page.json"));
+ assert.deepEqual(Object.keys(view.downloads), ["json", "csv"]);
+ } finally {
+ writeFileSync(file, saved);
+ }
+ truncateSync(path.join(dir, "evidence-pack.zip"), 25 * 1024 * 1024);
+ await compose("cited");
+ assert.ok(existsSync(pub("reports", REPORT, "report.html")));
+ assert.ok(!existsSync(pub("reports", REPORT, "evidence-pack.zip")));
+ const view = readJson<{ downloads: Record<string, string> }>(pub("reports", REPORT, "page.json"));
+ assert.deepEqual(Object.keys(view.downloads), ["html", "pdf", "md", "json", "csv"]);
+});
+
+test("reports export: no browser skips the PDF with a note; no zip fails the pack, naming it", async () => {
+ const r = await exportSiteReports({
+ siteId: "cited",
+ paths,
+ openPdfPrinter: async () => ({ missing: "Playwright is not available on this host" }),
+ zipBin: path.join(ROOT, "no-such-zip"),
+ onLog: () => {},
+ });
+ const dir = reportExportDir(paths, "cited", REPORT);
+ assert.deepEqual(readdirSync(dir).sort(), ["export.json", "report.html", "report.md"]);
+ assert.deepEqual(r.exported[0].manifest.notes, ["report.pdf skipped: Playwright is not available on this host"]);
+ assert.equal(r.problems.length, 1);
+ assert.equal(r.problems[0].format, "zip");
+ assert.match(r.problems[0].message, /needs the `zip` program/);
+ // Only what was asked for, and a report that is not published is refused.
+ await exportSiteReports({ siteId: "cited", paths, formats: ["md"], onLog: () => {} });
+ assert.deepEqual(readdirSync(dir).sort(), ["export.json", "report.md"]);
+ await assert.rejects(exportSiteReports({ siteId: "cited", paths, reportId: "nope", onLog: () => {} }), /does not publish a report "nope"/);
+ // The CLI: 2 for what does not exist, 1 for a problem.
+ const { main: exportMain } = await import("../bin/reports-export");
+ const quiet = { log: () => {}, error: () => {} };
+ assert.equal(await exportMain({ siteId: "no-such-site" }, quiet), 2);
+ assert.equal(await exportMain({ siteId: "cited", reportId: "nope" }, quiet), 2);
+ assert.equal(await exportMain({ siteId: "cited", formats: ["zip"], zipBin: path.join(ROOT, "no-such-zip") }, quiet), 1);
+ assert.equal(await exportMain({ siteId: "cited", formats: ["html", "md"] }, quiet), 0);
+});
diff --git a/common/publish/composeReports.ts b/common/publish/composeReports.ts
@@ -33,6 +33,11 @@
// an `archilyzer-citations` set;
// never a source's `saved` copy)
// reports/<id>/<still> each cited source still
+// reports/<id>/report.{html,pdf,md}, the report's exports, when
+// evidence-pack.zip `archilyzer reports export` made
+// them from the report as it is now
+// and each is within the publish
+// limit (./reportExportFiles.ts)
// m/index.json, m/<key>/moment.json one moment view per cited moment,
// "cited in" across every report
// media/clips/<channel>/<id>/<s>-<e>.mp4 a span's prepared clip (.m4a for
@@ -86,8 +91,10 @@ import {
momentViewPath,
orderedCitations,
reportCitationsDownloadPath,
+ reportExportDownloadPath,
reportIndexEntry,
reportViewPath,
+ REPORT_EXPORT_FORMATS,
type CitationView,
type CueLineView,
type MomentPageView,
@@ -95,6 +102,7 @@ import {
type RecordView,
type ReportIndexEntry,
type ReportIndexView,
+ type ReportDownloads,
type ReportPageView,
} from "../lib/report/views";
import { evidenceSpan, isAudioOnlyPlatform, type EvidenceSpan } from "../lib/evidenceClip-server";
@@ -106,6 +114,7 @@ import {
siteReportDir,
type ReportMediaEntry,
} from "./reportMedia";
+import { publishableReportExports, type PublishableReportExports } from "./reportExportFiles";
// The public dir's entries this stage owns. Every compose removes them first.
export const REPORT_PUBLIC_ENTRIES: readonly string[] = ["reports", "m", "media"];
@@ -382,25 +391,35 @@ const isFile = async (p: string) => (await stat(p).catch(() => null))?.isFile()
const sameSpan = (a: EvidenceSpan, b: EvidenceSpan) =>
Math.abs(a.from - b.from) < 0.001 && Math.abs(a.to - b.to) < 0.001;
-// ─── The stage ───
+// ─── Resolving: the reports as views, nothing written ───
-export async function composeReports(opts: ComposeReportsOptions): Promise<ComposedReports> {
+export type ResolveSiteReportsOptions = Omit<ComposeReportsOptions, "publicDir"> & {
+ // Each report's downloads, as its view carries them.
+ downloads?: (reportId: string) => ReportDownloads | undefined;
+};
+
+// The site's reports resolved against the corpus — verified, their views and
+// moment pages built, their prepared media matched — and nothing written.
+// compose writes what this answers; publish/reportExports.ts renders the same
+// views as files. Throws ComposeReportsError with every problem.
+export type ResolvedSiteReports = {
+ // The reports (verification computed) and their views, in the site's order.
+ reports: Report[];
+ views: ReportPageView[];
+ index: ReportIndexView;
+ moments: MomentPageView[];
+ // moment key → its prepared media, in `cacheDir`.
+ mediaOf: Map<string, ReportMediaEntry>;
+ cacheDir: string;
+ allowed: ComposeReportsProblem[];
+};
+
+export async function resolveSiteReports(opts: ResolveSiteReportsOptions): Promise<ResolvedSiteReports> {
const { paths, site } = opts;
- const publicDir = opts.publicDir ?? paths.exportPublicDir;
const log = opts.log ?? (() => {});
const now = (opts.now?.() ?? new Date()).toISOString();
const cited = isCitedSite(site);
- // Whatever an earlier compose left. rm removes a link, never its target (a
- // worktree's public/ entries may be links into the primary checkout).
- for (const entry of REPORT_PUBLIC_ENTRIES) {
- await rm(path.join(publicDir, entry), { recursive: true, force: true });
- }
- if ((site.reports ?? []).length === 0) {
- log("[reports] none published.");
- return { reports: [], moments: [], allowed: [] };
- }
-
const problems: ComposeReportsProblem[] = [];
const loaded = await loadSiteReports(paths, site);
for (const p of loaded.problems) {
@@ -627,10 +646,7 @@ export async function composeReports(opts: ComposeReportsOptions): Promise<Compo
const post = postsByChannel.get(c.channel)?.get(c.id);
return post ? { author: postAuthor(post), text: post.text, shot: postShot(c) } : undefined;
},
- downloads: {
- json: reportCitationsDownloadPath(report.id, "json"),
- csv: reportCitationsDownloadPath(report.id, "csv"),
- },
+ downloads: opts.downloads?.(report.id),
}),
);
@@ -708,6 +724,44 @@ export async function composeReports(opts: ComposeReportsOptions): Promise<Compo
});
}
moments.sort((a, b) => (a.key < b.key ? -1 : a.key > b.key ? 1 : 0));
+ return { reports, views, index, moments, mediaOf, cacheDir, allowed };
+}
+
+// ─── The stage ───
+
+export async function composeReports(opts: ComposeReportsOptions): Promise<ComposedReports> {
+ const { paths, site } = opts;
+ const publicDir = opts.publicDir ?? paths.exportPublicDir;
+ const log = opts.log ?? (() => {});
+
+ // Whatever an earlier compose left. rm removes a link, never its target (a
+ // worktree's public/ entries may be links into the primary checkout).
+ for (const entry of REPORT_PUBLIC_ENTRIES) {
+ await rm(path.join(publicDir, entry), { recursive: true, force: true });
+ }
+ if ((site.reports ?? []).length === 0) {
+ log("[reports] none published.");
+ return { reports: [], moments: [], allowed: [] };
+ }
+
+ // Each report's exports (publish/reportExports.ts), made from the report
+ // as it is now and small enough to publish; the citations as files always.
+ const exportsOf = new Map<string, PublishableReportExports>();
+ for (const id of site.reports ?? []) {
+ const found = await publishableReportExports(paths, site.siteId, id);
+ exportsOf.set(id, found);
+ for (const note of found.notes) log(`[reports] ${id}: ${note}`);
+ }
+ const { reports, views, index, moments, mediaOf, cacheDir, allowed } = await resolveSiteReports({
+ ...opts,
+ downloads: (id) => ({
+ ...Object.fromEntries(
+ REPORT_EXPORT_FORMATS.filter((f) => exportsOf.get(id)?.files[f]).map((f) => [f, reportExportDownloadPath(id, f)]),
+ ),
+ json: reportCitationsDownloadPath(id, "json"),
+ csv: reportCitationsDownloadPath(id, "csv"),
+ }),
+ });
// ─── Writing ───
@@ -723,6 +777,11 @@ export async function composeReports(opts: ComposeReportsOptions): Promise<Compo
const rel = (report.citations![c.id] as { image: string }).image;
await copyOut(publicDir, path.join(siteReportDir(paths, site.siteId, report.id), rel), c.image);
}
+ const exported = exportsOf.get(report.id)?.files ?? {};
+ for (const f of REPORT_EXPORT_FORMATS) {
+ const src = exported[f];
+ if (src) await copyOut(publicDir, src, reportExportDownloadPath(report.id, f));
+ }
}
await writeOut(
publicDir,
@@ -753,7 +812,7 @@ export async function composeReports(opts: ComposeReportsOptions): Promise<Compo
// A clip's published path: the moment's (lib/report/views.ts), `.m4a` for a
// clip cut as sound.
-function clipPath(m: SpanMoment, kind: "video" | "audio"): string {
+export function clipPath(m: SpanMoment, kind: "video" | "audio"): string {
const p = evidenceClipPath(m);
return kind === "audio" ? p.replace(/\.mp4$/, ".m4a") : p;
}
diff --git a/common/publish/reportExportFiles.ts b/common/publish/reportExportFiles.ts
@@ -0,0 +1,134 @@
+// WHERE A REPORT'S EXPORTS LIVE, and which of them may be published.
+//
+// `archilyzer reports export` (./reportExports.ts) writes a report's exports
+// to `.export-index/sites/<siteId>/report-exports/<reportId>/` — staging, not
+// served — beside a manifest, `export.json`, naming each file with its size
+// and checksum and the sha256 of the report.json it was made from. Compose
+// (./composeReports.ts) asks `publishableReportExports` which to copy into the
+// site: a file is published only when
+// - the manifest's report sha256 is the report.json's NOW (an export of an
+// earlier version of the report is never shipped beside the new one), and
+// - the file is within the publish limit (lib/builtExport.ts,
+// PUBLISH_MAX_FILE_BYTES) — an evidence pack over it stays local.
+//
+// Kept apart from ./reportExports.ts, which renders the exports from the
+// resolved reports and so imports compose: compose imports this.
+
+import { createHash } from "node:crypto";
+import { readFile, stat } from "node:fs/promises";
+import path from "node:path";
+import { publishFileSizeProblem } from "../lib/builtExport";
+import { readJsonFile } from "../lib/jsonFile-server";
+import type { Paths } from "../lib/paths";
+import type { ReportExportFooter } from "../lib/report/exportHtml";
+import { REPORT_EXPORT_FILENAMES, REPORT_EXPORT_FORMATS, type ReportExportFormat } from "../lib/report/views";
+import { siteIndexDir } from "../lib/site";
+import { siteReportFile } from "./reportMedia";
+
+export const REPORT_EXPORTS_DIRNAME = "report-exports";
+export const REPORT_EXPORT_MANIFEST_FILENAME = "export.json";
+export const REPORT_EXPORT_MANIFEST_FORMAT = "archilyzer-report-export";
+export const REPORT_EXPORT_MANIFEST_VERSION = 1;
+
+// `.export-index/sites/<siteId>/report-exports/`.
+export function reportExportsDir(paths: Paths, siteId: string): string {
+ return path.join(siteIndexDir(paths, siteId), REPORT_EXPORTS_DIRNAME);
+}
+
+// `.export-index/sites/<siteId>/report-exports/<reportId>/`.
+export function reportExportDir(paths: Paths, siteId: string, reportId: string): string {
+ return path.join(reportExportsDir(paths, siteId), reportId);
+}
+
+export type ReportExportFileEntry = {
+ // The file's name in the report's export dir (REPORT_EXPORT_FILENAMES).
+ file: string;
+ bytes: number;
+ sha256: string;
+ // Why it may not be published (over the limit), or absent when it may.
+ localOnly?: string;
+};
+
+export type ReportExportManifest = {
+ format: typeof REPORT_EXPORT_MANIFEST_FORMAT;
+ version: typeof REPORT_EXPORT_MANIFEST_VERSION;
+ siteId: string;
+ reportId: string;
+ // The sha256 of the report.json the exports were made from.
+ reportSha256: string;
+ exportedAt: string;
+ footer: ReportExportFooter;
+ files: Partial<Record<ReportExportFormat, ReportExportFileEntry>>;
+ // What was skipped, and why (a PDF with no browser, a pack with no `zip`).
+ notes: string[];
+};
+
+export function sha256Hex(data: string | Uint8Array): string {
+ return createHash("sha256").update(data).digest("hex");
+}
+
+// The sha256 of a report.json as it is on disk, or null when it is unreadable.
+export async function reportFileSha256(paths: Paths, siteId: string, reportId: string): Promise<string | null> {
+ try {
+ return sha256Hex(await readFile(siteReportFile(paths, siteId, reportId)));
+ } catch {
+ return null;
+ }
+}
+
+export async function readReportExportManifest(
+ paths: Paths,
+ siteId: string,
+ reportId: string,
+): Promise<ReportExportManifest | null> {
+ const read = await readJsonFile(path.join(reportExportDir(paths, siteId, reportId), REPORT_EXPORT_MANIFEST_FILENAME));
+ if (!read.ok) return null;
+ const v = read.value as Partial<ReportExportManifest> | null;
+ if (!v || v.format !== REPORT_EXPORT_MANIFEST_FORMAT || v.version !== REPORT_EXPORT_MANIFEST_VERSION) return null;
+ return v as ReportExportManifest;
+}
+
+export type PublishableReportExports = {
+ // format → the local file to publish.
+ files: Partial<Record<ReportExportFormat, string>>;
+ // Why an export is not published, one sentence each.
+ notes: string[];
+};
+
+// The exports of a report compose may publish: made from the report.json as
+// it is now, each file present and within the publish limit. A report never
+// exported has none, and no note.
+export async function publishableReportExports(
+ paths: Paths,
+ siteId: string,
+ reportId: string,
+): Promise<PublishableReportExports> {
+ const manifest = await readReportExportManifest(paths, siteId, reportId);
+ if (!manifest) return { files: {}, notes: [] };
+ const current = await reportFileSha256(paths, siteId, reportId);
+ if (current !== manifest.reportSha256) {
+ return {
+ files: {},
+ notes: ["its exports were made from another version of report.json — export again; none are published"],
+ };
+ }
+ const dir = reportExportDir(paths, siteId, reportId);
+ const files: PublishableReportExports["files"] = {};
+ const notes: string[] = [];
+ for (const f of REPORT_EXPORT_FORMATS) {
+ if (!manifest.files[f]) continue;
+ const file = path.join(dir, REPORT_EXPORT_FILENAMES[f]);
+ const st = await stat(file).catch(() => null);
+ if (!st?.isFile()) {
+ notes.push(`${REPORT_EXPORT_FILENAMES[f]} is named in export.json but missing — export again`);
+ continue;
+ }
+ const tooBig = publishFileSizeProblem(REPORT_EXPORT_FILENAMES[f], st.size);
+ if (tooBig) {
+ notes.push(`${tooBig}; it stays local`);
+ continue;
+ }
+ files[f] = file;
+ }
+ return { files, notes };
+}
diff --git a/common/publish/reportExports.ts b/common/publish/reportExports.ts
@@ -0,0 +1,628 @@
+// EXPORT A SITE'S REPORTS AS FILES — so a report survives a takedown as files
+// anyone can save and host again (plans/report-sites.md, "Exports").
+// `archilyzer reports export <siteId>` and the editor's `reports-export` job
+// run `exportSiteReports`; `reports prepare` runs it at its end.
+//
+// What it reads: the site's published reports, resolved exactly as compose
+// resolves them (./composeReports.ts resolveSiteReports — verified against the
+// corpus, the prepared media matched), and the media `reports prepare` left in
+// the site's report-media cache (./reportMedia.ts) with each report's stills.
+//
+// What it writes, per report, to
+// `.export-index/sites/<siteId>/report-exports/<reportId>/` (staging, never
+// served; ./reportExportFiles.ts names it):
+//
+// report.html ONE self-contained file (lib/report/exportHtml.ts):
+// inline CSS, no script, every still and post
+// screenshot a data: URI, recompressed here to WebP (or
+// JPEG) at most EXPORT_IMAGE_MAX_WIDTH wide; clips are
+// linked on the site, never inlined
+// report.pdf that HTML printed by headless Chromium (A4), its folded
+// evidence open. Where
+// Playwright or its browser is missing the PDF is skipped
+// with a note — never a failure
+// report.md plain Markdown, numbered references
+// (lib/report/exportMarkdown.ts)
+// evidence-pack.zip `<reportId>/report.html` pointing at its own media/
+// (the clips play offline, the stills and screenshots
+// as files), report.md, citations.json and .csv. Packed
+// by the system `zip`, deterministically (sorted names,
+// fixed times and modes, no extra attributes — the same
+// report checked at the same `now` packs to the same
+// bytes); a host without `zip` fails the format, naming it
+// export.json the manifest: each file's size and sha256, the sha256
+// of the report.json it was made from, the footer, notes
+//
+// Each run replaces the report's export dir whole: a format not asked for this
+// time is gone, never left from an older version of the report. Compose
+// publishes from here only what was made from the report.json as it is now,
+// each file within the publish limit (an evidence pack over 24 MiB stays
+// local).
+//
+// THE FOOTER (lib/report/exportHtml.ts ReportExportFooter) names which document
+// this is: today the report's date and the sha256 of its report.json;
+// `exportFooterFor` below is where the revision history (slice RH) fills in
+// the revision number and the commit.
+
+import { chmod, mkdir, mkdtemp, readdir, readFile, rm, utimes, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { execa } from "execa";
+import { publishFileSizeProblem } from "../lib/builtExport";
+import { writeFileAtomic, writeJsonAtomic } from "../lib/jsonFile-server";
+import { getPaths, type Paths } from "../lib/paths";
+import { getSite, listSiteIds, type Site } from "../lib/site";
+import { parseMomentKey, type SpanMoment } from "../lib/citations/moments";
+import type { Report } from "../lib/report/schema";
+import { reportExportHtml, type ExportClip, type ReportExportFooter } from "../lib/report/exportHtml";
+import { reportExportMarkdown } from "../lib/report/exportMarkdown";
+import {
+ REPORT_EXPORT_FILENAMES,
+ REPORT_EXPORT_FORMATS,
+ type ReportExportFormat,
+ type ReportPageView,
+ type SpanCitationView,
+} from "../lib/report/views";
+import { importPlaywright } from "../social/playwrightRuntime";
+import {
+ citationSet,
+ citationsCsv,
+ clipPath,
+ ComposeReportsError,
+ formatComposeReportsProblems,
+ resolveSiteReports,
+ type ResolvedSiteReports,
+} from "./composeReports";
+import {
+ REPORT_EXPORT_MANIFEST_FILENAME,
+ REPORT_EXPORT_MANIFEST_FORMAT,
+ REPORT_EXPORT_MANIFEST_VERSION,
+ reportExportDir,
+ reportExportsDir,
+ reportFileSha256,
+ sha256Hex,
+ type ReportExportFileEntry,
+ type ReportExportManifest,
+} from "./reportExportFiles";
+import { siteReportDir, type ReportMediaIndex } from "./reportMedia";
+
+// A still or a screenshot in report.html: at most this wide, recompressed.
+export const EXPORT_IMAGE_MAX_WIDTH = 1200;
+export const EXPORT_WEBP_QUALITY = 80;
+// The evidence pack's file times: fixed, so the same files pack the same.
+export const EVIDENCE_PACK_MTIME = new Date("2000-01-01T00:00:00Z");
+
+export type ReportExportProblem = {
+ // The report, absent for a problem of the whole run.
+ report?: string;
+ format?: ReportExportFormat;
+ message: string;
+};
+
+// Prints HTML to PDF. One is opened per run and closed at its end.
+export type PdfPrinter = {
+ print: (html: string) => Promise<Uint8Array>;
+ close: () => Promise<void>;
+};
+
+// A printer, or why there is none (the PDF is then skipped with that note).
+export type OpenPdfPrinter = () => Promise<PdfPrinter | { missing: string }>;
+
+export type ExportSiteReportsOptions = {
+ siteId: string;
+ paths?: Paths;
+ // One published report; default every one.
+ reportId?: string;
+ // Default: all of REPORT_EXPORT_FORMATS.
+ formats?: readonly ReportExportFormat[];
+ // Export even where a citation's media was not prepared (its clip is then
+ // neither linked nor packed), as compose's `--allow-missing-media`.
+ allowMissingMedia?: boolean;
+ onLog?: (line: string) => void;
+ signal?: AbortSignal;
+ // `social.x.visibility` and friends; default the live settings.
+ settings?: { social?: { x?: { visibility?: unknown } } };
+ now?: () => Date;
+ // Default: headless Chromium through Playwright (openPlaywrightPdfPrinter).
+ openPdfPrinter?: OpenPdfPrinter;
+ // Default: `zip` on PATH.
+ zipBin?: string;
+};
+
+export type ReportExportResult = {
+ reportId: string;
+ dir: string;
+ manifest: ReportExportManifest;
+};
+
+export type ExportSiteReportsResult = {
+ exported: ReportExportResult[];
+ problems: ReportExportProblem[];
+};
+
+const mib = (n: number) => `${(n / 1024 / 1024).toFixed(1)} MiB`;
+const firstLine = (e: unknown) => String((e as Error)?.message ?? e).split("\n")[0].slice(0, 200);
+
+export function parseReportExportFormats(v: string): ReportExportFormat[] | null {
+ const parts = v.split(",").map((s) => s.trim()).filter(Boolean);
+ if (parts.length === 0) return null;
+ for (const p of parts) if (!(REPORT_EXPORT_FORMATS as readonly string[]).includes(p)) return null;
+ return REPORT_EXPORT_FORMATS.filter((f) => parts.includes(f));
+}
+
+// THE FOOTER HOOK. Today: the report's own date and the sha256 of its
+// report.json. The revision history (slice RH) adds `revision` and `commit`
+// here — nothing else changes: the HTML, PDF and Markdown all print
+// reportExportFooterLine of what this returns, leaving out what is absent.
+export function exportFooterFor(view: Pick<ReportPageView, "published" | "updated">, reportSha256: string): ReportExportFooter {
+ const date = view.updated ?? view.published;
+ return { ...(date ? { date } : {}), reportSha256 };
+}
+
+// ─── PDF ───
+
+export const openPlaywrightPdfPrinter: OpenPdfPrinter = async () => {
+ let chromium;
+ try {
+ ({ chromium } = await importPlaywright());
+ } catch {
+ return { missing: "Playwright is not available on this host" };
+ }
+ let browser;
+ try {
+ browser = await chromium.launch({ headless: true });
+ } catch (e) {
+ return { missing: `headless Chromium did not start (${firstLine(e)})` };
+ }
+ const b = browser;
+ return {
+ print: async (html) => {
+ const context = await b.newContext({});
+ try {
+ const page = await context.newPage();
+ if (!page.setContent || !page.pdf) throw new Error("this Playwright cannot print a page");
+ await page.setContent(html, { waitUntil: "load" });
+ return await page.pdf({ format: "A4", printBackground: true });
+ } finally {
+ await context.close();
+ }
+ },
+ close: () => b.close(),
+ };
+};
+
+// ─── Images ───
+
+const IMAGE_MIME: Record<string, string> = {
+ ".png": "image/png",
+ ".jpg": "image/jpeg",
+ ".jpeg": "image/jpeg",
+ ".gif": "image/gif",
+ ".webp": "image/webp",
+ ".avif": "image/avif",
+};
+
+async function ffmpegImage(ffmpegBin: string, file: string, codec: "webp" | "jpeg", signal?: AbortSignal): Promise<Uint8Array | null> {
+ const scale = `scale='min(${EXPORT_IMAGE_MAX_WIDTH},iw)':-2`;
+ const tail =
+ codec === "webp"
+ ? ["-c:v", "libwebp", "-quality", String(EXPORT_WEBP_QUALITY), "-compression_level", "6", "-f", "webp"]
+ : ["-pix_fmt", "yuvj420p", "-c:v", "mjpeg", "-q:v", "4", "-f", "image2pipe"];
+ const r = await execa(
+ ffmpegBin,
+ ["-hide_banner", "-loglevel", "error", "-i", file, "-frames:v", "1", "-vf", scale, "-map_metadata", "-1",
+ "-fflags", "+bitexact", "-flags", "+bitexact", ...tail, "pipe:1"],
+ { reject: false, encoding: "buffer", timeout: 60_000, cancelSignal: signal },
+ ).catch(() => null);
+ if (!r || r.exitCode !== 0 || !(r.stdout instanceof Uint8Array) || r.stdout.length === 0) return null;
+ return r.stdout;
+}
+
+// An image as a data: URI for report.html: recompressed to WebP (else JPEG) at
+// most EXPORT_IMAGE_MAX_WIDTH wide when that is smaller than the file, else
+// the file as it is. Null when the file cannot be read.
+export async function exportImageDataUri(file: string, ffmpegBin: string, signal?: AbortSignal): Promise<string | null> {
+ let raw: Buffer;
+ try {
+ raw = await readFile(file);
+ } catch {
+ return null;
+ }
+ let best: { mime: string; data: Uint8Array } = {
+ mime: IMAGE_MIME[path.extname(file).toLowerCase()] ?? "application/octet-stream",
+ data: raw,
+ };
+ for (const codec of ["webp", "jpeg"] as const) {
+ const out = await ffmpegImage(ffmpegBin, file, codec, signal);
+ if (out && out.length < best.data.length) best = { mime: codec === "webp" ? "image/webp" : "image/jpeg", data: out };
+ if (out) break;
+ }
+ return `data:${best.mime};base64,${Buffer.from(best.data).toString("base64")}`;
+}
+
+// The image paths a view names (a source sentence's still, a post's shot).
+function viewImagePaths(view: ReportPageView): string[] {
+ const out = new Set<string>();
+ for (const c of Object.values(view.citations)) {
+ if (c.kind === "source" && c.image) out.add(c.image);
+ if (c.kind === "post" && c.shot) out.add(c.shot);
+ }
+ return [...out].sort();
+}
+
+// ─── Where a view's files are on this host ───
+
+export type ReportFiles = {
+ // A site-root path the view names → the file on disk.
+ local: (sitePath: string) => string | undefined;
+ // A span citation's prepared clip: its site-root path and file.
+ clip: (c: SpanCitationView) => { sitePath: string; file: string; kind: "video" | "audio" } | undefined;
+};
+
+function within(base: string, rel: string): string | undefined {
+ const b = path.resolve(base);
+ const f = path.resolve(b, rel);
+ return f.startsWith(b + path.sep) ? f : undefined;
+}
+
+function reportFiles(paths: Paths, siteId: string, view: ReportPageView, resolved: ResolvedSiteReports): ReportFiles {
+ const own = `/reports/${view.id}/`;
+ return {
+ local: (p) => {
+ if (p.startsWith(own)) return within(siteReportDir(paths, siteId, view.id), p.slice(own.length));
+ if (p.startsWith("/media/")) return within(resolved.cacheDir, p.slice("/media/".length));
+ return undefined;
+ },
+ clip: (c) => {
+ const entry = resolved.mediaOf.get(c.moment);
+ if (!entry || entry.kind === "post") return undefined;
+ const file = within(resolved.cacheDir, entry.file);
+ if (!file) return undefined;
+ return { sitePath: clipPath(parseMomentKey(c.moment) as SpanMoment, entry.kind), file, kind: entry.kind };
+ },
+ };
+}
+
+// A site-root path's place in the evidence pack: the site's media under
+// media/, the report's own files (its stills) under media/report/.
+function packPathOf(view: ReportPageView, sitePath: string): string | undefined {
+ const own = `/reports/${view.id}/`;
+ if (sitePath.startsWith(own)) return `media/report/${sitePath.slice(own.length)}`;
+ if (sitePath.startsWith("/media/")) return sitePath.slice(1);
+ return undefined;
+}
+
+// ─── The evidence pack ───
+
+async function zipAvailable(zipBin: string): Promise<boolean> {
+ const r = await execa(zipBin, ["-h"], { reject: false }).catch(() => null);
+ return !!r && r.exitCode === 0;
+}
+
+// `names` (relative to `cwd`) zipped into `outFile`, the same bytes for the
+// same files: names sorted, every time and mode fixed, no extra attributes,
+// times read in UTC. Media is stored, text deflated.
+export async function zipDeterministic(cwd: string, names: readonly string[], outFile: string, zipBin = "zip"): Promise<void> {
+ const sorted = [...names].sort();
+ for (const n of sorted) {
+ await chmod(path.join(cwd, n), 0o644);
+ await utimes(path.join(cwd, n), EVIDENCE_PACK_MTIME, EVIDENCE_PACK_MTIME);
+ }
+ await rm(outFile, { force: true });
+ const r = await execa(
+ zipBin,
+ ["-X", "-D", "-q", "-n", ".mp4:.m4a:.png:.jpg:.jpeg:.webp:.gif:.avif:.pdf:.zip", outFile, "-@"],
+ { cwd, input: `${sorted.join("\n")}\n`, env: { TZ: "UTC" }, reject: false },
+ );
+ if (r.exitCode !== 0) throw new Error(`zip failed (exit ${r.exitCode}): ${String(r.stderr ?? "").trim().slice(0, 300)}`);
+}
+
+async function buildEvidencePack(o: {
+ report: Report;
+ view: ReportPageView;
+ files: ReportFiles;
+ site: Pick<Site, "siteId" | "siteUrl" | "siteTitle">;
+ footer: ReportExportFooter;
+ markdown: string;
+ outFile: string;
+ zipBin: string;
+}): Promise<void> {
+ const tmp = await mkdtemp(path.join(os.tmpdir(), "report-evidence-pack-"));
+ try {
+ const top = o.view.id;
+ const names: string[] = [];
+ const put = async (rel: string, data: string | Uint8Array) => {
+ const f = path.join(tmp, top, rel);
+ await mkdir(path.dirname(f), { recursive: true });
+ await writeFile(f, data);
+ names.push(`${top}/${rel}`);
+ };
+ const images = new Map<string, string>();
+ for (const p of viewImagePaths(o.view)) {
+ const local = o.files.local(p);
+ const rel = packPathOf(o.view, p);
+ if (!local || !rel) continue;
+ const data = await readFile(local).catch(() => null);
+ if (!data) continue;
+ await put(rel, data);
+ images.set(p, rel);
+ }
+ const clips = new Map<string, ExportClip>();
+ for (const c of Object.values(o.view.citations)) {
+ if (c.kind !== "video" && c.kind !== "audio") continue;
+ const clip = o.files.clip(c);
+ if (!clip || clips.has(c.moment)) continue;
+ const rel = packPathOf(o.view, clip.sitePath);
+ const data = rel ? await readFile(clip.file).catch(() => null) : null;
+ if (!rel || !data) continue;
+ await put(rel, data);
+ clips.set(c.moment, { href: rel, kind: clip.kind, play: true });
+ }
+ const html = reportExportHtml(o.view, {
+ siteUrl: o.site.siteUrl || undefined,
+ siteTitle: o.site.siteTitle || undefined,
+ footer: o.footer,
+ image: (p) => images.get(p),
+ clip: (c) => clips.get(c.moment),
+ });
+ await put(REPORT_EXPORT_FILENAMES.html, html);
+ await put(REPORT_EXPORT_FILENAMES.md, o.markdown);
+ await put("citations.json", `${JSON.stringify(citationSet(o.report, o.view), null, 2)}\n`);
+ await put("citations.csv", citationsCsv(o.view));
+ const tmpZip = path.join(tmp, "pack.zip");
+ await zipDeterministic(tmp, names, tmpZip, o.zipBin);
+ await writeFileAtomic(o.outFile, await readFile(tmpZip));
+ } finally {
+ await rm(tmp, { recursive: true, force: true });
+ }
+}
+
+// ─── One report ───
+
+async function exportOneReport(o: {
+ paths: Paths;
+ site: Site;
+ report: Report;
+ view: ReportPageView;
+ resolved: ResolvedSiteReports;
+ formats: readonly ReportExportFormat[];
+ printer: () => Promise<PdfPrinter | { missing: string }>;
+ zipBin: string;
+ now: Date;
+ log: (line: string) => void;
+ signal?: AbortSignal;
+}): Promise<{ result?: ReportExportResult; problems: ReportExportProblem[] }> {
+ const { paths, site, report, view } = o;
+ const sha = await reportFileSha256(paths, site.siteId, report.id);
+ if (!sha) return { problems: [{ report: report.id, message: "its report.json cannot be read" }] };
+ return writeReportExports({
+ ...o,
+ dir: reportExportDir(paths, site.siteId, report.id),
+ reportSha256: sha,
+ files: reportFiles(paths, site.siteId, view, o.resolved),
+ ffmpegBin: paths.ffmpegBin,
+ });
+}
+
+// ONE REPORT'S EXPORTS, from its view and where its files are: every format
+// asked for and the manifest, into `dir` (emptied first). The site run above
+// calls it for each report; the report-site e2e stage calls it for its
+// fixture. `report` is the document as resolved (its verification
+// computed): the evidence pack's citations.json is its citation set.
+export async function writeReportExports(o: {
+ dir: string;
+ site: Pick<Site, "siteId" | "siteUrl" | "siteTitle">;
+ report: Report;
+ view: ReportPageView;
+ reportSha256: string;
+ files: ReportFiles;
+ ffmpegBin: string;
+ formats: readonly ReportExportFormat[];
+ printer: () => Promise<PdfPrinter | { missing: string }>;
+ zipBin: string;
+ now: Date;
+ log: (line: string) => void;
+ signal?: AbortSignal;
+}): Promise<{ result: ReportExportResult; problems: ReportExportProblem[] }> {
+ const { site, report, view, files } = o;
+ const problems: ReportExportProblem[] = [];
+ const sha = o.reportSha256;
+ const footer = exportFooterFor(view, sha);
+ const siteUrl = site.siteUrl || undefined;
+ const siteTitle = site.siteTitle || undefined;
+
+ // The one-file export's images, as data: URIs.
+ const dataUris = new Map<string, string>();
+ for (const p of viewImagePaths(view)) {
+ const local = files.local(p);
+ const uri = local ? await exportImageDataUri(local, o.ffmpegBin, o.signal) : null;
+ if (uri) dataUris.set(p, uri);
+ else problems.push({ report: report.id, format: "html", message: `the image ${p} cannot be read` });
+ }
+ const htmlOpts = {
+ siteUrl,
+ siteTitle,
+ footer,
+ image: (p: string) => dataUris.get(p),
+ clip: (c: SpanCitationView) => {
+ const clip = files.clip(c);
+ const href = clip && siteUrl ? `${siteUrl.replace(/\/+$/, "")}${clip.sitePath}` : undefined;
+ return clip && href ? { href, kind: clip.kind } : undefined;
+ },
+ };
+ const html = reportExportHtml(view, htmlOpts);
+ // The PDF prints what the page folds away (the evidence the document under
+ // review gave itself) open.
+ const printHtml = reportExportHtml(view, { ...htmlOpts, print: true });
+ const markdown = reportExportMarkdown(view, { siteUrl, siteTitle, footer });
+
+ const dir = o.dir;
+ await rm(dir, { recursive: true, force: true });
+ await mkdir(dir, { recursive: true });
+ const entries: ReportExportManifest["files"] = {};
+ const notes: string[] = [];
+ const record = async (f: ReportExportFormat) => {
+ const name = REPORT_EXPORT_FILENAMES[f];
+ const data = await readFile(path.join(dir, name));
+ const localOnly = publishFileSizeProblem(name, data.length);
+ const entry: ReportExportFileEntry = { file: name, bytes: data.length, sha256: sha256Hex(data), ...(localOnly ? { localOnly } : {}) };
+ entries[f] = entry;
+ o.log(` + ${report.id}/${name} (${mib(entry.bytes)})${localOnly ? ` — local only: ${localOnly}` : ""}`);
+ };
+
+ for (const f of o.formats) {
+ if (o.signal?.aborted) throw new Error("export cancelled");
+ const out = path.join(dir, REPORT_EXPORT_FILENAMES[f]);
+ switch (f) {
+ case "html":
+ await writeFileAtomic(out, html);
+ await record(f);
+ break;
+ case "md":
+ await writeFileAtomic(out, markdown);
+ await record(f);
+ break;
+ case "pdf": {
+ const printer = await o.printer();
+ if ("missing" in printer) {
+ notes.push(`report.pdf skipped: ${printer.missing}`);
+ o.log(` - ${report.id}/report.pdf skipped: ${printer.missing}`);
+ break;
+ }
+ try {
+ await writeFileAtomic(out, Buffer.from(await printer.print(printHtml)));
+ await record(f);
+ } catch (e) {
+ notes.push(`report.pdf skipped: printing failed (${firstLine(e)})`);
+ o.log(` - ${report.id}/report.pdf skipped: printing failed (${firstLine(e)})`);
+ }
+ break;
+ }
+ case "zip":
+ if (!(await zipAvailable(o.zipBin))) {
+ problems.push({
+ report: report.id,
+ format: "zip",
+ message: `the evidence pack needs the \`zip\` program (Info-ZIP), and "${o.zipBin}" cannot be run on this host — install zip, or export without zip (--formats html,pdf,md)`,
+ });
+ break;
+ }
+ try {
+ await buildEvidencePack({ report, view, files, site, footer, markdown, outFile: out, zipBin: o.zipBin });
+ await record(f);
+ } catch (e) {
+ problems.push({ report: report.id, format: "zip", message: `the evidence pack could not be packed: ${firstLine(e)}` });
+ }
+ break;
+ }
+ }
+
+ const manifest: ReportExportManifest = {
+ format: REPORT_EXPORT_MANIFEST_FORMAT,
+ version: REPORT_EXPORT_MANIFEST_VERSION,
+ siteId: site.siteId,
+ reportId: report.id,
+ reportSha256: sha,
+ exportedAt: o.now.toISOString(),
+ footer,
+ files: entries,
+ notes,
+ };
+ await writeJsonAtomic(path.join(dir, REPORT_EXPORT_MANIFEST_FILENAME), manifest);
+ return { result: { reportId: report.id, dir, manifest }, problems };
+}
+
+// ─── The run ───
+
+export async function exportSiteReports(opts: ExportSiteReportsOptions): Promise<ExportSiteReportsResult> {
+ const paths = opts.paths ?? getPaths();
+ const log = opts.onLog ?? (() => {});
+ const { siteId } = opts;
+ if (!listSiteIds(paths).includes(siteId)) throw new Error(`no site "${siteId}" (no sites/${siteId}/site.json)`);
+ const site = getSite(siteId, paths);
+ const published = site.reports ?? [];
+ if (opts.reportId && !published.includes(opts.reportId)) {
+ throw new Error(`site "${siteId}" does not publish a report "${opts.reportId}" (only a published report is exported)`);
+ }
+ const formats = opts.formats ?? REPORT_EXPORT_FORMATS;
+
+ let resolved: ResolvedSiteReports;
+ try {
+ resolved = await resolveSiteReports({
+ paths,
+ site,
+ allowMissingMedia: opts.allowMissingMedia,
+ settings: opts.settings,
+ now: opts.now,
+ log,
+ });
+ } catch (e) {
+ if (e instanceof ComposeReportsError) {
+ return {
+ exported: [],
+ problems: formatComposeReportsProblems(e.problems).map((message) => ({ message: `${message} (as compose would refuse it)` })),
+ };
+ }
+ throw e;
+ }
+
+ // A report no longer published keeps no exports.
+ if (!opts.reportId) {
+ for (const name of await readdir(reportExportsDir(paths, siteId)).catch(() => [] as string[])) {
+ if (!published.includes(name)) await rm(path.join(reportExportsDir(paths, siteId), name), { recursive: true, force: true });
+ }
+ }
+
+ // Opened on the first PDF, once for the run.
+ const opened: { printer?: Promise<PdfPrinter | { missing: string }> } = {};
+ const openPrinter = () => (opened.printer ??= (opts.openPdfPrinter ?? openPlaywrightPdfPrinter)());
+ const now = opts.now?.() ?? new Date();
+ const exported: ReportExportResult[] = [];
+ const problems: ReportExportProblem[] = [];
+ log(`${siteId}: exporting ${opts.reportId ?? `${resolved.reports.length} report(s)`} as ${formats.join(", ")}.`);
+ try {
+ for (let i = 0; i < resolved.reports.length; i++) {
+ const report = resolved.reports[i];
+ if (opts.reportId && report.id !== opts.reportId) continue;
+ if (opts.signal?.aborted) throw new Error("export cancelled");
+ const r = await exportOneReport({
+ paths,
+ site,
+ report,
+ view: resolved.views[i],
+ resolved,
+ formats,
+ printer: openPrinter,
+ zipBin: opts.zipBin ?? "zip",
+ now,
+ log,
+ signal: opts.signal,
+ });
+ if (r.result) exported.push(r.result);
+ problems.push(...r.problems);
+ }
+ } finally {
+ const p = await opened.printer;
+ if (p && !("missing" in p)) await p.close().catch(() => {});
+ }
+ log(`${exported.length} report(s) exported; ${problems.length} problem(s).`);
+ return { exported, problems };
+}
+
+export function formatReportExportProblems(problems: readonly ReportExportProblem[]): string[] {
+ return problems.map((p) => `${p.report ?? "site"}${p.format ? ` (${p.format})` : ""}: ${p.message}`);
+}
+
+// The end of `reports prepare`: export the reports when their media is
+// complete; a prepare with problems leaves the exports as they were (compose
+// would refuse the site anyway).
+export async function exportAfterPrepare(
+ index: ReportMediaIndex,
+ opts: ExportSiteReportsOptions,
+): Promise<ExportSiteReportsResult | null> {
+ if (index.problems.length > 0) {
+ opts.onLog?.("exports: not made — the evidence media is not complete.");
+ return null;
+ }
+ return exportSiteReports(opts);
+}
diff --git a/common/publish/reportMedia.test.ts b/common/publish/reportMedia.test.ts
@@ -225,5 +225,31 @@ test("a clean site: no problems, and a clip no longer cited leaves the cache", a
const files = readdirSync(reportMediaDir(paths, SITE)).sort();
const clip = index.moments[VIDEO_KEY].file;
assert.deepEqual(files, [clip.replace(/\.mp4$/, ".json"), clip, "index.json"].sort());
- assert.equal(await prepareMain({ siteId: SITE }, { log: () => {}, error: () => {} }), 0);
+});
+
+test("a clean prepare ends by exporting the reports; a host with no browser skips the PDF with a note", async () => {
+ // The record behind r2's citation, so the export can resolve it as compose does.
+ writeJson(path.join(videoDir(CH, "abc123"), "metadata.info.json"), {
+ id: "abc123",
+ title: "Demo stream",
+ upload_date: "20260110",
+ webpage_url: "https://www.youtube.com/watch?v=abc123",
+ extractor_key: "Youtube",
+ });
+ writeFileSync(
+ path.join(videoDir(CH, "abc123"), "transcript.en.vtt"),
+ "WEBVTT\n\n00:00:00.000 --> 00:00:06.000 align:start position:0%\none<00:00:00.000><c></c>\n",
+ );
+ const out: string[] = [];
+ const err: string[] = [];
+ const code = await prepareMain(
+ { siteId: SITE, exportOptions: { openPdfPrinter: async () => ({ missing: "no browser here" }) } },
+ { log: (l) => out.push(l), error: (l) => err.push(l) },
+ );
+ assert.equal(code, 0, err.join("\n"));
+ const dir = path.join(paths.exportSitesIndexDir, SITE, "report-exports", "r2");
+ assert.deepEqual(readdirSync(dir).sort(), ["evidence-pack.zip", "export.json", "report.html", "report.md"]);
+ const manifest = JSON.parse(readFileSync(path.join(dir, "export.json"), "utf8"));
+ assert.deepEqual(manifest.notes, ["report.pdf skipped: no browser here"]);
+ assert.ok(out.some((l) => l.includes("report.pdf skipped: no browser here")));
});
diff --git a/common/publish/source.ts b/common/publish/source.ts
@@ -42,6 +42,7 @@ import { runChildIntoLog } from "../jobs/runChild";
import { copyPublicFile, ownDir, writePublicFile } from "../bin/_publicFile";
import { getPaths, type Paths } from "../lib/paths";
import { PROJECT_NAME, PROJECT_URL } from "../lib/project";
+import { PUBLISH_MAX_FILE_BYTES } from "../lib/builtExport";
import {
CLONE_URL,
HISTORY_DIR,
@@ -118,7 +119,7 @@ export const HOME_REPLACEMENT = "/home/user";
// Cloudflare Pages allows 20,000 files per deployment and 25 MiB per file;
// the step refuses well inside both, leaving the rest of the site its room.
export const MAX_FILES = 15_000;
-export const MAX_FILE_BYTES = 24 * 1024 * 1024;
+export const MAX_FILE_BYTES = PUBLISH_MAX_FILE_BYTES;
// Packs are split at this size (under the per-file cap, with room to grow).
const PACK_SIZE = "20m";
diff --git a/common/social/playwrightRuntime.ts b/common/social/playwrightRuntime.ts
@@ -1,5 +1,6 @@
// Locating Playwright at runtime, for the modules that need a real browser
-// (the Nitter fetcher, the X fallback fetcher and the X session broker).
+// (the Nitter fetcher, the X fallback fetcher, the X session broker, and a
+// report's PDF export).
//
// Two constraints pull against each other:
// - `common` must NOT depend on Playwright. It is imported by the Docker
@@ -48,6 +49,14 @@ export type PageLike = {
fullPage?: boolean;
clip?: { x: number; y: number; width: number; height: number };
}) => Promise<Uint8Array>;
+ // A report's PDF export (publish/reportExports.ts): the export's HTML set as
+ // the page, printed. Optional: the fetchers' fakes do not print.
+ setContent?: (html: string, opts?: { waitUntil?: "load" | "domcontentloaded" | "networkidle" }) => Promise<void>;
+ pdf?: (opts?: {
+ format?: string;
+ printBackground?: boolean;
+ margin?: { top?: string; right?: string; bottom?: string; left?: string };
+ }) => Promise<Uint8Array>;
// The page's own request context (the profile's cookies): an X Article's
// inline images are fetched through it (xArticleCapture.ts).
request?: {
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -2,6 +2,7 @@
## [Unreleased]
- **A forum thread can be archived as a posts source.** A XenForo thread URL (`…/threads/<title>.<id>/`; Kiwi Farms is recognised by host) makes a forum-thread channel — platform "xenforo", one channel per thread, each forum post a post — searchable and readable like X and Bluesky posts, in the editor, the export and the MCP (`get_thread` gives a forum post's conversation: the posts it quotes and the posts quoting it). **Fetch posts** reads the thread in a headless browser, newest page first, one page at a time with a 10–20 s pause (the channel key `postPagePauseSeconds` sets it), and stops at already-archived posts; a **Latest N pages** box (`archilyzer posts fetch --pages N`) caps a run, and the next run continues where it stopped. The browser keeps one profile per forum host, so a browser check it clears once (KiwiFlare's proof of work, say) stays cleared; a check that does not clear within a minute, a captcha, a login wall or a refusal stops the run with the reason and keeps its place — never retried at once. **Connect forum session** on the channel page opens that profile in a window on the editor's machine, at the thread, for the operator to clear it or log in. **Import saved pages** (`archilyzer posts import-html <slug> <file-or-dir>…`) reads thread pages saved from a browser ("Save page as", complete or HTML only) through the same parser: new posts are added and a post saved again after an edit is updated. **Capture posts** works on forum posts: a screenshot of the post and its attached files, through the same profile. A forum post keeps its thread title, page, position, author id, last-edit time, quoted posts and its media links; quoted text is marked with "> " lines.
+- **A site's reports can be exported as files a reader saves and hosts again.** `archilyzer reports export <site> [--report <id>] [--formats html,pdf,md,zip]`, the `reports-export` job (`POST /api/ops/reports-export`, `pnpm ops reports-export`, and **Export reports** on a site's Reports tab) write each published report, checked as the build checks it, into `.export-index/sites/<site>/report-exports/<report>/`: `report.html`, one self-contained page (its own style, no script, stills and post screenshots inlined and recompressed, clips linked on the site); `report.pdf`, that page printed by headless Chromium, skipped with a note where there is none; `report.md`, plain Markdown with numbered references; and `evidence-pack.zip`, the page with its clips, stills and screenshots as files plus the Markdown and the citations, packed by the system `zip` (a host without it fails that format, naming it). An `export.json` names each file's size and checksum and the checksum of the report.json it was made from; every export ends with the report's date and the start of that checksum. Preparing the evidence media now exports at its end when nothing is missing, on the same queue. The build publishes an export beside the report only when it was made from the report as it is now and is at most 24 MiB — a larger evidence pack stays local — and the Reports tab lists each report's exports, their sizes and which the next build publishes. The 24 MiB limit the source mirror and the evidence clips already kept is now one shared number.
- **A report video's cue lookup names a site that publishes only its reports.** Pointing a report-to-video manifest at such a site (`corpus.json` `site.scope: "cited"`) used to fail with "channel … is not in corpus.json"; it now says the site publishes no transcripts and to use a full archive or a local corpus. The site form's **Publish** hint says a cited-only site is never listed on the homepage or the hub.
- **A site's build composes its reports, and a site that publishes only its reports ships nothing else.** Every site's compose now writes the reports its `site.json` publishes: each report's page and its citations as `citations.json` and `citations.csv` under `/reports/<id>/`, its cited stills, a page per cited moment with the record, the transcript lines around the span and every report that cites it, and the clips and post captures `archilyzer reports prepare` made for it, only the cited ones. Each quote is checked against the record as it is composed (a span's against its cues within 5 s either side, read from `en-orig` when the `en` track has no cues; a post's against its text) and the score, time and method are written into the citation, replacing any typed by hand. The build stops with the list of every problem before anything is written: an invalid report, a citation of a channel outside the site or of a post the site may not carry, a missing record, still or post, a quote that matches less than 60 % of what the record says, and a citation without prepared media or with media cut for another span (`--allow-missing-media` on `archilyzer compose site` and `build site` lets those two through, without a clip). A site with `publish: "cited"` removes everything corpus-shaped from `export/public` before it writes its reports, and its built `out/` is checked against what a cited site may hold: anything else, a file over 25 MiB or more than 20,000 files fails the build, and every deploy path (the Publish tab, `deploy site`, Build & deploy, Build & deploy all, the container build) refuses it, as it refuses a site set to cited whose last build was a full one. The hub's compose removes a report site's files too.
- **A site has a Reports tab.** `/sites/<site>/reports` lists every report under the site's `reports/` directory — the published ones in their order, then the drafts — with its kind, dates, sections, claims, citations by kind and, for a fact-check, how many claims carry each verdict. Each report's problems, from the same checker the prepare step and the build use, open under it. A draft with no problems can be published, and a published report moved up or down or unpublished; each writes only the site's `reports` list, applied to the list as it is on disk at that moment, so it never overwrites another change to the site. "Prepare evidence media" queues the `reports-prepare` job, and beside it the tab shows the last prepared media (moments by kind, total size, problems by kind) and links the last prepare job. What the site publishes (full or cited) is shown with a link to Settings, where it is changed.
diff --git a/editor/app/api/ops/reports-export/route.test.ts b/editor/app/api/ops/reports-export/route.test.ts
@@ -0,0 +1,55 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, rm } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+
+// Run with:
+// pnpm -C editor exec tsx --test "app/api/ops/reports-export/route.test.ts"
+//
+// The body's shape and the refusals that come before any job, answered from an
+// empty temp corpus: no job is queued and nothing is exported.
+
+const ROOT = await mkdtemp(path.join(os.tmpdir(), "reports-export-route-"));
+// Set before the route (and getPaths, which caches) is first imported.
+process.env.WORKER_TOKEN = "test-token";
+process.env.TRANSCRIPTS_DIR = ROOT;
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+const { POST } = await import("./route");
+test.after(() => rm(ROOT, { recursive: true, force: true }));
+
+async function post(body: Record<string, unknown>): Promise<{ status: number; error: string }> {
+ const res = await POST(
+ new Request("http://localhost/api/ops/reports-export", {
+ method: "POST",
+ headers: {
+ authorization: "Bearer test-token",
+ "content-type": "application/json",
+ },
+ body: JSON.stringify(body),
+ }),
+ );
+ return { status: res.status, error: ((await res.json()) as { error?: string }).error ?? "" };
+}
+
+test("siteId is required and a site id; reportId and formats are the other keys", async () => {
+ assert.match((await post({})).error, /"siteId" is required/);
+ const bad = await post({ siteId: "../x" });
+ assert.equal(bad.status, 400);
+ assert.match(bad.error, /not a valid site id/);
+ const unknown = await post({ siteId: "demo-site", pages: 1 });
+ assert.equal(unknown.status, 400);
+ assert.match(unknown.error, /unknown key\(s\): pages — this route accepts siteId, reportId, formats/);
+});
+
+test("a format not an export's is refused, naming it", async () => {
+ const r = await post({ siteId: "demo-site", formats: ["html", "docx"] });
+ assert.equal(r.status, 400);
+ assert.match(r.error, /1 of "formats" not in the export formats: docx/);
+});
+
+test("a site that does not exist is refused before any job", async () => {
+ const r = await post({ siteId: "demo-site", reportId: "r1", formats: ["md"] });
+ assert.equal(r.status, 400);
+ assert.match(r.error, /No site "demo-site"/);
+});
diff --git a/editor/app/api/ops/reports-export/route.ts b/editor/app/api/ops/reports-export/route.ts
@@ -0,0 +1,33 @@
+import { reportsExportAction } from "../../../sites/lib/reportsExportAction";
+import { jobResponse, OpsInputError, ops, optString, optSubset, reqString } from "../_lib";
+import { isValidSiteId } from "yt-dlp-transcript-common/lib/site";
+import { REPORT_EXPORT_FORMATS } from "yt-dlp-transcript-common/lib/report/views";
+
+export const dynamic = "force-dynamic";
+
+// POST { siteId: string, reportId?: string, formats?: ("html"|"pdf"|"md"|"zip")[] }
+// -> { ok: true, jobId }
+//
+// An adapter: one call to the action the site's Reports tab posts. Writes each
+// published report (or the one named) as report.html, report.pdf, report.md
+// and an evidence pack (or the formats named) into the site's report-exports
+// staging, as one `reports-export` job on prepare's queue. The job fails,
+// naming each, when an export cannot be written.
+export async function POST(request: Request) {
+ return ops(request, ["siteId", "reportId", "formats"], async (body) => {
+ const siteId = reqString(body, "siteId");
+ if (!isValidSiteId(siteId)) {
+ throw new OpsInputError(
+ `"${siteId}" is not a valid site id (lowercase letters, digits and "-"; must start with a letter or digit)`,
+ );
+ }
+ const reportId = optString(body, "reportId");
+ const formats = optSubset(body, "formats", REPORT_EXPORT_FORMATS, "the export formats");
+ return jobResponse(
+ await reportsExportAction(siteId, {
+ ...(reportId !== undefined ? { reportId } : {}),
+ ...(formats ? { formats } : {}),
+ }),
+ );
+ });
+}
diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts
@@ -55,6 +55,7 @@ import {
} from "../channels/[slug]/incompleteTranscriptActions";
import { replayFetchWindowAction } from "../channels/[slug]/videos/[id]/videoActions";
import { reportsPrepareAction } from "../sites/lib/reportsPrepareAction";
+import { reportsExportAction } from "../sites/lib/reportsExportAction";
export type ReplayHandler = (spec: JobSpec) => Promise<StreamActionResult>;
@@ -97,6 +98,17 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = {
const { p } = params(spec);
return reportsPrepareAction(str(p.siteId) ?? spec.slug);
},
+ // A report site's exports: the same site, report and formats, from the
+ // reports as they are now.
+ "reports-export": (spec) => {
+ const { p } = params(spec);
+ const reportId = str(p.reportId);
+ const formats = strings(p.formats);
+ return reportsExportAction(str(p.siteId) ?? spec.slug, {
+ ...(reportId ? { reportId } : {}),
+ ...(formats ? { formats } : {}),
+ });
+ },
// A clip window sourced for another tool. Replay RE-DERIVES from disk like
// every bucket job does: if the window (or a wider one covering it) has
// arrived since, the retry says so neutrally rather than paying twice.
diff --git a/editor/app/sites/[siteId]/reports/page.tsx b/editor/app/sites/[siteId]/reports/page.tsx
@@ -6,6 +6,11 @@ import { formatBytes } from "yt-dlp-transcript-common/lib/format";
import { isCitedSite } from "yt-dlp-transcript-common/lib/site";
import { readReportMediaIndex } from "yt-dlp-transcript-common/publish/reportMedia";
import { PrepareReportMediaButton } from "../../components/PrepareReportMediaButton";
+import { ExportReportsButton } from "../../components/ExportReportsButton";
+import {
+ REPORT_EXPORT_FILENAMES,
+ REPORT_EXPORT_FORMATS,
+} from "yt-dlp-transcript-common/lib/report/views";
import { ReportRowActions } from "../../components/ReportRowActions";
import {
publishRefusal,
@@ -13,7 +18,12 @@ import {
type ReportMediaSummary,
type ReportRow,
} from "../../lib/reportList";
-import { latestReportsPrepareJob, readSiteReportRows } from "../../lib/reportListServer";
+import {
+ latestSiteReportsJob,
+ readSiteReportExports,
+ readSiteReportRows,
+ type ReportExportsRow,
+} from "../../lib/reportListServer";
import { getSiteCached } from "../lib/siteCache";
export const dynamic = "force-dynamic";
@@ -25,7 +35,8 @@ export const metadata: Metadata = { title: "Reports" };
// draft, and what is wrong with it — the report's own checker, so this lists
// what prepare and the build would refuse. Publish, unpublish and reorder write
// that one key (lib/reportsActions.ts). The evidence media the published
-// reports cite is prepared here too, before a build.
+// reports cite is prepared here too, before a build, and the reports exported
+// as files (HTML, PDF, Markdown, an evidence pack) for the build to publish.
//
// The documents themselves are written elsewhere (a converter, or by hand);
// this tab never edits a report.json.
@@ -38,10 +49,12 @@ export default async function SiteReportsPage({
const site = getSiteCached(siteId);
if (!site) notFound();
const paths = getPaths();
- const [rows, mediaIndex, lastJob] = await Promise.all([
+ const [rows, mediaIndex, lastJob, exportRows, lastExportJob] = await Promise.all([
readSiteReportRows(paths, site),
readReportMediaIndex(paths, siteId),
- latestReportsPrepareJob(paths, siteId),
+ latestSiteReportsJob(paths, siteId, "reports-prepare"),
+ readSiteReportExports(paths, site),
+ latestSiteReportsJob(paths, siteId, "reports-export"),
]);
const publishedCount = rows.filter((r) => r.position !== null).length;
const cited = isCitedSite(site);
@@ -137,6 +150,44 @@ export default async function SiteReportsPage({
</p>
)}
</section>
+
+ <section
+ className="flex flex-col gap-3 border-t border-border pt-6"
+ aria-label="Exports"
+ >
+ <div>
+ <h2 className="text-lg font-semibold">Exports</h2>
+ <p className="text-sm text-muted-foreground">
+ Writes each published report as files a reader can save and host
+ again — one self-contained HTML page, a PDF of it, Markdown, and an
+ evidence pack (the page with its clips, stills and screenshots) —
+ for the site's next build to publish beside the report.
+ Preparing the evidence media exports too. A file over 24 MiB stays
+ on this machine.
+ </p>
+ </div>
+ <ExportReportsButton siteId={siteId} />
+ <p className="text-sm text-muted-foreground">
+ Last export job:{" "}
+ {lastExportJob ? (
+ <Link href={`/jobs/${lastExportJob.id}`} className="underline font-mono">
+ {lastExportJob.id}
+ </Link>
+ ) : (
+ "none among recent jobs"
+ )}
+ {lastExportJob && <> ({lastExportJob.status})</>}
+ </p>
+ {exportRows.length === 0 ? (
+ <p className="text-sm text-muted-foreground">No published report to export.</p>
+ ) : (
+ <ul className="flex flex-col gap-2" aria-label="Report exports">
+ {exportRows.map((row) => (
+ <ExportSummary key={row.reportId} row={row} />
+ ))}
+ </ul>
+ )}
+ </section>
</div>
);
}
@@ -298,3 +349,45 @@ function MediaSummary({ summary }: { summary: ReportMediaSummary }) {
</div>
);
}
+
+function ExportSummary({ row }: { row: ReportExportsRow }) {
+ const { manifest, publishable } = row;
+ return (
+ <li
+ data-testid="report-export"
+ data-report={row.reportId}
+ className="flex flex-col gap-1 rounded-md border border-border bg-card px-4 py-3 text-sm"
+ >
+ <p>
+ <code>{row.reportId}</code>{" "}
+ {manifest ? (
+ <span className="text-muted-foreground">
+ exported {new Date(manifest.exportedAt).toLocaleString()}
+ </span>
+ ) : (
+ <span className="text-muted-foreground">not exported yet</span>
+ )}
+ </p>
+ {manifest && (
+ <ul className="flex flex-wrap gap-x-4 gap-y-1 text-xs">
+ {REPORT_EXPORT_FORMATS.filter((f) => manifest.files[f]).map((f) => (
+ <li key={f}>
+ <code>{REPORT_EXPORT_FILENAMES[f]}</code>{" "}
+ {formatBytes(manifest.files[f]!.bytes)}{" "}
+ {publishable.files[f] ? (
+ <span className="text-success">published on build</span>
+ ) : (
+ <span className="text-warning">local only</span>
+ )}
+ </li>
+ ))}
+ </ul>
+ )}
+ {[...(manifest?.notes ?? []), ...publishable.notes].map((note, i) => (
+ <p key={i} className="text-xs text-muted-foreground">
+ {note}
+ </p>
+ ))}
+ </li>
+ );
+}
diff --git a/editor/app/sites/components/ExportReportsButton.tsx b/editor/app/sites/components/ExportReportsButton.tsx
@@ -0,0 +1,25 @@
+"use client";
+
+import { useRouter } from "next/navigation";
+import { StreamActionLog } from "yt-dlp-transcript-common/components/StreamActionLog";
+import { reportsExportAction } from "../lib/reportsExportAction";
+import { cancelJobAction } from "../../jobs/actions";
+
+// "Export reports": queues the site's reports-export job (every published
+// report as HTML, PDF, Markdown and an evidence pack) and streams its log.
+// When the job settles the tab is re-rendered, so the export list beside it
+// shows what the job just wrote.
+export function ExportReportsButton({ siteId }: { siteId: string }) {
+ const router = useRouter();
+ return (
+ <StreamActionLog
+ trigger={() => reportsExportAction(siteId)}
+ cancelAction={cancelJobAction}
+ buttonLabel="Export reports"
+ runningLabel="Exporting reports…"
+ onSettled={(started) => {
+ if (started) router.refresh();
+ }}
+ />
+ );
+}
diff --git a/editor/app/sites/lib/reportListServer.ts b/editor/app/sites/lib/reportListServer.ts
@@ -7,6 +7,12 @@ import { readJsonFile } from "yt-dlp-transcript-common/lib/jsonFile-server";
import { isReportId } from "yt-dlp-transcript-common/lib/report/schema";
import { siteDir, type Site } from "yt-dlp-transcript-common/lib/site";
import { siteReportFile } from "yt-dlp-transcript-common/publish/reportMedia";
+import {
+ publishableReportExports,
+ readReportExportManifest,
+ type PublishableReportExports,
+ type ReportExportManifest,
+} from "yt-dlp-transcript-common/publish/reportExportFiles";
import { getRegistry } from "yt-dlp-transcript-common/jobs/registry";
import { listAllJobs, type JobListEntry } from "yt-dlp-transcript-common/jobs/listJobs";
import { readJobMeta } from "yt-dlp-transcript-common/jobs/jobMeta";
@@ -43,19 +49,38 @@ export async function readSiteReportRows(paths: Paths, site: Site): Promise<Repo
// as /jobs lists them. A prepare older than that is not linked.
const PREPARE_JOB_LOOKBACK = 100;
-// The newest `reports-prepare` job for this site (its spec's slug is the site
-// id — reportsPrepareAction.ts), live or from an earlier server lifetime, or
-// null.
-export async function latestReportsPrepareJob(
+// The newest job of `kind` for this site (a reports job's spec's slug is the
+// site id — reportsPrepareAction.ts, reportsExportAction.ts), live or from an
+// earlier server lifetime, or null.
+export async function latestSiteReportsJob(
paths: Paths,
siteId: string,
+ kind: "reports-prepare" | "reports-export",
): Promise<JobListEntry | null> {
const page = await listAllJobs(paths, { limit: PREPARE_JOB_LOOKBACK });
const registry = getRegistry();
for (const entry of page.entries) {
- if (entry.kind !== "reports-prepare") continue;
+ if (entry.kind !== kind) continue;
const spec = registry.get(entry.id)?.spec ?? (await readJobMeta(paths, entry.id))?.spec;
if (spec?.slug === siteId) return entry;
}
return null;
}
+
+// One published report's exports, as the tab shows them: the manifest of the
+// last export (or null), and which files compose would publish now.
+export type ReportExportsRow = {
+ reportId: string;
+ manifest: ReportExportManifest | null;
+ publishable: PublishableReportExports;
+};
+
+export async function readSiteReportExports(paths: Paths, site: Site): Promise<ReportExportsRow[]> {
+ return Promise.all(
+ (site.reports ?? []).map(async (reportId) => ({
+ reportId,
+ manifest: await readReportExportManifest(paths, site.siteId, reportId),
+ publishable: await publishableReportExports(paths, site.siteId, reportId),
+ })),
+ );
+}
diff --git a/editor/app/sites/lib/reportsExportAction.ts b/editor/app/sites/lib/reportsExportAction.ts
@@ -0,0 +1,90 @@
+"use server";
+
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { isValidSiteId, listSiteIds } from "yt-dlp-transcript-common/lib/site";
+import { isReportId } from "yt-dlp-transcript-common/lib/report/schema";
+import {
+ REPORT_EXPORT_FORMATS,
+ type ReportExportFormat,
+} from "yt-dlp-transcript-common/lib/report/views";
+import {
+ runManagedFunction,
+ type StreamActionResult,
+} from "yt-dlp-transcript-common/jobs/streamCommand";
+import {
+ exportSiteReports,
+ formatReportExportProblems,
+} from "yt-dlp-transcript-common/publish/reportExports";
+import { REPORTS_PREPARE_QUEUE } from "./reportsQueue";
+
+// EXPORT A REPORT SITE'S REPORTS as a job: each published report as
+// report.html, report.pdf, report.md and an evidence pack, into
+// `.export-index/sites/<siteId>/report-exports/<reportId>/`, where compose
+// publishes them from. The work is publish/reportExports.ts's, the same
+// `archilyzer reports export` runs. The job FAILS when an export cannot be
+// written (the log names each); a PDF skipped for want of a browser is a
+// note in the log, not a failure.
+//
+// On prepare's queue: it reads the media cache a prepare writes and prunes.
+// The spec's `slug` is the SITE id; the replay handler reads `params`.
+export async function reportsExportAction(
+ siteId: string,
+ opts: { reportId?: string; formats?: string[] } = {},
+): Promise<StreamActionResult> {
+ const paths = getPaths();
+ const id = siteId.trim();
+ if (!isValidSiteId(id)) {
+ return { ok: false, error: `"${id}" is not a valid site id` };
+ }
+ if (!listSiteIds(paths).includes(id)) {
+ return { ok: false, error: `No site "${id}"` };
+ }
+ if (opts.reportId !== undefined && !isReportId(opts.reportId)) {
+ return { ok: false, error: `"${opts.reportId}" is not a report id` };
+ }
+ let formats: ReportExportFormat[] | undefined;
+ if (opts.formats !== undefined) {
+ const bad = opts.formats.filter(
+ (f) => !(REPORT_EXPORT_FORMATS as readonly string[]).includes(f),
+ );
+ if (bad.length > 0 || opts.formats.length === 0) {
+ return {
+ ok: false,
+ error: `formats are some of ${REPORT_EXPORT_FORMATS.join(", ")}${bad.length ? ` (not ${bad.join(", ")})` : ""}`,
+ };
+ }
+ formats = REPORT_EXPORT_FORMATS.filter((f) => opts.formats!.includes(f));
+ }
+ return runManagedFunction({
+ kind: "reports-export",
+ queueKey: REPORTS_PREPARE_QUEUE,
+ paths,
+ spec: {
+ kind: "reports-export",
+ slug: id,
+ params: {
+ siteId: id,
+ ...(opts.reportId ? { reportId: opts.reportId } : {}),
+ ...(formats ? { formats } : {}),
+ },
+ },
+ fn: async (onLog, signal) => {
+ const result = await exportSiteReports({
+ siteId: id,
+ paths,
+ onLog,
+ signal,
+ ...(opts.reportId ? { reportId: opts.reportId } : {}),
+ ...(formats ? { formats } : {}),
+ });
+ for (const r of result.exported) {
+ for (const note of r.manifest.notes) onLog(`note: ${r.reportId}: ${note}`);
+ }
+ if (result.problems.length === 0) return;
+ for (const line of formatReportExportProblems(result.problems)) onLog(line);
+ throw new Error(
+ `${result.problems.length} problem(s): the reports were not all exported`,
+ );
+ },
+ });
+}
diff --git a/editor/app/sites/lib/reportsPrepareAction.ts b/editor/app/sites/lib/reportsPrepareAction.ts
@@ -10,16 +10,20 @@ import {
formatReportMediaProblems,
prepareReportMedia,
} from "yt-dlp-transcript-common/publish/reportMedia";
-
-// Its own queue: one prepare at a time, waiting on no download and no build.
-const REPORTS_PREPARE_QUEUE = "reports-prepare";
+import {
+ exportAfterPrepare,
+ formatReportExportProblems,
+} from "yt-dlp-transcript-common/publish/reportExports";
+import { REPORTS_PREPARE_QUEUE } from "./reportsQueue";
// PREPARE A REPORT SITE'S EVIDENCE MEDIA as a job: every clip its published
// reports cite, cut from the media on disk, and every cited post capture,
// copied into `.export-index/sites/<siteId>/report-media/` before its build.
// The work is publish/reportMedia.ts's, the same `archilyzer reports prepare`
// runs. The job FAILS when any citation lacks its media — the log names each
-// one — and the manifest it writes carries the same list.
+// one — and the manifest it writes carries the same list. When nothing is
+// missing it ends by exporting the reports as files (reportsExportAction's
+// work, publish/reportExports.ts), and fails when an export does.
//
// The spec's `slug` is the SITE id (a spec needs one, and this job belongs to
// no channel); the replay handler reads `params.siteId`.
@@ -41,10 +45,20 @@ export async function reportsPrepareAction(
spec: { kind: "reports-prepare", slug: id, params: { siteId: id } },
fn: async (onLog, signal) => {
const index = await prepareReportMedia({ siteId: id, paths, onLog, signal });
- if (index.problems.length === 0) return;
- for (const line of formatReportMediaProblems(index.problems)) onLog(line);
+ if (index.problems.length > 0) {
+ for (const line of formatReportMediaProblems(index.problems)) onLog(line);
+ throw new Error(
+ `${index.problems.length} problem(s): the site's evidence media is not complete`,
+ );
+ }
+ const exported = await exportAfterPrepare(index, { siteId: id, paths, onLog, signal });
+ for (const r of exported?.exported ?? []) {
+ for (const note of r.manifest.notes) onLog(`note: ${r.reportId}: ${note}`);
+ }
+ if (!exported || exported.problems.length === 0) return;
+ for (const line of formatReportExportProblems(exported.problems)) onLog(line);
throw new Error(
- `${index.problems.length} problem(s): the site's evidence media is not complete`,
+ `${exported.problems.length} problem(s): the reports were not all exported`,
);
},
});
diff --git a/editor/app/sites/lib/reportsQueue.ts b/editor/app/sites/lib/reportsQueue.ts
@@ -0,0 +1,4 @@
+// The queue a report site's prepare and export jobs share: one at a time,
+// waiting on no download and no build — and never both at once, since an
+// export reads the media cache a prepare writes and prunes.
+export const REPORTS_PREPARE_QUEUE = "reports-prepare";
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **A report can be saved whole: as one HTML page, a PDF, Markdown, or an evidence pack.** A report page's download line now reads HTML · PDF · Markdown · Evidence pack · Citations JSON · CSV, each listed only when the site publishes it. The HTML is one file that opens with no network: the report with its verdicts, the document's sentences and the post screenshots inside it, numbered citations, and a reference list giving each quote's speaker, date, record, the original at its time and the moment page on the site. The PDF is that page printed. The Markdown is the same report as plain text with numbered references. The evidence pack is a zip of the page with its clips, stills and screenshots beside it, so the clips play offline. Each ends with a line naming the report's date and the start of its checksum. Needs `reports export` (or prepare) and a rebuild and deploy of each site with reports.
- **A report's claim can carry a flag, its title a byline, and a site with one report names it in the browser tab.** `report.json` claim `flag` (one line, at most 60 characters) shows as a small pill in the accent colour beside the claim's verdict, e.g. "No source given". On a report-only site with one report, the home page's tab title is the report's, as on the report's own page, where it was the site's title alone. A report's page follows its title with a byline from the document under review, "by <author> · <publisher>", in the same line when it fits; the publisher (or, with none, the author) links to the document. Under it, a step apart, the page names its own: "Fact-check by <site title>" ("Report by …"), then the dates, then the subtitle; the kind's label no longer sits above the title on the report's page. A claim's sentence from the document under review no longer links up to the document's box: it sits on a rail in the document's colour, and the box's left edge wears the same colour (a source's `accent`, `"#rrggbb"`; without one, the border colour). A sentence of another document keeps its "from <title>" link. A report's citation can say where its evidence came from (`origin`: `"subject"`, the document under review gave it; `"added"`, the report's author found it). A claim lists what the report added first, each card marked with the Archilyzer mark and "Not in the article" ("Not in the source"), then evidence of unknown origin, then what the document gave itself folded under "In the article (n)"; the reference list marks an added citation with the mark alone, and a claim's flag pill wears the same mark. A fact-check's page reads in three tiers, each opened by a hairline with one, two or three dots and its reading time (at 230 words a minute): the quick take (the tally, the summary, and links to what the check found, every claim and the downloads); **What the check found**, every ruled claim grouped by verdict (contradicted, not found, partly, untestable, corroborated), one line each linking to the claim, with its `gist` (a new optional claim field, one line, at most 240 characters) and its flag; and **Every claim, with its evidence**, which opens with **How it was checked** (`method`, a new optional report field in markdown). A report of kind `sweep` has the first and last tiers only. Needs a rebuild and deploy of the site.
- **Forum posts read like the other posts.** A post from a forum-thread channel shows its place in the thread (#N), an "edited" mark, the thread's title and its media as links, and opening its thread shows its conversation — the posts it quotes and the posts quoting it — rather than the whole forum thread.
- **A report-only site with one report opens on that report.** Its home page is the report itself, its header links nothing, and `/reports/` forwards home: there is no index of one. With more reports the home page is the list, without repeating the site's title under the header; a list entry is the report's name, subtitle and dates (its counts and tally are on its page). Pages a report-only site does not have link home.
diff --git a/export/app/components/reports/ReportArticle.tsx b/export/app/components/reports/ReportArticle.tsx
@@ -29,10 +29,20 @@ import { ArchiveList, ReportName, SourceBlock, dateLabel, textLink } from "./par
// its verdict and flag, the document's own sentence (its still), the findings with
// their inline citations, and the evidence cards (what the report added first,
// marked; what the document gave itself folded away last) — then the numbered
-// reference list the inline markers jump to, and the citations as files.
+// reference list the inline markers jump to, and the downloads: the report as
+// files (HTML, PDF, Markdown, the evidence pack) and its citations as data.
const proseClass = "text-[0.95rem] leading-relaxed text-foreground";
+const DOWNLOADS: readonly (readonly [keyof NonNullable<ReportPageView["downloads"]>, string])[] = [
+ ["html", "HTML"],
+ ["pdf", "PDF"],
+ ["md", "Markdown"],
+ ["zip", "Evidence pack"],
+ ["json", "Citations JSON"],
+ ["csv", "CSV"],
+];
+
// A claim's sentence in a quoted document: its still (or its words), on a rail
// in the document's colour (its `accent`, else the border colour) — the same
// colour as the document's box, so the sentences and the box read as one
@@ -251,12 +261,18 @@ export default function ReportArticle({ view, siteTitle }: { view: ReportPageVie
view.updated && view.updated !== view.published ? `Updated ${dateLabel(view.updated)}` : null,
].filter(Boolean);
const claimCount = view.sections.reduce((n, s) => n + s.claims.length, 0);
+ // The report as files (its exports), then its citations as data: only what
+ // compose published.
+ const downloads = DOWNLOADS.flatMap(([key, label]) => {
+ const href = view.downloads?.[key];
+ return href ? [[key, label, href] as const] : [];
+ });
const groups = isFactcheck ? foundGroups(view) : [];
const minutes = reportTierMinutes(view);
const jumps = [
...(groups.length > 0 ? [{ href: "#found", label: "What the check found" }] : []),
{ href: "#claims", label: "Every claim" },
- ...(view.downloads?.json || view.downloads?.csv ? [{ href: "#downloads", label: "Downloads" }] : []),
+ ...(downloads.length > 0 ? [{ href: "#downloads", label: "Downloads" }] : []),
];
// Documents quoted that are not the subject, each shown once before the
// references.
@@ -414,25 +430,22 @@ export default function ReportArticle({ view, siteTitle }: { view: ReportPageVie
References
</h2>
<ReferenceList citations={references} addedLabel={`Not in the ${subjectNoun}`} />
- {/* The jump links' "Downloads": the citation files follow the references. */}
- <span id="downloads" className="scroll-mt-20" />
</section>
)}
- {(view.downloads?.json || view.downloads?.csv) && (
- <p data-citation-downloads="" className="flex flex-wrap items-center gap-3 text-sm text-muted-foreground">
+ {downloads.length > 0 && (
+ <p
+ id="downloads"
+ data-citation-downloads=""
+ className="flex scroll-mt-20 flex-wrap items-center gap-x-3 gap-y-1 text-sm text-muted-foreground"
+ >
<ArrowDownToLine className="size-4" aria-hidden />
- Download citations:
- {view.downloads.json && (
- <a href={view.downloads.json} download className={textLink}>
- JSON
- </a>
- )}
- {view.downloads.csv && (
- <a href={view.downloads.csv} download className={textLink}>
- CSV
+ Download:
+ {downloads.map(([key, label, href]) => (
+ <a key={key} href={href} download data-download={key} className={textLink}>
+ {label}
</a>
- )}
+ ))}
</p>
)}
</article>
diff --git a/export/e2e-report/audit.spec.ts b/export/e2e-report/audit.spec.ts
@@ -28,6 +28,10 @@ test("it holds the reports, the moments, the cited media and the contract", () =
"reports/demo-factcheck/page.json",
"reports/demo-factcheck/citations.json",
"reports/demo-factcheck/citations.csv",
+ "reports/demo-factcheck/report.html",
+ "reports/demo-factcheck/report.pdf",
+ "reports/demo-factcheck/report.md",
+ "reports/demo-factcheck/evidence-pack.zip",
"reports/demo-factcheck/stills/a01.png",
"m/index.json",
"m/demo-channel/abc123/3126.00-3151.00/index.html",
diff --git a/export/e2e-report/contract.ts b/export/e2e-report/contract.ts
@@ -1,6 +1,8 @@
// The cited fixture site's compose step (stage.ts runs it, in a child process
// with the stage's environment): what compose's reports stage writes beside
-// the views — the citation downloads — and the cited site's contract, by
+// the views — the citation downloads and the report's exports (HTML, PDF,
+// Markdown, evidence pack, by publish/reportExports.ts's writer) — and the
+// cited site's contract, by
// compose's own functions (bin/compose-site.ts emitFederationFiles /
// emitAiFiles), so the stage's public/ is what compose would leave for this
// site. The view JSON, stills and media are the fixture's, already in place.
@@ -8,8 +10,9 @@
// Every import is dynamic: getPaths() reads the environment on first use.
import fs from "node:fs";
+import os from "node:os";
import path from "node:path";
-import type { MomentIndexView, ReportIndexView, ReportPageView } from "yt-dlp-transcript-common/lib/report/views";
+import type { MomentIndexView, MomentPageView, ReportIndexView, ReportPageView } from "yt-dlp-transcript-common/lib/report/views";
const { getPaths } = await import("yt-dlp-transcript-common/lib/paths");
const { getSite } = await import("yt-dlp-transcript-common/lib/site");
@@ -18,7 +21,12 @@ const { citationSet, citationsCsv } = await import("yt-dlp-transcript-common/pub
const { MOMENTS_INDEX_PATH, REPORTS_INDEX_PATH, reportCitationsDownloadPath, reportViewPath } = await import(
"yt-dlp-transcript-common/lib/report/views"
);
-const { readFixtureReport } = await import("../fixtures/report-site/fixture");
+const { FIXTURE_DIR, readFixtureReport } = await import("../fixtures/report-site/fixture");
+const { openPlaywrightPdfPrinter, writeReportExports } = await import("yt-dlp-transcript-common/publish/reportExports");
+const { sha256Hex } = await import("yt-dlp-transcript-common/publish/reportExportFiles");
+const { REPORT_EXPORT_FILENAMES, REPORT_EXPORT_FORMATS, momentViewPath, reportExportDownloadPath } = await import(
+ "yt-dlp-transcript-common/lib/report/views"
+);
const paths = getPaths();
const siteId = process.env.SITE_ID;
@@ -39,6 +47,44 @@ for (const entry of index.reports) {
`${JSON.stringify(citationSet(report, view), null, 2)}\n`,
);
fs.writeFileSync(pub(reportCitationsDownloadPath(entry.id, "csv")), citationsCsv(view));
+
+ // The report's exports, by the host step's own writer, from the files
+ // already in public/ (its stills, the post's shot, the clips), published
+ // where compose publishes them.
+ const printer = openPlaywrightPdfPrinter();
+ const dir = path.join(fs.mkdtempSync(path.join(os.tmpdir(), "report-site-exports-")), entry.id);
+ try {
+ const { result, problems } = await writeReportExports({
+ dir,
+ site,
+ report,
+ view,
+ reportSha256: sha256Hex(fs.readFileSync(path.join(FIXTURE_DIR, "source", entry.id, "report.json"))),
+ files: {
+ local: (p) => (fs.existsSync(pub(p)) ? pub(p) : undefined),
+ clip: (c) => {
+ const m = readJson<MomentPageView>(momentViewPath(c.moment));
+ return m.clip ? { sitePath: m.clip.src, file: pub(m.clip.src), kind: m.kind === "audio" ? "audio" : "video" } : undefined;
+ },
+ },
+ ffmpegBin: paths.ffmpegBin,
+ formats: REPORT_EXPORT_FORMATS,
+ printer: () => printer,
+ zipBin: "zip",
+ now: new Date(),
+ log: console.log,
+ });
+ if (problems.length > 0 || result.manifest.notes.length > 0) {
+ throw new Error(`contract.ts: the fixture's exports: ${JSON.stringify([...problems, ...result.manifest.notes])}`);
+ }
+ for (const f of REPORT_EXPORT_FORMATS) {
+ fs.copyFileSync(path.join(dir, REPORT_EXPORT_FILENAMES[f]), pub(reportExportDownloadPath(entry.id, f)));
+ }
+ } finally {
+ const p = await printer;
+ if (!("missing" in p)) await p.close();
+ fs.rmSync(path.dirname(dir), { recursive: true, force: true });
+ }
}
await emitFederationFiles(site, paths);
diff --git a/export/e2e-report/report-site.spec.ts b/export/e2e-report/report-site.spec.ts
@@ -213,16 +213,62 @@ test("a citation's number jumps to its entry in the reference list", async ({ pa
await expect(page.locator("#c-w01")).toContainText("Opened to traffic: April 2019.");
});
-test("the report's citations download as JSON and CSV", async ({ page, request }) => {
+test("the report downloads as HTML, PDF, Markdown and an evidence pack, its citations as JSON and CSV", async ({
+ page,
+ request,
+}) => {
await page.goto(REPORT);
const links = page.locator("[data-citation-downloads] a[download]");
- await expect(links).toHaveCount(2);
- for (const href of await links.evaluateAll((els) => els.map((el) => el.getAttribute("href")!))) {
+ await expect(links).toHaveText(["HTML", "PDF", "Markdown", "Evidence pack", "Citations JSON", "CSV"]);
+ const hrefs = await links.evaluateAll((els) => els.map((el) => el.getAttribute("href")!));
+ expect(hrefs).toEqual([
+ `${REPORT}report.html`,
+ `${REPORT}report.pdf`,
+ `${REPORT}report.md`,
+ `${REPORT}evidence-pack.zip`,
+ `${REPORT}citations.json`,
+ `${REPORT}citations.csv`,
+ ]);
+ for (const href of hrefs) {
const res = await request.get(href);
expect(res.status(), href).toBe(200);
- if (href.endsWith(".json")) expect((await res.json()).format).toBe("archilyzer-citations");
- else expect(await res.text()).toContain("c01");
+ const body = await res.body();
+ if (href.endsWith("citations.json")) expect(JSON.parse(body.toString()).format).toBe("archilyzer-citations");
+ else if (href.endsWith(".csv")) expect(body.toString()).toContain("c01");
+ else if (href.endsWith(".html")) expect(body.toString()).toContain("<title>Checking an example article</title>");
+ else if (href.endsWith(".pdf")) expect(body.subarray(0, 5).toString()).toBe("%PDF-");
+ else if (href.endsWith(".md")) expect(body.toString()).toContain("# Checking an example article");
+ else expect(body.subarray(0, 4).toString("hex")).toBe("504b0304");
+ }
+});
+
+test("report.html opens with no network: one file, no script, every image inlined", async ({ page, baseURL }) => {
+ const url = `${baseURL}${REPORT}report.html`;
+ const asked: string[] = [];
+ // Only the file itself may load; anything else it asked for would be refused.
+ await page.route("**/*", (route) => {
+ const u = route.request().url();
+ if (u === url) return route.continue();
+ asked.push(u);
+ return route.abort();
+ });
+ await page.goto(url);
+ await expect(page.locator("h1")).toContainText("Checking an example article");
+ await expect(page.locator("article[data-claim]")).toHaveCount(5);
+ await expect(page.locator("script")).toHaveCount(0);
+ const imgs = page.locator("img");
+ expect(await imgs.count()).toBeGreaterThan(0);
+ for (const img of await imgs.all()) {
+ expect(await img.getAttribute("src")).toMatch(/^data:image\//);
+ await expect.poll(() => img.evaluate((el) => (el as HTMLImageElement).naturalWidth)).toBeGreaterThan(0);
}
+ await expect(page.locator("[data-reference-list] > li")).toHaveCount(6);
+ // The page's tiers, its flag pill and the added mark, as the site has them.
+ await expect(page.locator("[data-report-tier]")).toHaveCount(3);
+ await expect(page.locator("[data-claim-flag]").first()).toContainText("No source given");
+ await expect(page.locator("[data-citation-added] svg").first()).toBeVisible();
+ await expect(page.locator("[data-export-footer]")).toContainText("report sha256");
+ expect(asked).toEqual([]);
});
test("a video citation opens its moment: the clip, the cue lines with the span marked, where it is cited", async ({
diff --git a/export/e2e-report/stage.ts b/export/e2e-report/stage.ts
@@ -19,9 +19,11 @@
// THE PUBLIC DIR is what compose's reports stage would write for the site:
// - the fixture report site (export/fixtures/report-site/public): the view
// JSON R4's builder made, the stills, the post's shot, the evidence clips;
-// - the citation downloads and the cited site's contract (site.json,
+// - the citation downloads, the report's exports (report.html, report.pdf,
+// report.md, evidence-pack.zip) and the cited site's contract (site.json,
// corpus.json, llms.txt, robots.txt, sitemap.xml, _headers), written by
-// compose's own functions in a child process (contract.ts);
+// compose's and the export step's own functions in a child process
+// (contract.ts);
// - export/public's checked-in assets (git-tracked files only — the
// checkout's composed corpus is never read).
//
diff --git a/export/fixtures/report-site/fixture.ts b/export/fixtures/report-site/fixture.ts
@@ -29,7 +29,9 @@ import {
evidenceClipPath,
momentViewPath,
reportCitationsDownloadPath,
+ reportExportDownloadPath,
reportIndexEntry,
+ REPORT_EXPORT_FORMATS,
reportViewPath,
type MomentPageView,
type RecordView,
@@ -101,7 +103,10 @@ export function buildFixtureReportView(report = readFixtureReport()): ReportPage
text: `${c.quote}\n\nPosted to settle it.`,
shot: shotFor(c),
}),
+ // Every download compose lists when the report was exported (the e2e
+ // stage writes the exports themselves: e2e-report/contract.ts).
downloads: {
+ ...Object.fromEntries(REPORT_EXPORT_FORMATS.map((f) => [f, reportExportDownloadPath(report.id, f)])),
json: reportCitationsDownloadPath(report.id, "json"),
csv: reportCitationsDownloadPath(report.id, "csv"),
},
diff --git a/export/fixtures/report-site/public/reports/demo-factcheck/page.json b/export/fixtures/report-site/public/reports/demo-factcheck/page.json
@@ -224,6 +224,10 @@
}
],
"downloads": {
+ "html": "/reports/demo-factcheck/report.html",
+ "pdf": "/reports/demo-factcheck/report.pdf",
+ "md": "/reports/demo-factcheck/report.md",
+ "zip": "/reports/demo-factcheck/evidence-pack.zip",
"json": "/reports/demo-factcheck/citations.json",
"csv": "/reports/demo-factcheck/citations.csv"
}
diff --git a/export/scripts/serve-out.mjs b/export/scripts/serve-out.mjs
@@ -53,6 +53,7 @@ const TYPES = {
".srt": "text/plain; charset=utf-8",
".md": "text/markdown; charset=utf-8",
".zip": "application/zip",
+ ".pdf": "application/pdf",
".wasm": "application/wasm",
};
diff --git a/plans/report-sites.md b/plans/report-sites.md
@@ -124,6 +124,39 @@ site with search off is report-only — what `publish: "cited"` was. A legacy `p
- The existing guard test that forbids publish code from naming `posts-media` is amended to allow exactly the
one module that copies CITED captures, with a test that it copies nothing else.
+## Exports
+
+A report must survive a takedown as files anyone can save and host again (slice RX).
+
+- `archilyzer reports export <siteId> [--report <id>] [--formats html,pdf,md,zip] [--allow-missing-media]`
+ and the editor's `reports-export` job (prepare's queue; Reports tab → **Export reports**) run
+ `common/publish/reportExports.ts`; `reports prepare` runs it at its end when nothing is missing. It resolves
+ the reports exactly as compose does (`resolveSiteReports`) and writes, per report, to
+ `.export-index/sites/<id>/report-exports/<reportId>/`:
+ - `report.html` — ONE file from `common/lib/report/exportHtml.ts` (pure view → HTML): inline CSS, no script;
+ stills and post screenshots recompressed on the host (WebP, else JPEG, ≤ 1200 px wide) into data URIs; clips
+ linked on the site, never inlined. The heading mirrors the site: the series on its own line in the accent,
+ the title with the reviewed document's byline inline, "Fact-check by <site title>", the dates, the subtitle.
+ Then the tally, sections → claims (verdict chip, the document's sentence as its still, findings with `[n]`
+ markers), the documents quoted with their archive links, and the numbered references (quote, speaker, date,
+ record, the original at its time, archive links, the moment page and clip when the site has a `siteUrl`).
+ - `report.pdf` — that HTML printed by headless Chromium (`importPlaywright`, A4); skipped with a note where
+ Playwright or its browser is missing, never a failure.
+ - `report.md` — plain Markdown (`exportMarkdown.ts`), `[label](cite:id)` → `label [n]`, numbered references.
+ - `evidence-pack.zip` — `<reportId>/report.html` playing its own `media/` (clips, stills, screenshots),
+ `report.md`, `citations.json`/`.csv`; packed by the system `zip` with sorted names, fixed times and modes
+ (the same report checked at the same time packs to the same bytes). No `zip` fails that format, naming it.
+ - `export.json` — each file's size and sha256, the sha256 of the report.json it was made from, the footer,
+ notes. Each run replaces the directory: nothing is left from another version.
+- Every export ends with the footer `Revision N · <date> · report sha256 <first 12> · commit <short>`; an unknown
+ part is left out. Today: the report's date and hash. The revision history (slice RH) fills `revision` and
+ `commit` in ONE place, `exportFooterFor` (`reportExports.ts`).
+- Compose (`publishableReportExports`, `common/publish/reportExportFiles.ts`) copies `report.html`, `report.pdf`,
+ `report.md` and the pack into `public/reports/<id>/` only when made from the report.json as it is now and each
+ is within the shared publish limit (`PUBLISH_MAX_FILE_BYTES`, 24 MiB, `lib/builtExport.ts`); a pack over it
+ stays local. The view's `downloads` lists what was published and the report page's download line shows HTML ·
+ PDF · Markdown · Evidence pack · Citations JSON · CSV. `reports/` is allowed wholesale by the cited audit.
+
## Compose and the contract
- `compose-site` gains a reports stage: resolve citations against the shared transcript trees, verify quotes,
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -41,6 +41,7 @@
// pnpm ops build-homepage --json '{"deploy":true}' --wait
// pnpm ops deploy-homepage --json '{"preview":"refresh"}' --wait
// pnpm ops reports-prepare --json '{"siteId":"demo-site"}' --wait
+// pnpm ops reports-export --json '{"siteId":"demo-site","formats":["html","md"]}' --wait
// pnpm ops get channel the-quartering
// pnpm ops tags --json '{"op":"define","tag":{"id":"eva-collab","label":"Collab"}}'
// pnpm ops tag-videos --file ids.json
@@ -132,6 +133,9 @@ const ACTIONS = [
// A report site's evidence media: cut every cited clip and copy every cited
// post capture into the site's report-media cache ({siteId}).
"reports-prepare",
+ // A report site's exports: each published report as HTML, PDF, Markdown and
+ // an evidence pack ({siteId, reportId?, formats?}).
+ "reports-export",
"lane",
// The curated-tag writers. `tags` edits the vocabulary (define/remove);
// `tag-videos` pins, unpins, suppresses or unsuppresses one tag over a batch
@@ -328,6 +332,11 @@ export function usage() {
'reports-prepare cuts every clip and copies every post capture a site\'s',
' published reports cite into its report-media cache, before its build:',
' {"siteId"}. The job fails, naming each one, when a citation lacks media.',
+ ' When nothing is missing it then exports the reports, as reports-export.',
+ "",
+ 'reports-export writes each published report as report.html, report.pdf,',
+ ' report.md and evidence-pack.zip for the site\'s build to publish:',
+ ' {"siteId", "reportId"?, "formats"?: ["html","pdf","md","zip"]}.',
"",
'retry-bucket runs one bucket of a channel\'s report as one job, past any',
' lane hold: {"slug", "bucket"}. "ids": [...] runs only those videos, and',
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -418,6 +418,14 @@ test("reports-prepare is a POST to its route, named in the usage", () => {
assert.match(usage(), /reports-prepare cuts every clip/);
});
+test("reports-export is a POST to its route, named in the usage", () => {
+ const p = parseArgs(["reports-export", "--json", '{"siteId":"demo-site","formats":["md"]}']);
+ assert.equal(p.method, "POST");
+ assert.equal(p.path, "/api/ops/reports-export");
+ assert.deepEqual(p.body, { siteId: "demo-site", formats: ["md"] });
+ assert.match(usage(), /reports-export writes each published report/);
+});
+
test("persist-videos is a POST to its route, named in the usage", () => {
const p = parseArgs([
"persist-videos",