Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit c5c067b5259e16380334d8a123a6f712f9fcbfe3
parent d4b2cd48ec0eaaeec4c74bf65cc467830198ea87
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 13:59:05 -0400

reports: tests for the exports — renderers, the host step, compose's publish gate

The HTML export is deterministic, has no script, and inlines every image as a
data URI; the heading, verdicts, numbered markers, references and footer are
held; the Markdown export numbers its references. The host step writes every
format from the resolved view, packs the same report checked at the same time
to the same bytes, skips the PDF with a note when there is no browser, and
fails the pack, naming it, when there is no zip. Compose publishes exports
and lists them as downloads, but not an export of another report.json, and
not a file over the 24 MiB publish limit.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/lib/builtExport.test.ts | 13+++++++++++++
Acommon/lib/report/export.test.ts | 197+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/publish/composeReports.test.ts | 136+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/publish/reportExports.ts | 5+++--
4 files changed, 349 insertions(+), 2 deletions(-)

diff --git a/common/lib/builtExport.test.ts b/common/lib/builtExport.test.ts @@ -13,7 +13,10 @@ import { builtSiteProblem, citedBuildProblem, deployAudienceProblem, + PAGES_MAX_FILE_BYTES, PAGES_MAX_FILES, + PUBLISH_MAX_FILE_BYTES, + publishFileSizeProblem, } from "./builtExport"; function tempOut(siteJson?: string): { dir: string; cleanup: () => void } { @@ -390,3 +393,13 @@ test("a site configured cited with a full build is refused at deploy; a cited bu cited.cleanup(); } }); + +test("the publish limit: 24 MiB, inside Pages' 25, one sentence naming the file over it", () => { + assert.equal(PUBLISH_MAX_FILE_BYTES, 24 * 1024 * 1024); + assert.ok(PUBLISH_MAX_FILE_BYTES < PAGES_MAX_FILE_BYTES); + assert.equal(publishFileSizeProblem("evidence-pack.zip", PUBLISH_MAX_FILE_BYTES), null); + assert.equal( + publishFileSizeProblem("evidence-pack.zip", PUBLISH_MAX_FILE_BYTES + 1), + "evidence-pack.zip is 24.0 MiB, over the publish limit of 24.0 MiB (Pages allows 25 MiB per file)", + ); +}); diff --git a/common/lib/report/export.test.ts b/common/lib/report/export.test.ts @@ -0,0 +1,197 @@ +// The report exports' renderers (exportHtml.ts, exportMarkdown.ts), from a +// fixture view: deterministic, self-contained, numbered. +// +// Run with: node_modules/.bin/tsx --test lib/report/export.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import type { Report } from "./schema"; +import { buildReportPageView, type RecordView } from "./views"; +import { + markdownToHtml, + reportExportFooterLine, + reportExportHtml, + subjectByline, + type ReportExportHtmlOptions, +} from "./exportHtml"; +import { citedMarkdownToPlain, mdText, reportExportMarkdown } from "./exportMarkdown"; + +const report: Report = { + format: "archilyzer-report", + version: 1, + id: "demo", + kind: "factcheck", + series: "Demo Checks", + title: "A demo fact-check", + subtitle: "Two claims, tested", + summary: "It starts [here](cite:v1). <script>alert(1)</script> and [a link](javascript:alert(1)).", + published: "2026-10-01", + updated: "2026-10-04T10:00:00Z", + subject: { source: "s0" }, + verdicts: { PARTLY: { label: "Half" } }, + sources: { + s0: { + kind: "article", + title: "An article", + url: "https://example.org/a", + author: "A. Writer", + publisher: "Example Gazette", + archives: [{ label: "archive", url: "https://archive.example.org/a", context: "the first edition" }], + saved: "sources/s0/page.html", + }, + }, + citations: { + v1: { kind: "video", channel: "demo-channel", id: "abc123", start: 3126, end: 3151, quote: "one * two", verification: { quoteScore: 0.97, quoteCheckedAt: "2026-10-04T12:00:00Z" } }, + a1: { kind: "audio", channel: "demo-podcast", id: "ep-1", start: 10.5, end: 20, quote: "two", speaker: "Guest" }, + p1: { kind: "post", channel: "demo-social", id: "123", quote: "three" }, + s1: { kind: "source", source: "s0", quote: "four", image: "stills/s1.png" }, + w1: { kind: "page", url: "https://example.org/p", title: "A page", quote: "five", archiveUrl: "https://archive.example.org/p" }, + }, + sections: [ + { + id: "one", + title: "One", + claims: [ + { + id: "c1", + title: "One", + text: "Claim one.", + verdict: "CONTRADICTED", + sourceQuote: { citation: "s1" }, + findings: "See [this](cite:w1), **bold**, and `[code](cite:never)`.\n\n- a list item\n- another", + citations: ["v1", "a1"], + }, + ], + }, + { + id: "two", + title: "Two", + body: "A body citing [a post](cite:p1).", + claims: [{ id: "c2", text: "Claim two.", verdict: "PARTLY", citations: ["a1"] }], + }, + ], +}; + +const RECORDS: Record<string, RecordView> = { + "demo-channel/abc123": { channel: "demo-channel", channelTitle: "Demo Channel", id: "abc123", title: "Demo stream", date: "2026-01-10", originalUrl: "https://media.example.org/abc123?t=3126" }, + "demo-podcast/ep-1": { channel: "demo-podcast", channelTitle: "Demo Podcast", id: "ep-1", title: "Episode 1", date: "2026-02-02", originalUrl: "https://media.example.org/ep-1" }, + "demo-social/123": { channel: "demo-social", channelTitle: "Demo Social", id: "123", date: "2025-03-14", originalUrl: "https://social.example.org/123" }, +}; + +const view = buildReportPageView(report, { + record: (c) => RECORDS[`${c.channel}/${c.id}`], + post: () => ({ author: "@demo", text: "three", shot: "/media/posts/demo-social/123/shot.png" }), +}); + +const PNG = "data:image/png;base64,iVBORw0KGgo="; +const opts: ReportExportHtmlOptions = { + siteUrl: "https://reports.example.org/", + siteTitle: "Demo Site", + footer: { date: "2026-10-04", reportSha256: "0123456789abcdef".repeat(4) }, + image: () => PNG, + clip: (c) => ({ href: `https://reports.example.org/media/clips/${c.moment}.mp4`, kind: c.kind }), +}; + +test("the HTML export is deterministic, holds no script, and every image is a data: URI", () => { + const html = reportExportHtml(view, opts); + assert.equal(reportExportHtml(view, opts), html); + assert.doesNotMatch(html, /<script/i, "a report's raw HTML is escaped, and the export has no script"); + assert.match(html, /&lt;script&gt;alert\(1\)&lt;\/script&gt;/); + assert.doesNotMatch(html, /javascript:/, "a link no reader should follow is left as its label"); + const srcs = [...html.matchAll(/<img [^>]*src="([^"]+)"/g)].map((m) => m[1]); + assert.equal(srcs.length, 2, "the claim's still and the post's screenshot"); + for (const s of srcs) assert.ok(s.startsWith("data:image/"), s); + // No stylesheet, font or script is fetched. + assert.doesNotMatch(html, /<link |@import|url\(/); +}); + +test("the heading: the series on its own line, the title with the document's byline, then ours", () => { + const html = reportExportHtml(view, opts); + assert.match( + html, + /<h1><span class="series" data-report-series="">Demo Checks<\/span><span class="title">A demo fact-check <span class="by" data-report-byline="">by A\. Writer · <a href="https:\/\/example\.org\/a">Example Gazette<\/a><\/span><\/span><\/h1>/, + ); + assert.match(html, /<p class="site-line" data-report-kind="">Fact-check by Demo Site<\/p>\n<p class="dates">Published 2026-10-01 · Updated 2026-10-04<\/p>\n<p class="subtitle">Two claims, tested<\/p>/); + assert.match(html, /<title>Demo Checks: A demo fact-check<\/title>/); + assert.doesNotMatch(html, /class="eyebrow"/); +}); + +test("subjectByline: author and publisher; the author alone takes the link; neither is none", () => { + assert.deepEqual(subjectByline(view), { author: "A. Writer", publisher: "Example Gazette", url: "https://example.org/a" }); + const only = (s: object) => subjectByline({ subject: "s0", sources: { s0: { id: "s0", kind: "article", title: "t", archives: [], ...s } } }); + assert.deepEqual(only({ author: "A", url: "https://x.example" }), { author: "A", url: "https://x.example" }); + assert.equal(only({ url: "https://x.example" }), undefined); + assert.match(reportExportHtml({ ...view, sources: { s0: { ...view.sources.s0, publisher: undefined } } }, opts), /by <a href="https:\/\/example\.org\/a">A\. Writer<\/a>/); +}); + +test("verdicts, inline markers [n] to the reference list, and references with every link", () => { + const html = reportExportHtml(view, opts); + assert.match(html, /data-verdict="CONTRADICTED" style="--v:#[0-9a-f]{6}">/i); + assert.match(html, /data-verdict="PARTLY"[^>]*>Half/, "the report's override"); + assert.match(html, /here<sup class="cite"><a href="#c-v1">\[1\]<\/a><\/sup>/); + assert.match(html, /<code>\[code\]\(cite:never\)<\/code>/, "code is not citing"); + const refs = [...html.matchAll(/<li id="c-([a-z0-9]+)" value="(\d+)"/g)].map((m) => `${m[2]}:${m[1]}`); + assert.deepEqual(refs, ["1:v1", "2:s1", "3:w1", "4:a1", "5:p1"]); + // A span: the original at its time, the moment page and the clip on the site. + assert.match(html, /Original at 52:06: <a href="https:\/\/media\.example\.org\/abc123\?t=3126">/); + assert.match(html, /Moment page: <a href="https:\/\/reports\.example\.org\/m\/demo-channel\/abc123\/3126\.00-3151\.00\/">/); + assert.match(html, /Clip: <a href="https:\/\/reports\.example\.org\/media\/clips\//); + assert.doesNotMatch(html, /<video/, "a clip is linked, not played, in the one-file export"); + assert.match(html, /Archived: <a href="https:\/\/archive\.example\.org\/p">/); + assert.match(html, /quote match 97%/); + assert.match(html, /<footer class="export-footer" data-export-footer=""><p>2026-10-04 · report sha256 0123456789ab<\/p><p>Published at <a href="https:\/\/reports\.example\.org\/reports\/demo\/">/); +}); + +test("without a site URL nothing links to the site; the pack's clip plays in place", () => { + const html = reportExportHtml(view, { ...opts, siteUrl: undefined, siteTitle: undefined, clip: undefined }); + assert.doesNotMatch(html, /reports\.example\.org/); + assert.match(html, /<p class="site-line" data-report-kind="">Fact-check<\/p>/); + const pack = reportExportHtml(view, { ...opts, image: () => "media/x.png", clip: () => ({ href: "media/clips/a.mp4", kind: "video", play: true }) }); + assert.match(pack, /<video controls preload="none" src="media\/clips\/a\.mp4"><\/video>/); + assert.match(pack, /<img src="media\/x\.png"/); +}); + +test("the footer line names what is known and leaves the rest out", () => { + const sha = "f".repeat(64); + assert.equal(reportExportFooterLine({ reportSha256: sha }), "report sha256 ffffffffffff"); + assert.equal( + reportExportFooterLine({ revision: 3, date: "2026-10-05T09:00:00Z", reportSha256: sha, commit: "abcdef1234567890" }), + "Revision 3 · 2026-10-05 · report sha256 ffffffffffff · commit abcdef123456", + ); +}); + +test("markdownToHtml: paragraphs, lists, quotes, code, headings under the page's own", () => { + const html = markdownToHtml("# Head\n\nA *word* and __strong__.\n\n1. one\n2. two\n\n> quoted\n\n```\n<b>\n```", () => undefined); + assert.equal( + html, + "<h3>Head</h3>\n<p>A <em>word</em> and <strong>strong</strong>.</p>\n<ol><li>one</li><li>two</li></ol>\n" + + "<blockquote><p>quoted</p></blockquote>\n<pre><code>&lt;b&gt;</code></pre>", + ); + assert.equal(markdownToHtml("[site](/reports/x/)", () => undefined, { siteUrl: "https://s.example" }), '<p><a href="https://s.example/reports/x/">site</a></p>'); + assert.equal(markdownToHtml("[site](/reports/x/)", () => undefined), "<p>site</p>"); +}); + +test("the Markdown export: the heading lines, `label [n]` citations, numbered references, the footer", () => { + const md = reportExportMarkdown(view, opts); + assert.equal(reportExportMarkdown(view, opts), md); + assert.ok( + md.startsWith( + "**Demo Checks**\n\n# A demo fact-check\n\nby A. Writer · [Example Gazette](https://example.org/a)\n\n" + + "Fact-check by Demo Site · Published 2026-10-01 · Updated 2026-10-04\n\n*Two claims, tested*\n", + ), + md.slice(0, 300), + ); + assert.match(md, /It starts here \[1\]\./); + assert.match(md, /See this \[3\], \*\*bold\*\*/); + assert.match(md, /^### One$/m); + assert.match(md, /^\*\*Verdict: Contradicted\*\*$/m); + const refs = [...md.matchAll(/^(\d+)\. “(.*)”/gm)].map((m) => `${m[1]}:${m[2]}`); + assert.deepEqual(refs, ["1:one \\* two", "2:four", "3:five", "4:two", "5:three"]); + assert.match(md, /Moment page: <https:\/\/reports\.example\.org\/m\/demo-channel\/abc123\/3126\.00-3151\.00\/>/); + assert.match(md, /---\n\n2026-10-04 · report sha256 0123456789ab\n\nPublished at <https:\/\/reports\.example\.org\/reports\/demo\/>\n$/); +}); + +test("mdText escapes what Markdown would read; citedMarkdownToPlain numbers citations", () => { + assert.equal(mdText("# a *b* [c]"), "\\# a \\*b\\* \\[c\\]"); + assert.equal(citedMarkdownToPlain("x [y](cite:p1) [z](https://e.example)", view), "x y [5] [z](https://e.example)"); +}); diff --git a/common/publish/composeReports.test.ts b/common/publish/composeReports.test.ts @@ -551,3 +551,139 @@ test("a site with no reports ships none of the last site's", async () => { const corpus = readJson<{ reports?: unknown }>(pub("corpus.json")); assert.equal(corpus.reports, undefined); }); + +// ─── Exports (publish/reportExports.ts) ─── + +const { exportSiteReports } = await import("./reportExports"); +const { reportExportDir } = await import("./reportExportFiles"); +const { execFileSync } = await import("node:child_process"); +const { createHash } = await import("node:crypto"); +const { truncateSync } = await import("node:fs"); + +const sha = (file: string) => createHash("sha256").update(readFileSync(file)).digest("hex"); +const fakePrinter = (printed: string[]) => async () => ({ + print: async (html: string) => { + printed.push(html); + return Buffer.from("%PDF-1.4 fake"); + }, + close: async () => {}, +}); + +test("reports export: every format from the resolved view, a self-contained HTML, a deterministic pack", async () => { + const printed: string[] = []; + // `now` is the verification's time, which citations.json carries: the + // same report, checked at the same time, packs to the same bytes. + const now = () => new Date("2026-10-05T12:00:00Z"); + const r = await exportSiteReports({ siteId: "cited", paths, now, openPdfPrinter: fakePrinter(printed), onLog: () => {} }); + assert.deepEqual(r.problems, []); + const dir = reportExportDir(paths, "cited", REPORT); + assert.deepEqual(readdirSync(dir).sort(), ["evidence-pack.zip", "export.json", "report.html", "report.md", "report.pdf"]); + const html = readFileSync(path.join(dir, "report.html"), "utf8"); + assert.deepEqual(printed, [html], "the PDF is the HTML, printed"); + assert.doesNotMatch(html, /<script/i); + const imgs = [...html.matchAll(/<img [^>]*src="([^"]+)"/g)].map((m) => m[1]); + assert.equal(imgs.length, 2, "the claim's still and the post's screenshot"); + for (const src of imgs) assert.match(src, /^data:image\//); + // Verified as compose verifies, the clip linked on the site, never inlined. + assert.match(html, /quote match 100%/); + assert.match(html, /https:\/\/cited\.example\.test\/media\/clips\/demo-channel\/abc123\/10\.00-20\.00\.mp4/); + assert.doesNotMatch(html, /<video/); + const md = readFileSync(path.join(dir, "report.md"), "utf8"); + assert.match(md, /^1\. “He opened the bridge himself\.”/m); + + const manifest = readJson<{ reportSha256: string; files: Record<string, { bytes: number; sha256: string }>; notes: string[] }>( + path.join(dir, "export.json"), + ); + assert.equal(manifest.reportSha256, sha(path.join(paths.sitesDir, "cited", "reports", REPORT, "report.json"))); + assert.deepEqual(Object.keys(manifest.files), ["html", "pdf", "md", "zip"]); + assert.equal(manifest.files.html.sha256, sha(path.join(dir, "report.html"))); + assert.deepEqual(manifest.notes, []); + assert.match(html, new RegExp(`report sha256 ${manifest.reportSha256.slice(0, 12)}`)); + + // The pack: the HTML playing its own media, the Markdown, the citations. + const zip = path.join(dir, "evidence-pack.zip"); + const names = execFileSync("unzip", ["-Z1", zip], { encoding: "utf8" }).trim().split("\n"); + assert.deepEqual(names, [ + `${REPORT}/citations.csv`, + `${REPORT}/citations.json`, + `${REPORT}/media/clips/${VIDEOS}/abc123/10.00-20.00.mp4`, + `${REPORT}/media/clips/${VIDEOS}/def456/5.00-9.00.m4a`, + `${REPORT}/media/posts/${SOCIAL}/3kabc/shot.png`, + `${REPORT}/media/report/stills/a01.png`, + `${REPORT}/report.html`, + `${REPORT}/report.md`, + ]); + const packed = execFileSync("unzip", ["-p", zip, `${REPORT}/report.html`], { encoding: "utf8" }); + assert.match(packed, /<video controls preload="none" src="media\/clips\/demo-channel\/abc123\/10\.00-20\.00\.mp4">/); + assert.match(packed, /<img src="media\/report\/stills\/a01\.png"/); + + // The same report exports to the same bytes. + const before = { html: sha(path.join(dir, "report.html")), zip: sha(zip) }; + await exportSiteReports({ siteId: "cited", paths, now, openPdfPrinter: fakePrinter([]), onLog: () => {} }); + assert.deepEqual({ html: sha(path.join(dir, "report.html")), zip: sha(zip) }, before); +}); + +test("compose publishes the exports beside the report and lists them as downloads", async () => { + await compose("cited"); + for (const f of ["report.html", "report.pdf", "report.md", "evidence-pack.zip"]) { + assert.equal(sha(pub("reports", REPORT, f)), sha(path.join(reportExportDir(paths, "cited", REPORT), f)), f); + } + const view = readJson<{ downloads: Record<string, string> }>(pub("reports", REPORT, "page.json")); + assert.deepEqual(view.downloads, { + html: `/reports/${REPORT}/report.html`, + pdf: `/reports/${REPORT}/report.pdf`, + md: `/reports/${REPORT}/report.md`, + zip: `/reports/${REPORT}/evidence-pack.zip`, + json: `/reports/${REPORT}/citations.json`, + csv: `/reports/${REPORT}/citations.csv`, + }); + assert.equal(citedBuildProblem(paths.exportPublicDir), null, "the exports are within reports/, which the audit allows"); +}); + +test("an export of another version of the report is not published; a pack over the limit stays local", async () => { + const dir = reportExportDir(paths, "cited", REPORT); + const file = path.join(paths.sitesDir, "cited", "reports", REPORT, "report.json"); + const saved = readFileSync(file, "utf8"); + writeFileSync(file, `${saved}\n`); + try { + await compose("cited"); + assert.ok(!existsSync(pub("reports", REPORT, "report.html"))); + const view = readJson<{ downloads: Record<string, string> }>(pub("reports", REPORT, "page.json")); + assert.deepEqual(Object.keys(view.downloads), ["json", "csv"]); + } finally { + writeFileSync(file, saved); + } + truncateSync(path.join(dir, "evidence-pack.zip"), 25 * 1024 * 1024); + await compose("cited"); + assert.ok(existsSync(pub("reports", REPORT, "report.html"))); + assert.ok(!existsSync(pub("reports", REPORT, "evidence-pack.zip"))); + const view = readJson<{ downloads: Record<string, string> }>(pub("reports", REPORT, "page.json")); + assert.deepEqual(Object.keys(view.downloads), ["html", "pdf", "md", "json", "csv"]); +}); + +test("reports export: no browser skips the PDF with a note; no zip fails the pack, naming it", async () => { + const r = await exportSiteReports({ + siteId: "cited", + paths, + openPdfPrinter: async () => ({ missing: "Playwright is not available on this host" }), + zipBin: path.join(ROOT, "no-such-zip"), + onLog: () => {}, + }); + const dir = reportExportDir(paths, "cited", REPORT); + assert.deepEqual(readdirSync(dir).sort(), ["export.json", "report.html", "report.md"]); + assert.deepEqual(r.exported[0].manifest.notes, ["report.pdf skipped: Playwright is not available on this host"]); + assert.equal(r.problems.length, 1); + assert.equal(r.problems[0].format, "zip"); + assert.match(r.problems[0].message, /needs the `zip` program/); + // Only what was asked for, and a report that is not published is refused. + await exportSiteReports({ siteId: "cited", paths, formats: ["md"], onLog: () => {} }); + assert.deepEqual(readdirSync(dir).sort(), ["export.json", "report.md"]); + await assert.rejects(exportSiteReports({ siteId: "cited", paths, reportId: "nope", onLog: () => {} }), /does not publish a report "nope"/); + // The CLI: 2 for what does not exist, 1 for a problem. + const { main: exportMain } = await import("../bin/reports-export"); + const quiet = { log: () => {}, error: () => {} }; + assert.equal(await exportMain({ siteId: "no-such-site" }, quiet), 2); + assert.equal(await exportMain({ siteId: "cited", reportId: "nope" }, quiet), 2); + assert.equal(await exportMain({ siteId: "cited", formats: ["zip"], zipBin: path.join(ROOT, "no-such-zip") }, quiet), 1); + assert.equal(await exportMain({ siteId: "cited", formats: ["html", "md"] }, quiet), 0); +}); diff --git a/common/publish/reportExports.ts b/common/publish/reportExports.ts @@ -26,8 +26,9 @@ // (the clips play offline, the stills and screenshots // as files), report.md, citations.json and .csv. Packed // by the system `zip`, deterministically (sorted names, -// fixed times and modes, no extra attributes); a host -// without `zip` fails the format, naming it +// fixed times and modes, no extra attributes — the same +// report checked at the same `now` packs to the same +// bytes); a host without `zip` fails the format, naming it // export.json the manifest: each file's size and sha256, the sha256 // of the report.json it was made from, the footer, notes //