// Integration: the reports stage of compose, through the REAL site compose // (bin/compose-site.ts main) over a temp corpus. // // One video channel (a record whose cues are fresh enough to read from its // VTT, and one whose `en` track parses to no cues so `en-orig` is read), one // Bluesky channel with a posts archive, a fact-check citing all five kinds // (and defining one citation it never cites), stills, a saved source copy, and // a prepared media manifest with fake clips and a capture — the cache // `archilyzer reports prepare` would have written. Three sites over it: a // CITED one, a FULL one with the same report, and a full one with none. // // Run with: node_modules/.bin/tsx --test publish/composeReports.test.ts import { after, test } from "node:test"; import assert from "node:assert/strict"; import { existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; const ROOT = mkdtempSync(path.join(tmpdir(), "compose-reports-")); const PINNED: Record = { TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), SITES_DIR: path.join(ROOT, "transcripts", "sites"), SETTINGS_FILE: path.join(ROOT, "settings.json"), EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"), EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"), CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"), SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"), CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"), ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"), }; Object.assign(process.env, PINNED); delete process.env.REPORTS_ALLOW_MISSING_MEDIA; after(() => rmSync(ROOT, { recursive: true, force: true })); const { getPaths } = await import("../lib/paths"); const { buildIndex } = await import("../controller/buildIndex"); const { writePosts } = await import("../lib/posts-server"); const { main: composeSite } = await import("../bin/compose-site"); const { ComposeReportsError, CITATIONS_CSV_COLUMNS } = await import("./composeReports"); const { REPORT_MEDIA_FORMAT, REPORT_MEDIA_VERSION, reportMediaDir, reportMediaIndexFile } = await import("./reportMedia"); const { QUOTE_CHECK_METHOD } = await import("../lib/citations/verify"); const { parseCitationSet } = await import("../lib/citations/validate"); const { CONTRACT } = await import("../lib/archive/contract"); const { citedBuildProblem, builtBundleProblem, reportHistoryProblem } = await import("../lib/builtExport"); const paths = getPaths(); const VIDEOS = "demo-channel"; const SOCIAL = "demo-social"; const REPORT = "demo-report"; const NOW_FLOOR = new Date().toISOString(); // What a report's notes.json says: if it shows up in anything built, a note leaked. const NOTES_SENTINEL = "operator-note-never-published-7f3a"; const writeJson = (file: string, value: unknown) => { mkdirSync(path.dirname(file), { recursive: true }); writeFileSync(file, JSON.stringify(value, null, 2)); }; const writeText = (file: string, text: string) => { mkdirSync(path.dirname(file), { recursive: true }); writeFileSync(file, text); }; const readJson = >(file: string): T => JSON.parse(readFileSync(file, "utf8")) as T; const pub = (...p: string[]) => path.join(paths.exportPublicDir, ...p); // Every file under a directory, relative, sorted. function filesUnder(dir: string): string[] { const out: string[] = []; const walk = (d: string, rel: string) => { for (const e of readdirSync(d, { withFileTypes: true })) { const r = rel ? `${rel}/${e.name}` : e.name; if (e.isDirectory()) walk(path.join(d, e.name), r); else out.push(r); } }; if (existsSync(dir)) walk(dir, ""); return out.sort(); } // A YouTube-shaped VTT: parseVtt keeps only lines carrying inline timing. const ts = (s: number) => new Date(s * 1000).toISOString().slice(11, 23); const vtt = (cues: [number, number, string][], tagged = true) => "WEBVTT\nKind: captions\nLanguage: en\n\n" + cues .map(([a, b, text]) => `${ts(a)} --> ${ts(b)} align:start position:0%\n${text}${tagged ? `<${ts(a)}>` : ""}\n`) .join("\n"); const CLIP = "0".repeat(31) + "1"; const AUDIO_CLIP = "0".repeat(31) + "2"; const UNCITED_CLIP = "0".repeat(31) + "3"; function report(over: { c01Quote?: string } = {}) { return { format: "archilyzer-report", version: 1, id: REPORT, kind: "factcheck", title: "Checking a demo article", subtitle: "Four claims, one stream", published: "2026-10-01", subject: { source: "s0" }, sources: { s0: { kind: "article", title: "A demo article", url: "https://example.org/article", saved: "sources/s0/page.html", }, }, citations: { c01: { kind: "video", channel: VIDEOS, id: "abc123", start: 10, end: 20, pad: { before: 2, after: 3 }, quote: over.c01Quote ?? "The bridge opened in the spring, I was there for it.", // Hand-typed: compose overwrites it. verification: { quoteScore: 1, quoteCheckedAt: "2020-01-01T00:00:00Z", voiceChecked: true, method: "by hand" }, }, c02: { kind: "audio", channel: VIDEOS, id: "def456", start: 5, end: 9, quote: "words only the original track has" }, p01: { kind: "post", channel: SOCIAL, id: "3kabc", quote: "Posted to settle it", date: "2025-03-14" }, a01: { kind: "source", source: "s0", quote: "He opened the bridge himself.", image: "stills/a01.png" }, w01: { kind: "page", url: "https://example.org/page", quote: "a page says so", verification: { quoteScore: 0.5, quoteCheckedAt: "2020-01-01T00:00:00Z" }, }, u01: { kind: "video", channel: VIDEOS, id: "abc123", start: 40, end: 45, quote: "never cited" }, }, sections: [ { id: "bridge", title: "The bridge", claims: [ { id: "claim-1", text: "He opened the bridge himself.", verdict: "CONTRADICTED", sourceQuote: { citation: "a01" }, findings: "He says [it opened without him](cite:c01), and [posted so](cite:p01).", citations: ["c01", "c02", "w01"], }, ], }, ], }; } function seedSite(siteId: string, extra: Record = {}, withReport = true) { writeJson(path.join(paths.sitesDir, siteId, "site.json"), { siteId, siteTitle: `Site ${siteId}`, siteDescription: "fixture", headerTitle: siteId, homeTagline: "", socialLinks: [], groups: [{ id: "default", name: "All channels", selectedByDefault: true }], defaultGroupId: "default", channels: [VIDEOS, SOCIAL].map((slug) => ({ slug, groupId: "default" })), siteUrl: `https://${siteId}.example.test`, archives: false, ...extra, }); if (!withReport) return; const dir = path.join(paths.sitesDir, siteId, "reports", REPORT); writeJson(path.join(dir, "report.json"), report()); writeText(path.join(dir, "stills", "a01.png"), "png-bytes"); writeText(path.join(dir, "sources", "s0", "page.html"), "

saved copy, never published

"); // umtool's operator notes live beside report.json and are never published. writeJson(path.join(dir, "notes.json"), { format: "umtool-notes", version: 1, notes: [{ text: NOTES_SENTINEL }] }); seedMedia(siteId); } // What `archilyzer reports prepare` leaves: the manifest, the clips with // their sidecars, the cited capture. function seedMedia(siteId: string, drop: string[] = []) { const dir = reportMediaDir(paths, siteId); const clip = (hash: string, kind: "video" | "audio", span: { from: number; to: number }) => { const file = `${hash}${kind === "video" ? ".mp4" : ".m4a"}`; writeText(path.join(dir, file), `${kind}-bytes-${hash}`); const media = { kind, file, bytes: 20, sha256: "f".repeat(64), width: null, height: null, durationSec: span.to - span.from }; writeJson(path.join(dir, `${hash}.json`), { ...media, profile: "evidence-v1", span, source: { kind: "corpus-window", name: "x" } }); return media; }; writeText(path.join(dir, "posts", SOCIAL, "3kabc", "shot.png"), "shot"); writeText(path.join(dir, "posts", SOCIAL, "3kabc", "photo.jpg"), "photo"); const moments: Record = { [`${VIDEOS}/abc123/10.00-20.00`]: clip(CLIP, "video", { from: 8, to: 23 }), [`${VIDEOS}/def456/5.00-9.00`]: clip(AUDIO_CLIP, "audio", { from: 5, to: 9 }), [`${VIDEOS}/abc123/40.00-45.00`]: clip(UNCITED_CLIP, "video", { from: 40, to: 45 }), [`${SOCIAL}/3kabc`]: { kind: "post", file: `posts/${SOCIAL}/3kabc/shot.png`, bytes: 4, sha256: "e".repeat(64), width: null, height: null, durationSec: null, media: [{ file: `posts/${SOCIAL}/3kabc/photo.jpg`, bytes: 5, sha256: "d".repeat(64) }], }, }; for (const k of drop) delete moments[k]; writeJson(reportMediaIndexFile(paths, siteId), { format: REPORT_MEDIA_FORMAT, version: REPORT_MEDIA_VERSION, siteId, preparedAt: "2026-10-05T00:00:00.000Z", moments, problems: [], }); } async function seedCorpus() { writeJson(paths.settingsFile, {}); writeJson(path.join(paths.channelsDir, VIDEOS, "config.json"), { handling: "youtube", name: "Demo Channel", url: "https://www.youtube.com/@demo/videos", }); const meta = (id: string) => ({ id, title: `Demo stream ${id}`, channel: VIDEOS, upload_date: "20260110", duration: 600, webpage_url: `https://www.youtube.com/watch?v=${id}`, extractor_key: "Youtube", }); const abc = path.join(paths.channelsDir, VIDEOS, "data", "abc123"); writeJson(path.join(abc, "metadata.info.json"), meta("abc123")); writeText( path.join(abc, "transcript.en.vtt"), vtt([ [0, 6, "Okay so somebody asked about the bridge."], [6, 10, "Let me be clear about this one."], [10, 15, "The bridge opened in the spring,"], [15, 20, "I was there for it."], [20, 30, "Anyway, back to the mail."], [30, 40, "Next letter."], [40, 45, "never cited"], [60, 70, "Far outside any window."], ]), ); // The served `en` track is a rewrite of what was said; the original track // has the words. The quote must be checked (and shown) against the original. const def = path.join(paths.channelsDir, VIDEOS, "data", "def456"); writeJson(path.join(def, "metadata.info.json"), meta("def456")); writeText(path.join(def, "transcript.en.vtt"), vtt([[5, 9, "Some words a rewrite changed."]])); writeText(path.join(def, "transcript.en-orig.vtt"), vtt([[5, 9, "Words only the original track has."]])); writeJson(path.join(paths.channelsDir, SOCIAL, "config.json"), { handling: "youtube", name: "Demo Social", url: "https://bsky.app/profile/demo.example", sourceKind: "social", platform: "bluesky", socialHandle: "demo.example", }); await writePosts(path.join(paths.channelsDir, SOCIAL), [ { id: "3kabc", slug: `${SOCIAL}/3kabc`, channelSlug: SOCIAL, author: "demo.example", authorName: "Demo", createdAt: "2025-03-14T12:00:00.000Z", uploadDate: "20250314", text: "Posted to settle it. The bridge was open before I got there.", url: "https://bsky.app/profile/demo.example/post/3kabc", platform: "bluesky", isReply: false, isRepost: false, links: [], }, ]); seedSite("cited", { search: false, reports: [REPORT] }); seedSite("full", { reports: [REPORT] }); seedSite("plain", {}, false); await buildIndex({ paths, onLog: () => {} }); } async function compose(siteId: string, opts: { allowMissingMedia?: boolean } = {}) { const log = console.log; console.log = () => {}; try { await composeSite({ siteId, paths, ...opts }); } finally { console.log = log; } } await seedCorpus(); test("a cited site: exactly its reports, moments, cited media and contract — the corpus pruned", async () => { // A full compose leaves the corpus in public/, as the shared public dir // would hold it after another site's build; and some stray files besides. await compose("plain"); assert.ok(existsSync(pub("transcripts", VIDEOS))); for (const f of ["sw.js", "tags.json", "duplicates.json", "hub-sites.json", "chart-templates.json"]) writeText(pub(f), "stale"); writeText(pub("archives", "manifest.json"), "{}"); writeText(pub("digests", VIDEOS, "manifest.json"), "{}"); await compose("cited"); const abc = `${VIDEOS}/abc123/10.00-20.00`; const def = `${VIDEOS}/def456/5.00-9.00`; assert.deepEqual(filesUnder(paths.exportPublicDir), [ "_headers", "corpus.json", "llms.txt", `m/${VIDEOS}/abc123/10.00-20.00/moment.json`, `m/${VIDEOS}/def456/5.00-9.00/moment.json`, `m/${SOCIAL}/3kabc/moment.json`, "m/index.json", `media/clips/${abc}.mp4`, `media/clips/${def}.m4a`, `media/posts/${SOCIAL}/3kabc/photo.jpg`, `media/posts/${SOCIAL}/3kabc/shot.png`, `reports/${REPORT}/citations.csv`, `reports/${REPORT}/citations.json`, `reports/${REPORT}/page.json`, `reports/${REPORT}/stills/a01.png`, "reports/index.json", "robots.txt", "site.json", "sitemap.xml", ]); assert.equal(readFileSync(pub("media", "clips", `${abc}.mp4`), "utf8"), `video-bytes-${CLIP}`); assert.equal(readFileSync(pub("media", "clips", `${def}.m4a`), "utf8"), `audio-bytes-${AUDIO_CLIP}`); // The uncited citation's clip, prepared, is not published; nor its page. assert.ok(!JSON.stringify(readJson(pub("m", "index.json"))).includes("40.00")); }); test("a cited site's contract: site.json without channels, corpus.json scope cited at spec 5, the cited llms.txt and sitemap", () => { const site = readJson<{ siteId: string; channels: unknown[]; pwa: boolean }>(pub("site.json")); assert.equal(site.siteId, "cited"); assert.deepEqual(site.channels, []); assert.equal(site.pwa, false); const corpus = readJson<{ spec: number; site: { id: string; scope?: string; audience?: string }; channels: unknown[]; totals: { channels: number; videos: number }; reports?: { index: string; count: number }; }>(pub("corpus.json")); assert.equal(corpus.spec, CONTRACT.corpusSpec); assert.equal(corpus.spec, 5); assert.equal(corpus.site.id, "cited"); assert.equal(corpus.site.scope, "cited"); assert.equal(corpus.site.audience, "public"); assert.deepEqual(corpus.channels, []); assert.deepEqual(corpus.totals, { channels: 0, videos: 0 }); assert.equal(corpus.reports?.index, "https://cited.example.test/reports/index.json"); assert.equal(corpus.reports?.count, 1); const llms = readFileSync(pub("llms.txt"), "utf8"); assert.match(llms, /^# Site cited/); assert.match(llms, /\[Checking a demo article\]\(https:\/\/cited\.example\.test\/reports\/demo-report\/\): Four claims, one stream/); assert.match(llms, /there is no searchable corpus here/); assert.doesNotMatch(llms, /## Channels|shardScheme|page-|transcripts\//); const sitemap = readFileSync(pub("sitemap.xml"), "utf8"); const locs = [...sitemap.matchAll(/([^<]+)<\/loc>/g)].map((m) => m[1].replace("https://cited.example.test", "")); assert.deepEqual(locs, [ "/", "/reports/", "/reports/demo-report/", `/m/${VIDEOS}/abc123/10.00-20.00/`, `/m/${VIDEOS}/def456/5.00-9.00/`, `/m/${SOCIAL}/3kabc/`, ]); // The bundle names itself and its scope: the deploy guards take it. assert.equal(builtBundleProblem(paths.exportPublicDir, "cited"), null); assert.equal(citedBuildProblem(paths.exportPublicDir), null); }); type View = { citations: Record; record?: Record; shot?: string; text?: string; author?: string }> }; test("verification is computed at compose: a hand-typed block is overwritten, a page's dropped, en-orig read when en has no cues", () => { const view = readJson(pub("reports", REPORT, "page.json")); const v = view.citations.c01.verification!; assert.equal(v.quoteScore, 1); assert.equal(v.method, `${QUOTE_CHECK_METHOD}; text: transcript.en.vtt`, "the method names the transcript that matched"); assert.ok((v.quoteCheckedAt as string) >= NOW_FLOOR, "checked now, not when the document says"); assert.ok(!("voiceChecked" in v), "a voice is never vouched for by compose"); assert.equal(view.citations.c02.verification?.quoteScore, 1, "checked against the en-orig track"); assert.match(String(view.citations.c02.verification?.method), /text: transcript\.en-orig\.vtt$/, "a rewritten `en` track loses to the original"); assert.equal(view.citations.p01.verification?.quoteScore, 1, "a post's quote against its text"); assert.equal(view.citations.w01.verification, undefined); assert.equal(view.citations.a01.verification, undefined); assert.equal(view.citations.u01, undefined, "a citation never cited is not in the view"); // The record, resolved from the corpus; a cited site links no corpus. assert.deepEqual(view.citations.c01.record, { channel: VIDEOS, channelTitle: "Demo Channel", id: "abc123", title: "Demo stream abc123", date: "2026-01-10", platform: "youtube", originalUrl: "https://www.youtube.com/watch?v=abc123&t=10s", }); assert.equal(view.citations.p01.shot, `/media/posts/${SOCIAL}/3kabc/shot.png`); assert.equal(view.citations.p01.author, "Demo (@demo.example)"); }); test("moment pages: the clip with its pad, the bounded cue context, the post's capture, cited in", () => { const span = readJson & { cues: { start: number; text: string; inSpan: boolean }[]; citedIn: { href: string }[] }>( pub("m", VIDEOS, "abc123", "10.00-20.00", "moment.json"), ); assert.equal(span.kind, "video"); assert.deepEqual(span.clip, { src: `/media/clips/${VIDEOS}/abc123/10.00-20.00.mp4`, start: 8, end: 23 }); assert.equal(span.start, 10); assert.equal(span.end, 20); // ±15 s of context, never the whole record. assert.deepEqual( span.cues.map((q) => [q.start, q.inSpan]), [ [0, false], [6, false], [10, true], [15, true], [20, false], [30, false], ], ); assert.deepEqual( span.citedIn.map((e) => e.href), [`/reports/${REPORT}/#claim-1`], ); const audio = readJson>(pub("m", VIDEOS, "def456", "5.00-9.00", "moment.json")); assert.equal(audio.kind, "audio"); assert.deepEqual(audio.clip, { src: `/media/clips/${VIDEOS}/def456/5.00-9.00.m4a`, start: 5, end: 9 }); const post = readJson & { post: unknown; record: Record }>( pub("m", SOCIAL, "3kabc", "moment.json"), ); assert.equal(post.kind, "post"); assert.equal(post.date, "2025-03-14"); assert.deepEqual(post.post, { author: "Demo (@demo.example)", text: "Posted to settle it. The bridge was open before I got there.", shot: `/media/posts/${SOCIAL}/3kabc/shot.png`, media: [{ src: `/media/posts/${SOCIAL}/3kabc/photo.jpg`, kind: "image" }], }); assert.equal(post.record.originalUrl, "https://bsky.app/profile/demo.example/post/3kabc"); }); test("the citations as files: a valid citation set without the saved copy, and one CSV row per citation", () => { const set = readJson(pub("reports", REPORT, "citations.json")); const parsed = parseCitationSet(set); assert.ok(parsed.ok, JSON.stringify(parsed.problems)); assert.deepEqual(parsed.problems, []); assert.deepEqual(Object.keys((set as { citations: object }).citations), ["a01", "c01", "p01", "c02", "w01"]); assert.ok(!JSON.stringify(set).includes("saved"), "a source's saved copy is never published"); const csv = readFileSync(pub("reports", REPORT, "citations.csv"), "utf8").trimEnd().split("\r\n"); assert.equal(csv[0], CITATIONS_CSV_COLUMNS.join(",")); assert.equal(csv.length, 6); assert.equal( csv[2], `c01,2,video,${VIDEOS},abc123,10,20,"The bridge opened in the spring, I was there for it.",,2026-01-10,https://www.youtube.com/watch?v=abc123&t=10s,/m/${VIDEOS}/abc123/10.00-20.00/,1`, ); assert.match(csv[1], /^a01,1,source,,,,,He opened the bridge himself\.,,,https:\/\/example\.org\/article,,$/); }); test("a quote that drifted from its cues fails compose, before anything is written", async () => { const file = path.join(paths.sitesDir, "cited", "reports", REPORT, "report.json"); writeJson(file, report({ c01Quote: "He said he cut the ribbon himself that morning." })); try { await assert.rejects(compose("cited"), (e: unknown) => { assert.ok(e instanceof ComposeReportsError); assert.deepEqual( e.problems.map((p) => [p.kind, p.citation]), [["quote-drift", `${REPORT}#c01`]], ); return true; }); assert.ok(!existsSync(pub("reports"))); assert.ok(!existsSync(pub("site.json")), "a failed compose leaves public/ naming no site"); } finally { writeJson(file, report()); } }); test("a citation without prepared media fails compose with the list, unless --allow-missing-media", async () => { seedMedia("cited", [`${VIDEOS}/abc123/10.00-20.00`]); try { await assert.rejects(compose("cited"), (e: unknown) => { assert.ok(e instanceof ComposeReportsError); assert.deepEqual( e.problems.map((p) => [p.kind, p.moment]), [["missing-media", `${VIDEOS}/abc123/10.00-20.00`]], ); assert.match(e.message, /cited by demo-report#c01/); return true; }); await compose("cited", { allowMissingMedia: true }); const m = readJson>(pub("m", VIDEOS, "abc123", "10.00-20.00", "moment.json")); assert.equal(m.clip, undefined); assert.ok(!existsSync(pub("media", "clips", VIDEOS, "abc123"))); } finally { seedMedia("cited"); } }); test("a report's video: copied beside its page and named in the view; a missing or oversize one fails compose", async () => { const dir = path.join(paths.sitesDir, "cited", "reports", REPORT); const file = path.join(dir, "report.json"); writeJson(file, { ...report(), video: { src: "video.mp4", poster: "poster.jpg", caption: "The cut" } }); try { await assert.rejects(compose("cited"), (e: unknown) => { assert.ok(e instanceof ComposeReportsError); assert.deepEqual(e.problems.map((p) => [p.kind, p.report]), [["report-video", REPORT], ["report-video", REPORT]]); assert.match(e.message, /video\.mp4 does not exist/); return true; }); writeText(path.join(dir, "video.mp4"), "mp4-bytes"); writeText(path.join(dir, "poster.jpg"), "jpg-bytes"); await compose("cited"); assert.equal(readFileSync(pub("reports", REPORT, "video.mp4"), "utf8"), "mp4-bytes"); assert.equal(readFileSync(pub("reports", REPORT, "poster.jpg"), "utf8"), "jpg-bytes"); const view = readJson<{ video?: unknown }>(pub("reports", REPORT, "page.json")); assert.deepEqual(view.video, { src: `/reports/${REPORT}/video.mp4`, poster: `/reports/${REPORT}/poster.jpg`, caption: "The cut" }); writeFileSync(path.join(dir, "video.mp4"), Buffer.alloc(25 * 1024 * 1024)); await assert.rejects(compose("cited"), (e: unknown) => { assert.ok(e instanceof ComposeReportsError); assert.deepEqual(e.problems.map((p) => p.kind), ["report-video"]); assert.match(e.message, /over the publish limit/); return true; }); } finally { writeJson(file, report()); rmSync(path.join(dir, "video.mp4"), { force: true }); rmSync(path.join(dir, "poster.jpg"), { force: true }); await compose("cited"); } }); test("a clip prepared for another span is stale: the reports changed since prepare", async () => { const sidecar = path.join(reportMediaDir(paths, "cited"), `${CLIP}.json`); const saved = readFileSync(sidecar, "utf8"); writeJson(sidecar, { ...JSON.parse(saved), span: { from: 10, to: 20 } }); try { await assert.rejects(compose("cited"), (e: unknown) => { assert.ok(e instanceof ComposeReportsError); assert.deepEqual(e.problems.map((p) => p.kind), ["stale-media"]); return true; }); } finally { writeFileSync(sidecar, saved); } }); test("a full site with reports keeps its corpus and links each moment into it", async () => { await compose("full"); assert.ok(existsSync(pub("transcripts", VIDEOS)), "the corpus is composed as ever"); assert.ok(existsSync(pub("summaries", "manifest.json"))); const corpus = readJson<{ spec: number; site: { scope?: string; audience?: string }; channels: unknown[]; reports?: { count: number } }>( pub("corpus.json"), ); assert.equal(corpus.spec, 5); assert.equal(corpus.site.scope, undefined); assert.equal(corpus.site.audience, undefined); assert.equal(corpus.channels.length, 2); assert.equal(corpus.reports?.count, 1); const m = readJson<{ record: { corpusUrl?: string } }>(pub("m", VIDEOS, "abc123", "10.00-20.00", "moment.json")); assert.equal(m.record.corpusUrl, `/?v=${VIDEOS}%2Fabc123&t=10`); const post = readJson<{ record: { corpusUrl?: string } }>(pub("m", SOCIAL, "3kabc", "moment.json")); assert.equal(post.record.corpusUrl, `/?v=${SOCIAL}%2F3kabc&vm=post`); const llms = readFileSync(pub("llms.txt"), "utf8"); assert.match(llms, /## Reports[^]*Checking a demo article[^]*## Channels/); assert.match(readFileSync(pub("sitemap.xml"), "utf8"), /\/reports\/demo-report\//); assert.equal(citedBuildProblem(paths.exportPublicDir), null, "the audit leaves a full build alone"); }); test("a site with no reports ships none of the last site's", async () => { await compose("cited"); await compose("plain"); for (const entry of ["reports", "m", "media"]) assert.ok(!existsSync(pub(entry)), entry); const corpus = readJson<{ reports?: unknown }>(pub("corpus.json")); assert.equal(corpus.reports, undefined); }); test("a report's timeline: its feeds beside the page on a site with a public URL, none without", async () => { const withTimeline = { ...report(), entries: [ { id: "e-old", date: "2026-10-02", title: "First & ", body: "It [opened](cite:c01) after all." }, { id: "e-new", date: "2026-10-05T08:30:00Z", title: "Second update", body: "Nothing new." }, ], }; for (const [siteId, extra] of [ ["timeline", {}], ["timeline-private", { siteUrl: undefined, audience: "private" }], ] as const) { seedSite(siteId, { search: false, reports: [REPORT], ...extra }); writeJson(path.join(paths.sitesDir, siteId, "reports", REPORT, "report.json"), withTimeline); } await compose("timeline"); const page = readJson<{ entries: { id: string }[]; updated?: string; feeds?: Record }>(pub("reports", REPORT, "page.json")); assert.deepEqual(page.entries.map((e) => e.id), ["e-new", "e-old"]); assert.equal(page.updated, "2026-10-05T08:30:00Z"); assert.deepEqual(page.feeds, { rss: `https://timeline.example.test/reports/${REPORT}/feed.xml`, json: `https://timeline.example.test/reports/${REPORT}/feed.json`, }); const index = readJson<{ reports: { updated?: string }[] }>(pub("reports", "index.json")); assert.equal(index.reports[0].updated, "2026-10-05T08:30:00Z"); const rss = readFileSync(pub("reports", REPORT, "feed.xml"), "utf8"); const items = [...rss.matchAll(/([^<]+)<\/link>/g)].map((m) => m[1]); assert.deepEqual(items, [ `https://timeline.example.test/reports/${REPORT}/`, `https://timeline.example.test/reports/${REPORT}/#e-new`, `https://timeline.example.test/reports/${REPORT}/#e-old`, ]); assert.match(rss, /First & <update><\/title>/); const feed = readJson<{ version: string; items: { id: string }[] }>(pub("reports", REPORT, "feed.json")); assert.equal(feed.version, "https://jsonfeed.org/version/1.1"); assert.deepEqual(feed.items.map((i) => i.id.split("#")[1]), ["e-new", "e-old"]); // Served as what they are, and readable cross-origin. assert.match(readFileSync(pub("_headers"), "utf8"), /\/reports\/:report\/feed\.xml\n {2}Content-Type: application\/rss\+xml; charset=utf-8\n/); // The cited audit allows them (reports/ is the stage's own). assert.equal(citedBuildProblem(paths.exportPublicDir), null); // No public URL: the page and its timeline, no feed — absolute links have nowhere to point. await compose("timeline-private"); const priv = readJson<{ entries: unknown[]; feeds?: unknown }>(pub("reports", REPORT, "page.json")); assert.equal(priv.entries.length, 2); assert.equal(priv.feeds, undefined); assert.ok(!existsSync(pub("reports", REPORT, "feed.xml"))); assert.ok(!existsSync(pub("reports", REPORT, "feed.json"))); }); // ─── Exports (publish/reportExports.ts) ─── const { exportSiteReports } = await import("./reportExports"); const { reportExportDir } = await import("./reportExportFiles"); const { execFileSync } = await import("node:child_process"); const { createHash } = await import("node:crypto"); const { truncateSync } = await import("node:fs"); const sha = (file: string) => createHash("sha256").update(readFileSync(file)).digest("hex"); const fakePrinter = (printed: string[]) => async () => ({ print: async (html: string) => { printed.push(html); return Buffer.from("%PDF-1.4 fake"); }, close: async () => {}, }); test("reports export: every format from the resolved view, a self-contained HTML, a deterministic pack", async () => { const printed: string[] = []; // `now` is the verification's time, which citations.json carries: the // same report, checked at the same time, packs to the same bytes. const now = () => new Date("2026-10-05T12:00:00Z"); const r = await exportSiteReports({ siteId: "cited", paths, now, openPdfPrinter: fakePrinter(printed), onLog: () => {} }); assert.deepEqual(r.problems, []); const dir = reportExportDir(paths, "cited", REPORT); assert.deepEqual(readdirSync(dir).sort(), [ "evidence-pack.zip", "export.json", "report.html", "report.md", "report.pdf", "slides.html", "slides.pdf", ]); const html = readFileSync(path.join(dir, "report.html"), "utf8"); const slides = readFileSync(path.join(dir, "slides.html"), "utf8"); assert.deepEqual(printed, [html, slides], "each PDF is its HTML, printed"); // The slides: one file, the site's stylesheet and its few lines of script // inline, every picture inlined, one page per slide. assert.match(slides, /\.rs-slide\s*\{/); assert.equal([...slides.matchAll(/<script>/g)].length, 1); for (const m of slides.matchAll(/<img [^>]*src="([^"]+)"/g)) assert.match(m[1], /^data:image\//); assert.ok([...slides.matchAll(/class="rs-page" id="s-\d+"/g)].length >= 4); assert.doesNotMatch(html, /<script/i); const imgs = [...html.matchAll(/<img [^>]*src="([^"]+)"/g)].map((m) => m[1]); assert.equal(imgs.length, 2, "the claim's still and the post's screenshot"); for (const src of imgs) assert.match(src, /^data:image\//); // Verified as compose verifies, the clip linked on the site, never inlined. assert.match(html, /quote match 100%/); assert.match(html, /https:\/\/cited\.example\.test\/media\/clips\/demo-channel\/abc123\/10\.00-20\.00\.mp4/); assert.doesNotMatch(html, /<video/); const md = readFileSync(path.join(dir, "report.md"), "utf8"); assert.match(md, /^1\. “He opened the bridge himself\.”/m); const manifest = readJson<{ reportSha256: string; files: Record<string, { bytes: number; sha256: string }>; notes: string[] }>( path.join(dir, "export.json"), ); assert.equal(manifest.reportSha256, sha(path.join(paths.sitesDir, "cited", "reports", REPORT, "report.json"))); assert.deepEqual(Object.keys(manifest.files), ["html", "pdf", "md", "slides", "slides-pdf", "zip"]); assert.equal(manifest.files.html.sha256, sha(path.join(dir, "report.html"))); assert.deepEqual(manifest.notes, []); assert.match(html, new RegExp(`report sha256 ${manifest.reportSha256.slice(0, 12)}`)); // The pack: the HTML playing its own media, the Markdown, the citations. const zip = path.join(dir, "evidence-pack.zip"); const names = execFileSync("unzip", ["-Z1", zip], { encoding: "utf8" }).trim().split("\n"); assert.deepEqual(names, [ `${REPORT}/citations.csv`, `${REPORT}/citations.json`, `${REPORT}/media/clips/${VIDEOS}/abc123/10.00-20.00.mp4`, `${REPORT}/media/clips/${VIDEOS}/def456/5.00-9.00.m4a`, `${REPORT}/media/posts/${SOCIAL}/3kabc/shot.png`, `${REPORT}/media/report/stills/a01.png`, `${REPORT}/report.html`, `${REPORT}/report.md`, `${REPORT}/slides.html`, ]); const packed = execFileSync("unzip", ["-p", zip, `${REPORT}/report.html`], { encoding: "utf8" }); assert.match(packed, /<video controls preload="none" src="media\/clips\/demo-channel\/abc123\/10\.00-20\.00\.mp4">/); assert.match(packed, /<img src="media\/report\/stills\/a01\.png"/); // The same report exports to the same bytes. const before = { html: sha(path.join(dir, "report.html")), slides: sha(path.join(dir, "slides.html")), zip: sha(zip) }; await exportSiteReports({ siteId: "cited", paths, now, openPdfPrinter: fakePrinter([]), onLog: () => {} }); assert.deepEqual({ html: sha(path.join(dir, "report.html")), slides: sha(path.join(dir, "slides.html")), zip: sha(zip) }, before); }); test("compose publishes the exports beside the report and lists them as downloads", async () => { await compose("cited"); for (const f of ["report.html", "report.pdf", "report.md", "slides.html", "slides.pdf", "evidence-pack.zip"]) { assert.equal(sha(pub("reports", REPORT, f)), sha(path.join(reportExportDir(paths, "cited", REPORT), f)), f); } const view = readJson<{ downloads: Record<string, string> }>(pub("reports", REPORT, "page.json")); assert.deepEqual(view.downloads, { html: `/reports/${REPORT}/report.html`, pdf: `/reports/${REPORT}/report.pdf`, md: `/reports/${REPORT}/report.md`, slides: `/reports/${REPORT}/slides.html`, "slides-pdf": `/reports/${REPORT}/slides.pdf`, zip: `/reports/${REPORT}/evidence-pack.zip`, json: `/reports/${REPORT}/citations.json`, csv: `/reports/${REPORT}/citations.csv`, }); assert.equal(citedBuildProblem(paths.exportPublicDir), null, "the exports are within reports/, which the audit allows"); }); test("reports export commits a revision when report.json changed; compose publishes the history, the page names it", async () => { const dir = reportExportDir(paths, "cited", REPORT); const file = path.join(paths.sitesDir, "cited", "reports", REPORT, "report.json"); const saved = readFileSync(file, "utf8"); // The exports above made revision 1; exporting the same report.json again made none. const first = readJson<{ revision: { revision: number; committed: boolean; commit: string }; footer: { revision?: number } }>( path.join(dir, "export.json"), ); assert.equal(first.revision.revision, 1); assert.equal(first.revision.committed, false); assert.equal(first.footer.revision, 1); assert.match(readFileSync(path.join(dir, "report.md"), "utf8"), /Revision 1 · 2026-10-01 · report sha256 [0-9a-f]{12}/); const edited = report(); edited.sections[0].claims[0] = { ...edited.sections[0].claims[0], text: "He opened the old bridge himself.", verdict: "PARTLY" }; writeJson(file, edited); try { const r = await exportSiteReports({ siteId: "cited", paths, now: () => new Date("2026-10-06T08:00:00Z"), openPdfPrinter: fakePrinter([]), onLog: () => {}, }); assert.deepEqual(r.problems, []); const m = r.exported[0].manifest; assert.equal(m.revision?.revision, 2); assert.equal(m.revision?.committed, true); assert.deepEqual(m.revision?.summary, ["Verdict changed: claim-1 CONTRADICTED → PARTLY", "Claim edited: claim-1 (text)"]); assert.match(readFileSync(path.join(dir, "report.html"), "utf8"), /Revision 2 · 2026-10-01 · report sha256/); await compose("cited"); const history = readJson<{ clone: string; revisions: { revision: number; commit: string; reportSha256: string; date: string; summary: string[]; claims: { id: string }[] }[]; }>(pub("reports", REPORT, "history", "history.json")); assert.equal(history.clone, `https://cited.example.test/reports/${REPORT}/history/repo`); assert.deepEqual(history.revisions.map((x) => x.revision), [1, 2]); assert.equal(history.revisions[1].commit, m.revision?.commit); assert.equal(history.revisions[1].reportSha256, sha(file)); assert.equal(history.revisions[1].date, "2026-10-06T08:00:00Z"); assert.deepEqual(history.revisions[1].claims.map((c) => c.id), ["claim-1"]); const repo = filesUnder(pub("reports", REPORT, "history", "repo")); assert.deepEqual(repo.filter((f) => !f.startsWith("objects/pack/")), ["HEAD", "info/refs", "objects/info/packs", "packed-refs", "refs/heads/main"]); assert.ok(repo.some((f) => f.endsWith(".pack"))); const view = readJson<{ history: unknown }>(pub("reports", REPORT, "page.json")); assert.deepEqual(view.history, { revision: 2, date: "2026-10-06T08:00:00Z", href: `/reports/${REPORT}/history/`, current: true }); assert.match(readFileSync(pub("sitemap.xml"), "utf8"), new RegExp(`/reports/${REPORT}/history/`)); assert.equal(citedBuildProblem(paths.exportPublicDir), null); assert.equal(reportHistoryProblem(paths.exportPublicDir), null); } finally { writeFileSync(file, saved); } // Edited since the newest revision and not exported: the page says so. await compose("cited"); const view = readJson<{ history: { revision: number; current: boolean } }>(pub("reports", REPORT, "page.json")); assert.deepEqual([view.history.revision, view.history.current], [2, false]); // Exported again, the way back is revision 3. const back = await exportSiteReports({ siteId: "cited", paths, now: () => new Date("2026-10-06T09:00:00Z"), openPdfPrinter: fakePrinter([]), onLog: () => {} }); assert.equal(back.exported[0].manifest.revision?.revision, 3); assert.deepEqual(back.exported[0].manifest.revision?.summary, ["Verdict changed: claim-1 PARTLY → CONTRADICTED", "Claim edited: claim-1 (text)"]); // A site with no reports ships no history. await compose("plain"); assert.ok(!existsSync(pub("reports"))); }); test("an export of another version of the report is not published; a pack over the limit stays local", async () => { const dir = reportExportDir(paths, "cited", REPORT); const file = path.join(paths.sitesDir, "cited", "reports", REPORT, "report.json"); const saved = readFileSync(file, "utf8"); writeFileSync(file, `${saved}\n`); try { await compose("cited"); assert.ok(!existsSync(pub("reports", REPORT, "report.html"))); const view = readJson<{ downloads: Record<string, string> }>(pub("reports", REPORT, "page.json")); assert.deepEqual(Object.keys(view.downloads), ["json", "csv"]); } finally { writeFileSync(file, saved); } truncateSync(path.join(dir, "evidence-pack.zip"), 25 * 1024 * 1024); await compose("cited"); assert.ok(existsSync(pub("reports", REPORT, "report.html"))); assert.ok(!existsSync(pub("reports", REPORT, "evidence-pack.zip"))); const view = readJson<{ downloads: Record<string, string> }>(pub("reports", REPORT, "page.json")); assert.deepEqual(Object.keys(view.downloads), ["html", "pdf", "md", "slides", "slides-pdf", "json", "csv"]); }); test("reports export: no browser skips the PDF with a note; no zip fails the pack, naming it", async () => { const r = await exportSiteReports({ siteId: "cited", paths, openPdfPrinter: async () => ({ missing: "Playwright is not available on this host" }), zipBin: path.join(ROOT, "no-such-zip"), onLog: () => {}, }); const dir = reportExportDir(paths, "cited", REPORT); assert.deepEqual(readdirSync(dir).sort(), ["export.json", "report.html", "report.md", "slides.html"]); assert.deepEqual(r.exported[0].manifest.notes, [ "report.pdf skipped: Playwright is not available on this host", "slides.pdf skipped: Playwright is not available on this host", ]); assert.equal(r.problems.length, 1); assert.equal(r.problems[0].format, "zip"); assert.match(r.problems[0].message, /needs the `zip` program/); // Only what was asked for, and a report that is not published is refused. await exportSiteReports({ siteId: "cited", paths, formats: ["md"], onLog: () => {} }); assert.deepEqual(readdirSync(dir).sort(), ["export.json", "report.md"]); await assert.rejects(exportSiteReports({ siteId: "cited", paths, reportId: "nope", onLog: () => {} }), /does not publish a report "nope"/); // The CLI: 2 for what does not exist, 1 for a problem. const { main: exportMain } = await import("../bin/reports-export"); const quiet = { log: () => {}, error: () => {} }; assert.equal(await exportMain({ siteId: "no-such-site" }, quiet), 2); assert.equal(await exportMain({ siteId: "cited", reportId: "nope" }, quiet), 2); assert.equal(await exportMain({ siteId: "cited", formats: ["zip"], zipBin: path.join(ROOT, "no-such-zip") }, quiet), 1); assert.equal(await exportMain({ siteId: "cited", formats: ["html", "md"] }, quiet), 0); }); test("a report's notes.json (umtool's operator notes) is never published, exported or committed to its history", async () => { const { reportHistoryGitDir } = await import("./reportHistory"); const { execFileSync } = await import("node:child_process"); const notes = path.join(paths.sitesDir, "cited", "reports", REPORT, "notes.json"); assert.ok(existsSync(notes)); await exportSiteReports({ siteId: "cited", paths, now: () => new Date("2026-10-08T08:00:00Z"), openPdfPrinter: fakePrinter([]), onLog: () => {} }); await compose("cited"); const leaks = (root: string) => filesUnder(root).filter((f) => f.endsWith("notes.json") || readFileSync(path.join(root, f)).includes(NOTES_SENTINEL)); assert.deepEqual(leaks(paths.exportPublicDir), []); assert.deepEqual(leaks(reportExportDir(paths, "cited", REPORT)), []); const gitDir = reportHistoryGitDir(paths, "cited", REPORT); const revs = execFileSync("git", ["--git-dir", gitDir, "rev-list", "--all"], { encoding: "utf8" }).trim().split("\n").filter(Boolean); assert.ok(revs.length > 0); for (const rev of revs) { const names = execFileSync("git", ["--git-dir", gitDir, "ls-tree", "-r", "--name-only", rev], { encoding: "utf8" }).trim().split("\n"); assert.deepEqual(names.filter((n) => n.includes("notes")), [], rev); for (const n of names) { const blob = execFileSync("git", ["--git-dir", gitDir, "cat-file", "blob", `${rev}:${n}`]); assert.ok(!blob.includes(NOTES_SENTINEL), `${rev}:${n}`); } } });