commit d1cf5afe5eb7d362485444da55fc6d78a2e4145a
parent 23dd41318f263c99f8feb4e9f010996c5ec32e91
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 03:33:41 -0400
common: tests — compose over a temp corpus (exact files, prune, contract, verification, drift, missing and stale media, full-site colocation), the cited audit, the build env
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
3 files changed, 695 insertions(+), 0 deletions(-)
diff --git a/common/lib/builtExport.test.ts b/common/lib/builtExport.test.ts
@@ -8,8 +8,12 @@ import {
builtHomepageAt,
builtHomepageProblem,
builtHubProblem,
+ builtScopeProblem,
builtSiteIdIn,
builtSiteProblem,
+ citedBuildProblem,
+ deployAudienceProblem,
+ PAGES_MAX_FILES,
} from "./builtExport";
function tempOut(siteJson?: string): { dir: string; cleanup: () => void } {
@@ -256,3 +260,133 @@ test("builtSiteProblem refuses exactly what builtBundleProblem refuses", () => {
torn.cleanup();
}
});
+
+// ─── The cited out/ audit ───
+
+// A cited build as next build leaves it: compose's files, Next's shell, the
+// reports, moments and cited media.
+function citedOut(extra: string[] = []): { dir: string; cleanup: () => void } {
+ const t = bundle({
+ site: { siteId: "reports-site", channels: [] },
+ corpus: { spec: 5, kind: "site", site: { id: "reports-site", scope: "cited", audience: "public" } },
+ });
+ const files = [
+ "_next/static/chunks/app.js",
+ "_not-found/index.html",
+ "404/index.html",
+ "404.html",
+ "index.html",
+ "index.txt",
+ "__next._tree.txt",
+ "__next.!KHdvcmtzcGFjZSk.__PAGE__.txt",
+ "ask/index.html",
+ "changelog/index.html",
+ "downloads/index.html",
+ "duplicates/index.html",
+ "offline/index.html",
+ "reports/index.json",
+ "reports/index.html",
+ "reports/_none/index.html",
+ "reports/demo-report/index.html",
+ "reports/demo-report/page.json",
+ "reports/demo-report/stills/a01.png",
+ "m/index.json",
+ "m/_none/index.html",
+ "m/demo-channel/abc123/10.00-20.00/index.html",
+ "media/clips/demo-channel/abc123/10.00-20.00.mp4",
+ "media/posts/demo-social/3kabc/shot.png",
+ "icons/icon-192.png",
+ "favicon.ico",
+ "manifest.webmanifest",
+ "llms.txt",
+ "robots.txt",
+ "sitemap.xml",
+ "_headers",
+ "next.svg",
+ ...extra,
+ ];
+ for (const f of files) {
+ mkdirSync(path.dirname(path.join(t.dir, f)), { recursive: true });
+ writeFileSync(path.join(t.dir, f), "x");
+ }
+ return t;
+}
+
+test("a cited build holding only reports, moments, cited media and the shell passes, and deploys", () => {
+ const t = citedOut();
+ try {
+ assert.equal(citedBuildProblem(t.dir), null);
+ assert.equal(builtBundleProblem(t.dir, "reports-site"), null);
+ assert.equal(builtSiteProblem(t.dir, "reports-site"), null);
+ } finally {
+ t.cleanup();
+ }
+});
+
+test("a cited build holding anything corpus-shaped is refused — by the audit and by both deploy checks", () => {
+ for (const extra of [
+ "summaries/manifest.json",
+ "transcripts/demo-channel/page-0000.json",
+ "posts/demo-social/manifest.json",
+ "archives/manifest.json",
+ "search-aliases.json",
+ "tags.json",
+ "sw.js",
+ "media/thumbnails/a.jpg",
+ "media/stray.mp4",
+ ]) {
+ const t = citedOut([extra]);
+ try {
+ const problem = citedBuildProblem(t.dir);
+ assert.ok(problem, extra);
+ assert.match(problem, /is a cited build, which publishes only reports and their moments, but it also holds/);
+ assert.ok(problem.includes(extra.startsWith("media/") ? extra.split("/").slice(0, 2).join("/") : extra.split("/")[0]), `${extra}: ${problem}`);
+ assert.equal(builtBundleProblem(t.dir, "reports-site"), problem, extra);
+ assert.equal(builtSiteProblem(t.dir, "reports-site"), problem, extra);
+ } finally {
+ t.cleanup();
+ }
+ }
+});
+
+test("the audit leaves a full build alone, whatever it holds", () => {
+ const t = bundle({ site: { siteId: "anilyzer" }, corpus: { spec: 5, site: { id: "anilyzer" } } });
+ mkdirSync(path.join(t.dir, "transcripts", "x"), { recursive: true });
+ try {
+ assert.equal(citedBuildProblem(t.dir), null);
+ assert.equal(builtBundleProblem(t.dir, "anilyzer"), null);
+ } finally {
+ t.cleanup();
+ }
+});
+
+test("a cited build over the Pages file limit is refused", () => {
+ const t = citedOut();
+ try {
+ const dir = path.join(t.dir, "m", "many");
+ mkdirSync(dir, { recursive: true });
+ for (let i = 0; i <= PAGES_MAX_FILES; i++) writeFileSync(path.join(dir, String(i)), "");
+ assert.match(citedBuildProblem(t.dir)!, /over Pages' limit of 20000/);
+ } finally {
+ t.cleanup();
+ }
+});
+
+test("a site configured cited with a full build is refused at deploy; a cited build of it is not", () => {
+ const full = bundle({ site: { siteId: "reports-site" }, corpus: { spec: 5, site: { id: "reports-site" } } });
+ const cited = citedOut();
+ try {
+ const site = { siteId: "reports-site", publish: "cited" };
+ assert.match(builtScopeProblem(site, full.dir)!, /publishes only its reports \(publish: cited\), but .* holds a full build/);
+ assert.equal(deployAudienceProblem(site, full.dir), builtScopeProblem(site, full.dir));
+ assert.equal(builtScopeProblem(site, cited.dir), null);
+ assert.equal(builtScopeProblem({ siteId: "reports-site" }, full.dir), null);
+ // Nothing built: the identity checks answer that, not this one.
+ const empty = tempOut();
+ assert.equal(builtScopeProblem(site, empty.dir), null);
+ empty.cleanup();
+ } finally {
+ full.cleanup();
+ cited.cleanup();
+ }
+});
diff --git a/common/publish/build.test.ts b/common/publish/build.test.ts
@@ -114,6 +114,16 @@ test("buildSiteSteps: skipData drops the data phase; skipArchives sets BUILD_ARC
);
});
+test("buildSiteSteps: allowMissingMedia lets compose through a report citation with no prepared media", () => {
+ const [compose] = buildSiteSteps({ siteId: "a", paths, skipData: true, allowMissingMedia: true, baseEnv: {} });
+ assert.equal(compose.args.join(" "), "run compose:site");
+ assert.equal(compose.env.REPORTS_ALLOW_MISSING_MEDIA, "1");
+ assert.equal(
+ buildSiteSteps({ siteId: "a", paths, baseEnv: {} })[0].env.REPORTS_ALLOW_MISSING_MEDIA,
+ undefined,
+ );
+});
+
test("buildHubSteps: compose:hub, then next build with INSTANCE_MODE=hub, in export/", () => {
const steps = buildHubSteps({ paths, baseEnv: { PATH: "/bin" } });
const env = {
diff --git a/common/publish/composeReports.test.ts b/common/publish/composeReports.test.ts
@@ -0,0 +1,551 @@
+// Integration: the reports stage of compose, through the REAL site compose
+// (bin/compose-site.ts main) over a temp corpus.
+//
+// One video channel (a record whose cues are fresh enough to read from its
+// VTT, and one whose `en` track parses to no cues so `en-orig` is read), one
+// Bluesky channel with a posts archive, a fact-check citing all five kinds
+// (and defining one citation it never cites), stills, a saved source copy, and
+// a prepared media manifest with fake clips and a capture — the cache
+// `archilyzer reports prepare` would have written. Three sites over it: a
+// CITED one, a FULL one with the same report, and a full one with none.
+//
+// Run with: node_modules/.bin/tsx --test publish/composeReports.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "compose-reports-"));
+const PINNED: Record<string, string> = {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"),
+ EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"),
+ CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"),
+ SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"),
+ CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+ ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"),
+};
+Object.assign(process.env, PINNED);
+delete process.env.REPORTS_ALLOW_MISSING_MEDIA;
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("../controller/buildIndex");
+const { writePosts } = await import("../lib/posts-server");
+const { main: composeSite } = await import("../bin/compose-site");
+const { ComposeReportsError, CITATIONS_CSV_COLUMNS } = await import("./composeReports");
+const { REPORT_MEDIA_FORMAT, REPORT_MEDIA_VERSION, reportMediaDir, reportMediaIndexFile } = await import("./reportMedia");
+const { QUOTE_CHECK_METHOD } = await import("../lib/citations/verify");
+const { parseCitationSet } = await import("../lib/citations/validate");
+const { CONTRACT } = await import("../lib/archive/contract");
+const { citedBuildProblem, builtBundleProblem } = await import("../lib/builtExport");
+
+const paths = getPaths();
+const VIDEOS = "demo-channel";
+const SOCIAL = "demo-social";
+const REPORT = "demo-report";
+const NOW_FLOOR = new Date().toISOString();
+
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const writeText = (file: string, text: string) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, text);
+};
+const readJson = <T = Record<string, unknown>>(file: string): T => JSON.parse(readFileSync(file, "utf8")) as T;
+const pub = (...p: string[]) => path.join(paths.exportPublicDir, ...p);
+
+// Every file under a directory, relative, sorted.
+function filesUnder(dir: string): string[] {
+ const out: string[] = [];
+ const walk = (d: string, rel: string) => {
+ for (const e of readdirSync(d, { withFileTypes: true })) {
+ const r = rel ? `${rel}/${e.name}` : e.name;
+ if (e.isDirectory()) walk(path.join(d, e.name), r);
+ else out.push(r);
+ }
+ };
+ if (existsSync(dir)) walk(dir, "");
+ return out.sort();
+}
+
+// A YouTube-shaped VTT: parseVtt keeps only lines carrying inline timing.
+const ts = (s: number) => new Date(s * 1000).toISOString().slice(11, 23);
+const vtt = (cues: [number, number, string][], tagged = true) =>
+ "WEBVTT\nKind: captions\nLanguage: en\n\n" +
+ cues
+ .map(([a, b, text]) => `${ts(a)} --> ${ts(b)} align:start position:0%\n${text}${tagged ? `<${ts(a)}><c></c>` : ""}\n`)
+ .join("\n");
+
+const CLIP = "0".repeat(31) + "1";
+const AUDIO_CLIP = "0".repeat(31) + "2";
+const UNCITED_CLIP = "0".repeat(31) + "3";
+
+function report(over: { c01Quote?: string } = {}) {
+ return {
+ format: "archilyzer-report",
+ version: 1,
+ id: REPORT,
+ kind: "factcheck",
+ title: "Checking a demo article",
+ subtitle: "Four claims, one stream",
+ published: "2026-10-01",
+ subject: { source: "s0" },
+ sources: {
+ s0: {
+ kind: "article",
+ title: "A demo article",
+ url: "https://example.org/article",
+ saved: "sources/s0/page.html",
+ },
+ },
+ citations: {
+ c01: {
+ kind: "video",
+ channel: VIDEOS,
+ id: "abc123",
+ start: 10,
+ end: 20,
+ pad: { before: 2, after: 3 },
+ quote: over.c01Quote ?? "The bridge opened in the spring, I was there for it.",
+ // Hand-typed: compose overwrites it.
+ verification: { quoteScore: 1, quoteCheckedAt: "2020-01-01T00:00:00Z", voiceChecked: true, method: "by hand" },
+ },
+ c02: { kind: "audio", channel: VIDEOS, id: "def456", start: 5, end: 9, quote: "words only the original track has" },
+ p01: { kind: "post", channel: SOCIAL, id: "3kabc", quote: "Posted to settle it", date: "2025-03-14" },
+ a01: { kind: "source", source: "s0", quote: "He opened the bridge himself.", image: "stills/a01.png" },
+ w01: {
+ kind: "page",
+ url: "https://example.org/page",
+ quote: "a page says so",
+ verification: { quoteScore: 0.5, quoteCheckedAt: "2020-01-01T00:00:00Z" },
+ },
+ u01: { kind: "video", channel: VIDEOS, id: "abc123", start: 40, end: 45, quote: "never cited" },
+ },
+ sections: [
+ {
+ id: "bridge",
+ title: "The bridge",
+ claims: [
+ {
+ id: "claim-1",
+ text: "He opened the bridge himself.",
+ verdict: "CONTRADICTED",
+ sourceQuote: { citation: "a01" },
+ findings: "He says [it opened without him](cite:c01), and [posted so](cite:p01).",
+ citations: ["c01", "c02", "w01"],
+ },
+ ],
+ },
+ ],
+ };
+}
+
+function seedSite(siteId: string, extra: Record<string, unknown> = {}, withReport = true) {
+ writeJson(path.join(paths.sitesDir, siteId, "site.json"), {
+ siteId,
+ siteTitle: `Site ${siteId}`,
+ siteDescription: "fixture",
+ headerTitle: siteId,
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [VIDEOS, SOCIAL].map((slug) => ({ slug, groupId: "default" })),
+ siteUrl: `https://${siteId}.example.test`,
+ archives: false,
+ ...extra,
+ });
+ if (!withReport) return;
+ const dir = path.join(paths.sitesDir, siteId, "reports", REPORT);
+ writeJson(path.join(dir, "report.json"), report());
+ writeText(path.join(dir, "stills", "a01.png"), "png-bytes");
+ writeText(path.join(dir, "sources", "s0", "page.html"), "<p>saved copy, never published</p>");
+ seedMedia(siteId);
+}
+
+// What `archilyzer reports prepare` leaves: the manifest, the clips with
+// their sidecars, the cited capture.
+function seedMedia(siteId: string, drop: string[] = []) {
+ const dir = reportMediaDir(paths, siteId);
+ const clip = (hash: string, kind: "video" | "audio", span: { from: number; to: number }) => {
+ const file = `${hash}${kind === "video" ? ".mp4" : ".m4a"}`;
+ writeText(path.join(dir, file), `${kind}-bytes-${hash}`);
+ const media = { kind, file, bytes: 20, sha256: "f".repeat(64), width: null, height: null, durationSec: span.to - span.from };
+ writeJson(path.join(dir, `${hash}.json`), { ...media, profile: "evidence-v1", span, source: { kind: "corpus-window", name: "x" } });
+ return media;
+ };
+ writeText(path.join(dir, "posts", SOCIAL, "3kabc", "shot.png"), "shot");
+ writeText(path.join(dir, "posts", SOCIAL, "3kabc", "photo.jpg"), "photo");
+ const moments: Record<string, unknown> = {
+ [`${VIDEOS}/abc123/10.00-20.00`]: clip(CLIP, "video", { from: 8, to: 23 }),
+ [`${VIDEOS}/def456/5.00-9.00`]: clip(AUDIO_CLIP, "audio", { from: 5, to: 9 }),
+ [`${VIDEOS}/abc123/40.00-45.00`]: clip(UNCITED_CLIP, "video", { from: 40, to: 45 }),
+ [`${SOCIAL}/3kabc`]: {
+ kind: "post",
+ file: `posts/${SOCIAL}/3kabc/shot.png`,
+ bytes: 4,
+ sha256: "e".repeat(64),
+ width: null,
+ height: null,
+ durationSec: null,
+ media: [{ file: `posts/${SOCIAL}/3kabc/photo.jpg`, bytes: 5, sha256: "d".repeat(64) }],
+ },
+ };
+ for (const k of drop) delete moments[k];
+ writeJson(reportMediaIndexFile(paths, siteId), {
+ format: REPORT_MEDIA_FORMAT,
+ version: REPORT_MEDIA_VERSION,
+ siteId,
+ preparedAt: "2026-10-05T00:00:00.000Z",
+ moments,
+ problems: [],
+ });
+}
+
+async function seedCorpus() {
+ writeJson(paths.settingsFile, {});
+ writeJson(path.join(paths.channelsDir, VIDEOS, "config.json"), {
+ handling: "youtube",
+ name: "Demo Channel",
+ url: "https://www.youtube.com/@demo/videos",
+ });
+ const meta = (id: string) => ({
+ id,
+ title: `Demo stream ${id}`,
+ channel: VIDEOS,
+ upload_date: "20260110",
+ duration: 600,
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ const abc = path.join(paths.channelsDir, VIDEOS, "data", "abc123");
+ writeJson(path.join(abc, "metadata.info.json"), meta("abc123"));
+ writeText(
+ path.join(abc, "transcript.en.vtt"),
+ vtt([
+ [0, 6, "Okay so somebody asked about the bridge."],
+ [6, 10, "Let me be clear about this one."],
+ [10, 15, "The bridge opened in the spring,"],
+ [15, 20, "I was there for it."],
+ [20, 30, "Anyway, back to the mail."],
+ [30, 40, "Next letter."],
+ [40, 45, "never cited"],
+ [60, 70, "Far outside any window."],
+ ]),
+ );
+ // The served `en` track parses to no cues; the original track has them.
+ const def = path.join(paths.channelsDir, VIDEOS, "data", "def456");
+ writeJson(path.join(def, "metadata.info.json"), meta("def456"));
+ writeText(path.join(def, "transcript.en.vtt"), vtt([[5, 9, "lost words"]], false));
+ writeText(path.join(def, "transcript.en-orig.vtt"), vtt([[5, 9, "Words only the original track has."]]));
+
+ writeJson(path.join(paths.channelsDir, SOCIAL, "config.json"), {
+ handling: "youtube",
+ name: "Demo Social",
+ url: "https://bsky.app/profile/demo.example",
+ sourceKind: "social",
+ platform: "bluesky",
+ socialHandle: "demo.example",
+ });
+ await writePosts(path.join(paths.channelsDir, SOCIAL), [
+ {
+ id: "3kabc",
+ slug: `${SOCIAL}/3kabc`,
+ channelSlug: SOCIAL,
+ author: "demo.example",
+ authorName: "Demo",
+ createdAt: "2025-03-14T12:00:00.000Z",
+ uploadDate: "20250314",
+ text: "Posted to settle it. The bridge was open before I got there.",
+ url: "https://bsky.app/profile/demo.example/post/3kabc",
+ platform: "bluesky",
+ isReply: false,
+ isRepost: false,
+ links: [],
+ },
+ ]);
+
+ seedSite("cited", { publish: "cited", reports: [REPORT] });
+ seedSite("full", { reports: [REPORT] });
+ seedSite("plain", {}, false);
+ await buildIndex({ paths, onLog: () => {} });
+}
+
+async function compose(siteId: string, opts: { allowMissingMedia?: boolean } = {}) {
+ const log = console.log;
+ console.log = () => {};
+ try {
+ await composeSite({ siteId, paths, ...opts });
+ } finally {
+ console.log = log;
+ }
+}
+
+await seedCorpus();
+
+test("a cited site: exactly its reports, moments, cited media and contract — the corpus pruned", async () => {
+ // A full compose leaves the corpus in public/, as the shared public dir
+ // would hold it after another site's build; and some stray files besides.
+ await compose("plain");
+ assert.ok(existsSync(pub("transcripts", VIDEOS)));
+ for (const f of ["sw.js", "tags.json", "duplicates.json", "hub-sites.json", "chart-templates.json"]) writeText(pub(f), "stale");
+ writeText(pub("archives", "manifest.json"), "{}");
+ writeText(pub("digests", VIDEOS, "manifest.json"), "{}");
+
+ await compose("cited");
+ const abc = `${VIDEOS}/abc123/10.00-20.00`;
+ const def = `${VIDEOS}/def456/5.00-9.00`;
+ assert.deepEqual(filesUnder(paths.exportPublicDir), [
+ "_headers",
+ "corpus.json",
+ "llms.txt",
+ `m/${VIDEOS}/abc123/10.00-20.00/moment.json`,
+ `m/${VIDEOS}/def456/5.00-9.00/moment.json`,
+ `m/${SOCIAL}/3kabc/moment.json`,
+ "m/index.json",
+ `media/clips/${abc}.mp4`,
+ `media/clips/${def}.m4a`,
+ `media/posts/${SOCIAL}/3kabc/photo.jpg`,
+ `media/posts/${SOCIAL}/3kabc/shot.png`,
+ `reports/${REPORT}/citations.csv`,
+ `reports/${REPORT}/citations.json`,
+ `reports/${REPORT}/page.json`,
+ `reports/${REPORT}/stills/a01.png`,
+ "reports/index.json",
+ "robots.txt",
+ "site.json",
+ "sitemap.xml",
+ ]);
+ assert.equal(readFileSync(pub("media", "clips", `${abc}.mp4`), "utf8"), `video-bytes-${CLIP}`);
+ assert.equal(readFileSync(pub("media", "clips", `${def}.m4a`), "utf8"), `audio-bytes-${AUDIO_CLIP}`);
+ // The uncited citation's clip, prepared, is not published; nor its page.
+ assert.ok(!JSON.stringify(readJson(pub("m", "index.json"))).includes("40.00"));
+});
+
+test("a cited site's contract: site.json without channels, corpus.json scope cited at spec 5, the cited llms.txt and sitemap", () => {
+ const site = readJson<{ siteId: string; channels: unknown[]; pwa: boolean }>(pub("site.json"));
+ assert.equal(site.siteId, "cited");
+ assert.deepEqual(site.channels, []);
+ assert.equal(site.pwa, false);
+ const corpus = readJson<{
+ spec: number;
+ site: { id: string; scope?: string; audience?: string };
+ channels: unknown[];
+ totals: { channels: number; videos: number };
+ reports?: { index: string; count: number };
+ }>(pub("corpus.json"));
+ assert.equal(corpus.spec, CONTRACT.corpusSpec);
+ assert.equal(corpus.spec, 5);
+ assert.equal(corpus.site.id, "cited");
+ assert.equal(corpus.site.scope, "cited");
+ assert.equal(corpus.site.audience, "public");
+ assert.deepEqual(corpus.channels, []);
+ assert.deepEqual(corpus.totals, { channels: 0, videos: 0 });
+ assert.equal(corpus.reports?.index, "https://cited.example.test/reports/index.json");
+ assert.equal(corpus.reports?.count, 1);
+
+ const llms = readFileSync(pub("llms.txt"), "utf8");
+ assert.match(llms, /^# Site cited/);
+ assert.match(llms, /\[Checking a demo article\]\(https:\/\/cited\.example\.test\/reports\/demo-report\/\): Four claims, one stream/);
+ assert.match(llms, /there is no searchable corpus here/);
+ assert.doesNotMatch(llms, /## Channels|shardScheme|page-<NNNN>|transcripts\//);
+
+ const sitemap = readFileSync(pub("sitemap.xml"), "utf8");
+ const locs = [...sitemap.matchAll(/<loc>([^<]+)<\/loc>/g)].map((m) => m[1].replace("https://cited.example.test", ""));
+ assert.deepEqual(locs, [
+ "/",
+ "/reports/",
+ "/reports/demo-report/",
+ `/m/${VIDEOS}/abc123/10.00-20.00/`,
+ `/m/${VIDEOS}/def456/5.00-9.00/`,
+ `/m/${SOCIAL}/3kabc/`,
+ ]);
+ // The bundle names itself and its scope: the deploy guards take it.
+ assert.equal(builtBundleProblem(paths.exportPublicDir, "cited"), null);
+ assert.equal(citedBuildProblem(paths.exportPublicDir), null);
+});
+
+type View = { citations: Record<string, { verification?: Record<string, unknown>; record?: Record<string, unknown>; shot?: string; text?: string; author?: string }> };
+
+test("verification is computed at compose: a hand-typed block is overwritten, a page's dropped, en-orig read when en has no cues", () => {
+ const view = readJson<View>(pub("reports", REPORT, "page.json"));
+ const v = view.citations.c01.verification!;
+ assert.equal(v.quoteScore, 1);
+ assert.equal(v.method, QUOTE_CHECK_METHOD);
+ assert.ok((v.quoteCheckedAt as string) >= NOW_FLOOR, "checked now, not when the document says");
+ assert.ok(!("voiceChecked" in v), "a voice is never vouched for by compose");
+ assert.equal(view.citations.c02.verification?.quoteScore, 1, "checked against the en-orig track");
+ assert.equal(view.citations.p01.verification?.quoteScore, 1, "a post's quote against its text");
+ assert.equal(view.citations.w01.verification, undefined);
+ assert.equal(view.citations.a01.verification, undefined);
+ assert.equal(view.citations.u01, undefined, "a citation never cited is not in the view");
+ // The record, resolved from the corpus; a cited site links no corpus.
+ assert.deepEqual(view.citations.c01.record, {
+ channel: VIDEOS,
+ channelTitle: "Demo Channel",
+ id: "abc123",
+ title: "Demo stream abc123",
+ date: "2026-01-10",
+ platform: "youtube",
+ originalUrl: "https://www.youtube.com/watch?v=abc123&t=10s",
+ });
+ assert.equal(view.citations.p01.shot, `/media/posts/${SOCIAL}/3kabc/shot.png`);
+ assert.equal(view.citations.p01.author, "Demo (@demo.example)");
+});
+
+test("moment pages: the clip with its pad, the bounded cue context, the post's capture, cited in", () => {
+ const span = readJson<Record<string, unknown> & { cues: { start: number; text: string; inSpan: boolean }[]; citedIn: { href: string }[] }>(
+ pub("m", VIDEOS, "abc123", "10.00-20.00", "moment.json"),
+ );
+ assert.equal(span.kind, "video");
+ assert.deepEqual(span.clip, { src: `/media/clips/${VIDEOS}/abc123/10.00-20.00.mp4`, start: 8, end: 23 });
+ assert.equal(span.start, 10);
+ assert.equal(span.end, 20);
+ // ±15 s of context, never the whole record.
+ assert.deepEqual(
+ span.cues.map((q) => [q.start, q.inSpan]),
+ [
+ [0, false],
+ [6, false],
+ [10, true],
+ [15, true],
+ [20, false],
+ [30, false],
+ ],
+ );
+ assert.deepEqual(
+ span.citedIn.map((e) => e.href),
+ [`/reports/${REPORT}/#claim-1`],
+ );
+ const audio = readJson<Record<string, unknown>>(pub("m", VIDEOS, "def456", "5.00-9.00", "moment.json"));
+ assert.equal(audio.kind, "audio");
+ assert.deepEqual(audio.clip, { src: `/media/clips/${VIDEOS}/def456/5.00-9.00.m4a`, start: 5, end: 9 });
+
+ const post = readJson<Record<string, unknown> & { post: unknown; record: Record<string, unknown> }>(
+ pub("m", SOCIAL, "3kabc", "moment.json"),
+ );
+ assert.equal(post.kind, "post");
+ assert.equal(post.date, "2025-03-14");
+ assert.deepEqual(post.post, {
+ author: "Demo (@demo.example)",
+ text: "Posted to settle it. The bridge was open before I got there.",
+ shot: `/media/posts/${SOCIAL}/3kabc/shot.png`,
+ media: [{ src: `/media/posts/${SOCIAL}/3kabc/photo.jpg`, kind: "image" }],
+ });
+ assert.equal(post.record.originalUrl, "https://bsky.app/profile/demo.example/post/3kabc");
+});
+
+test("the citations as files: a valid citation set without the saved copy, and one CSV row per citation", () => {
+ const set = readJson(pub("reports", REPORT, "citations.json"));
+ const parsed = parseCitationSet(set);
+ assert.ok(parsed.ok, JSON.stringify(parsed.problems));
+ assert.deepEqual(parsed.problems, []);
+ assert.deepEqual(Object.keys((set as { citations: object }).citations), ["a01", "c01", "p01", "c02", "w01"]);
+ assert.ok(!JSON.stringify(set).includes("saved"), "a source's saved copy is never published");
+
+ const csv = readFileSync(pub("reports", REPORT, "citations.csv"), "utf8").trimEnd().split("\r\n");
+ assert.equal(csv[0], CITATIONS_CSV_COLUMNS.join(","));
+ assert.equal(csv.length, 6);
+ assert.equal(
+ csv[2],
+ `c01,2,video,${VIDEOS},abc123,10,20,"The bridge opened in the spring, I was there for it.",,2026-01-10,https://www.youtube.com/watch?v=abc123&t=10s,/m/${VIDEOS}/abc123/10.00-20.00/,1`,
+ );
+ assert.match(csv[1], /^a01,1,source,,,,,He opened the bridge himself\.,,,https:\/\/example\.org\/article,,$/);
+});
+
+test("a quote that drifted from its cues fails compose, before anything is written", async () => {
+ const file = path.join(paths.sitesDir, "cited", "reports", REPORT, "report.json");
+ writeJson(file, report({ c01Quote: "He said he cut the ribbon himself that morning." }));
+ try {
+ await assert.rejects(compose("cited"), (e: unknown) => {
+ assert.ok(e instanceof ComposeReportsError);
+ assert.deepEqual(
+ e.problems.map((p) => [p.kind, p.citation]),
+ [["quote-drift", `${REPORT}#c01`]],
+ );
+ return true;
+ });
+ assert.ok(!existsSync(pub("reports")));
+ assert.ok(!existsSync(pub("site.json")), "a failed compose leaves public/ naming no site");
+ } finally {
+ writeJson(file, report());
+ }
+});
+
+test("a citation without prepared media fails compose with the list, unless --allow-missing-media", async () => {
+ seedMedia("cited", [`${VIDEOS}/abc123/10.00-20.00`]);
+ try {
+ await assert.rejects(compose("cited"), (e: unknown) => {
+ assert.ok(e instanceof ComposeReportsError);
+ assert.deepEqual(
+ e.problems.map((p) => [p.kind, p.moment]),
+ [["missing-media", `${VIDEOS}/abc123/10.00-20.00`]],
+ );
+ assert.match(e.message, /cited by demo-report#c01/);
+ return true;
+ });
+ await compose("cited", { allowMissingMedia: true });
+ const m = readJson<Record<string, unknown>>(pub("m", VIDEOS, "abc123", "10.00-20.00", "moment.json"));
+ assert.equal(m.clip, undefined);
+ assert.ok(!existsSync(pub("media", "clips", VIDEOS, "abc123")));
+ } finally {
+ seedMedia("cited");
+ }
+});
+
+test("a clip prepared for another span is stale: the reports changed since prepare", async () => {
+ const sidecar = path.join(reportMediaDir(paths, "cited"), `${CLIP}.json`);
+ const saved = readFileSync(sidecar, "utf8");
+ writeJson(sidecar, { ...JSON.parse(saved), span: { from: 10, to: 20 } });
+ try {
+ await assert.rejects(compose("cited"), (e: unknown) => {
+ assert.ok(e instanceof ComposeReportsError);
+ assert.deepEqual(e.problems.map((p) => p.kind), ["stale-media"]);
+ return true;
+ });
+ } finally {
+ writeFileSync(sidecar, saved);
+ }
+});
+
+test("a full site with reports keeps its corpus and links each moment into it", async () => {
+ await compose("full");
+ assert.ok(existsSync(pub("transcripts", VIDEOS)), "the corpus is composed as ever");
+ assert.ok(existsSync(pub("summaries", "manifest.json")));
+ const corpus = readJson<{ spec: number; site: { scope?: string; audience?: string }; channels: unknown[]; reports?: { count: number } }>(
+ pub("corpus.json"),
+ );
+ assert.equal(corpus.spec, 5);
+ assert.equal(corpus.site.scope, undefined);
+ assert.equal(corpus.site.audience, undefined);
+ assert.equal(corpus.channels.length, 2);
+ assert.equal(corpus.reports?.count, 1);
+ const m = readJson<{ record: { corpusUrl?: string } }>(pub("m", VIDEOS, "abc123", "10.00-20.00", "moment.json"));
+ assert.equal(m.record.corpusUrl, `/?v=${VIDEOS}%2Fabc123&t=10`);
+ const post = readJson<{ record: { corpusUrl?: string } }>(pub("m", SOCIAL, "3kabc", "moment.json"));
+ assert.equal(post.record.corpusUrl, `/?v=${SOCIAL}%2F3kabc&vm=post`);
+ const llms = readFileSync(pub("llms.txt"), "utf8");
+ assert.match(llms, /## Reports[^]*Checking a demo article[^]*## Channels/);
+ assert.match(readFileSync(pub("sitemap.xml"), "utf8"), /\/reports\/demo-report\//);
+ assert.equal(citedBuildProblem(paths.exportPublicDir), null, "the audit leaves a full build alone");
+});
+
+test("a site with no reports ships none of the last site's", async () => {
+ await compose("cited");
+ await compose("plain");
+ for (const entry of ["reports", "m", "media"]) assert.ok(!existsSync(pub(entry)), entry);
+ const corpus = readJson<{ reports?: unknown }>(pub("corpus.json"));
+ assert.equal(corpus.reports, undefined);
+});