Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit d1cf5afe5eb7d362485444da55fc6d78a2e4145a
parent 23dd41318f263c99f8feb4e9f010996c5ec32e91
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 03:33:41 -0400

common: tests — compose over a temp corpus (exact files, prune, contract, verification, drift, missing and stale media, full-site colocation), the cited audit, the build env

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/lib/builtExport.test.ts | 134+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/publish/build.test.ts | 10++++++++++
Acommon/publish/composeReports.test.ts | 551+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
3 files changed, 695 insertions(+), 0 deletions(-)

diff --git a/common/lib/builtExport.test.ts b/common/lib/builtExport.test.ts @@ -8,8 +8,12 @@ import { builtHomepageAt, builtHomepageProblem, builtHubProblem, + builtScopeProblem, builtSiteIdIn, builtSiteProblem, + citedBuildProblem, + deployAudienceProblem, + PAGES_MAX_FILES, } from "./builtExport"; function tempOut(siteJson?: string): { dir: string; cleanup: () => void } { @@ -256,3 +260,133 @@ test("builtSiteProblem refuses exactly what builtBundleProblem refuses", () => { torn.cleanup(); } }); + +// ─── The cited out/ audit ─── + +// A cited build as next build leaves it: compose's files, Next's shell, the +// reports, moments and cited media. +function citedOut(extra: string[] = []): { dir: string; cleanup: () => void } { + const t = bundle({ + site: { siteId: "reports-site", channels: [] }, + corpus: { spec: 5, kind: "site", site: { id: "reports-site", scope: "cited", audience: "public" } }, + }); + const files = [ + "_next/static/chunks/app.js", + "_not-found/index.html", + "404/index.html", + "404.html", + "index.html", + "index.txt", + "__next._tree.txt", + "__next.!KHdvcmtzcGFjZSk.__PAGE__.txt", + "ask/index.html", + "changelog/index.html", + "downloads/index.html", + "duplicates/index.html", + "offline/index.html", + "reports/index.json", + "reports/index.html", + "reports/_none/index.html", + "reports/demo-report/index.html", + "reports/demo-report/page.json", + "reports/demo-report/stills/a01.png", + "m/index.json", + "m/_none/index.html", + "m/demo-channel/abc123/10.00-20.00/index.html", + "media/clips/demo-channel/abc123/10.00-20.00.mp4", + "media/posts/demo-social/3kabc/shot.png", + "icons/icon-192.png", + "favicon.ico", + "manifest.webmanifest", + "llms.txt", + "robots.txt", + "sitemap.xml", + "_headers", + "next.svg", + ...extra, + ]; + for (const f of files) { + mkdirSync(path.dirname(path.join(t.dir, f)), { recursive: true }); + writeFileSync(path.join(t.dir, f), "x"); + } + return t; +} + +test("a cited build holding only reports, moments, cited media and the shell passes, and deploys", () => { + const t = citedOut(); + try { + assert.equal(citedBuildProblem(t.dir), null); + assert.equal(builtBundleProblem(t.dir, "reports-site"), null); + assert.equal(builtSiteProblem(t.dir, "reports-site"), null); + } finally { + t.cleanup(); + } +}); + +test("a cited build holding anything corpus-shaped is refused — by the audit and by both deploy checks", () => { + for (const extra of [ + "summaries/manifest.json", + "transcripts/demo-channel/page-0000.json", + "posts/demo-social/manifest.json", + "archives/manifest.json", + "search-aliases.json", + "tags.json", + "sw.js", + "media/thumbnails/a.jpg", + "media/stray.mp4", + ]) { + const t = citedOut([extra]); + try { + const problem = citedBuildProblem(t.dir); + assert.ok(problem, extra); + assert.match(problem, /is a cited build, which publishes only reports and their moments, but it also holds/); + assert.ok(problem.includes(extra.startsWith("media/") ? extra.split("/").slice(0, 2).join("/") : extra.split("/")[0]), `${extra}: ${problem}`); + assert.equal(builtBundleProblem(t.dir, "reports-site"), problem, extra); + assert.equal(builtSiteProblem(t.dir, "reports-site"), problem, extra); + } finally { + t.cleanup(); + } + } +}); + +test("the audit leaves a full build alone, whatever it holds", () => { + const t = bundle({ site: { siteId: "anilyzer" }, corpus: { spec: 5, site: { id: "anilyzer" } } }); + mkdirSync(path.join(t.dir, "transcripts", "x"), { recursive: true }); + try { + assert.equal(citedBuildProblem(t.dir), null); + assert.equal(builtBundleProblem(t.dir, "anilyzer"), null); + } finally { + t.cleanup(); + } +}); + +test("a cited build over the Pages file limit is refused", () => { + const t = citedOut(); + try { + const dir = path.join(t.dir, "m", "many"); + mkdirSync(dir, { recursive: true }); + for (let i = 0; i <= PAGES_MAX_FILES; i++) writeFileSync(path.join(dir, String(i)), ""); + assert.match(citedBuildProblem(t.dir)!, /over Pages' limit of 20000/); + } finally { + t.cleanup(); + } +}); + +test("a site configured cited with a full build is refused at deploy; a cited build of it is not", () => { + const full = bundle({ site: { siteId: "reports-site" }, corpus: { spec: 5, site: { id: "reports-site" } } }); + const cited = citedOut(); + try { + const site = { siteId: "reports-site", publish: "cited" }; + assert.match(builtScopeProblem(site, full.dir)!, /publishes only its reports \(publish: cited\), but .* holds a full build/); + assert.equal(deployAudienceProblem(site, full.dir), builtScopeProblem(site, full.dir)); + assert.equal(builtScopeProblem(site, cited.dir), null); + assert.equal(builtScopeProblem({ siteId: "reports-site" }, full.dir), null); + // Nothing built: the identity checks answer that, not this one. + const empty = tempOut(); + assert.equal(builtScopeProblem(site, empty.dir), null); + empty.cleanup(); + } finally { + full.cleanup(); + cited.cleanup(); + } +}); diff --git a/common/publish/build.test.ts b/common/publish/build.test.ts @@ -114,6 +114,16 @@ test("buildSiteSteps: skipData drops the data phase; skipArchives sets BUILD_ARC ); }); +test("buildSiteSteps: allowMissingMedia lets compose through a report citation with no prepared media", () => { + const [compose] = buildSiteSteps({ siteId: "a", paths, skipData: true, allowMissingMedia: true, baseEnv: {} }); + assert.equal(compose.args.join(" "), "run compose:site"); + assert.equal(compose.env.REPORTS_ALLOW_MISSING_MEDIA, "1"); + assert.equal( + buildSiteSteps({ siteId: "a", paths, baseEnv: {} })[0].env.REPORTS_ALLOW_MISSING_MEDIA, + undefined, + ); +}); + test("buildHubSteps: compose:hub, then next build with INSTANCE_MODE=hub, in export/", () => { const steps = buildHubSteps({ paths, baseEnv: { PATH: "/bin" } }); const env = { diff --git a/common/publish/composeReports.test.ts b/common/publish/composeReports.test.ts @@ -0,0 +1,551 @@ +// Integration: the reports stage of compose, through the REAL site compose +// (bin/compose-site.ts main) over a temp corpus. +// +// One video channel (a record whose cues are fresh enough to read from its +// VTT, and one whose `en` track parses to no cues so `en-orig` is read), one +// Bluesky channel with a posts archive, a fact-check citing all five kinds +// (and defining one citation it never cites), stills, a saved source copy, and +// a prepared media manifest with fake clips and a capture — the cache +// `archilyzer reports prepare` would have written. Three sites over it: a +// CITED one, a FULL one with the same report, and a full one with none. +// +// Run with: node_modules/.bin/tsx --test publish/composeReports.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +const ROOT = mkdtempSync(path.join(tmpdir(), "compose-reports-")); +const PINNED: Record<string, string> = { + TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), + SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), + SITES_DIR: path.join(ROOT, "transcripts", "sites"), + SETTINGS_FILE: path.join(ROOT, "settings.json"), + EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), + EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), + EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), + EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"), + EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"), + CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"), + SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"), + CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"), + ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), + ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"), +}; +Object.assign(process.env, PINNED); +delete process.env.REPORTS_ALLOW_MISSING_MEDIA; +after(() => rmSync(ROOT, { recursive: true, force: true })); + +const { getPaths } = await import("../lib/paths"); +const { buildIndex } = await import("../controller/buildIndex"); +const { writePosts } = await import("../lib/posts-server"); +const { main: composeSite } = await import("../bin/compose-site"); +const { ComposeReportsError, CITATIONS_CSV_COLUMNS } = await import("./composeReports"); +const { REPORT_MEDIA_FORMAT, REPORT_MEDIA_VERSION, reportMediaDir, reportMediaIndexFile } = await import("./reportMedia"); +const { QUOTE_CHECK_METHOD } = await import("../lib/citations/verify"); +const { parseCitationSet } = await import("../lib/citations/validate"); +const { CONTRACT } = await import("../lib/archive/contract"); +const { citedBuildProblem, builtBundleProblem } = await import("../lib/builtExport"); + +const paths = getPaths(); +const VIDEOS = "demo-channel"; +const SOCIAL = "demo-social"; +const REPORT = "demo-report"; +const NOW_FLOOR = new Date().toISOString(); + +const writeJson = (file: string, value: unknown) => { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, JSON.stringify(value, null, 2)); +}; +const writeText = (file: string, text: string) => { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, text); +}; +const readJson = <T = Record<string, unknown>>(file: string): T => JSON.parse(readFileSync(file, "utf8")) as T; +const pub = (...p: string[]) => path.join(paths.exportPublicDir, ...p); + +// Every file under a directory, relative, sorted. +function filesUnder(dir: string): string[] { + const out: string[] = []; + const walk = (d: string, rel: string) => { + for (const e of readdirSync(d, { withFileTypes: true })) { + const r = rel ? `${rel}/${e.name}` : e.name; + if (e.isDirectory()) walk(path.join(d, e.name), r); + else out.push(r); + } + }; + if (existsSync(dir)) walk(dir, ""); + return out.sort(); +} + +// A YouTube-shaped VTT: parseVtt keeps only lines carrying inline timing. +const ts = (s: number) => new Date(s * 1000).toISOString().slice(11, 23); +const vtt = (cues: [number, number, string][], tagged = true) => + "WEBVTT\nKind: captions\nLanguage: en\n\n" + + cues + .map(([a, b, text]) => `${ts(a)} --> ${ts(b)} align:start position:0%\n${text}${tagged ? `<${ts(a)}><c></c>` : ""}\n`) + .join("\n"); + +const CLIP = "0".repeat(31) + "1"; +const AUDIO_CLIP = "0".repeat(31) + "2"; +const UNCITED_CLIP = "0".repeat(31) + "3"; + +function report(over: { c01Quote?: string } = {}) { + return { + format: "archilyzer-report", + version: 1, + id: REPORT, + kind: "factcheck", + title: "Checking a demo article", + subtitle: "Four claims, one stream", + published: "2026-10-01", + subject: { source: "s0" }, + sources: { + s0: { + kind: "article", + title: "A demo article", + url: "https://example.org/article", + saved: "sources/s0/page.html", + }, + }, + citations: { + c01: { + kind: "video", + channel: VIDEOS, + id: "abc123", + start: 10, + end: 20, + pad: { before: 2, after: 3 }, + quote: over.c01Quote ?? "The bridge opened in the spring, I was there for it.", + // Hand-typed: compose overwrites it. + verification: { quoteScore: 1, quoteCheckedAt: "2020-01-01T00:00:00Z", voiceChecked: true, method: "by hand" }, + }, + c02: { kind: "audio", channel: VIDEOS, id: "def456", start: 5, end: 9, quote: "words only the original track has" }, + p01: { kind: "post", channel: SOCIAL, id: "3kabc", quote: "Posted to settle it", date: "2025-03-14" }, + a01: { kind: "source", source: "s0", quote: "He opened the bridge himself.", image: "stills/a01.png" }, + w01: { + kind: "page", + url: "https://example.org/page", + quote: "a page says so", + verification: { quoteScore: 0.5, quoteCheckedAt: "2020-01-01T00:00:00Z" }, + }, + u01: { kind: "video", channel: VIDEOS, id: "abc123", start: 40, end: 45, quote: "never cited" }, + }, + sections: [ + { + id: "bridge", + title: "The bridge", + claims: [ + { + id: "claim-1", + text: "He opened the bridge himself.", + verdict: "CONTRADICTED", + sourceQuote: { citation: "a01" }, + findings: "He says [it opened without him](cite:c01), and [posted so](cite:p01).", + citations: ["c01", "c02", "w01"], + }, + ], + }, + ], + }; +} + +function seedSite(siteId: string, extra: Record<string, unknown> = {}, withReport = true) { + writeJson(path.join(paths.sitesDir, siteId, "site.json"), { + siteId, + siteTitle: `Site ${siteId}`, + siteDescription: "fixture", + headerTitle: siteId, + homeTagline: "", + socialLinks: [], + groups: [{ id: "default", name: "All channels", selectedByDefault: true }], + defaultGroupId: "default", + channels: [VIDEOS, SOCIAL].map((slug) => ({ slug, groupId: "default" })), + siteUrl: `https://${siteId}.example.test`, + archives: false, + ...extra, + }); + if (!withReport) return; + const dir = path.join(paths.sitesDir, siteId, "reports", REPORT); + writeJson(path.join(dir, "report.json"), report()); + writeText(path.join(dir, "stills", "a01.png"), "png-bytes"); + writeText(path.join(dir, "sources", "s0", "page.html"), "<p>saved copy, never published</p>"); + seedMedia(siteId); +} + +// What `archilyzer reports prepare` leaves: the manifest, the clips with +// their sidecars, the cited capture. +function seedMedia(siteId: string, drop: string[] = []) { + const dir = reportMediaDir(paths, siteId); + const clip = (hash: string, kind: "video" | "audio", span: { from: number; to: number }) => { + const file = `${hash}${kind === "video" ? ".mp4" : ".m4a"}`; + writeText(path.join(dir, file), `${kind}-bytes-${hash}`); + const media = { kind, file, bytes: 20, sha256: "f".repeat(64), width: null, height: null, durationSec: span.to - span.from }; + writeJson(path.join(dir, `${hash}.json`), { ...media, profile: "evidence-v1", span, source: { kind: "corpus-window", name: "x" } }); + return media; + }; + writeText(path.join(dir, "posts", SOCIAL, "3kabc", "shot.png"), "shot"); + writeText(path.join(dir, "posts", SOCIAL, "3kabc", "photo.jpg"), "photo"); + const moments: Record<string, unknown> = { + [`${VIDEOS}/abc123/10.00-20.00`]: clip(CLIP, "video", { from: 8, to: 23 }), + [`${VIDEOS}/def456/5.00-9.00`]: clip(AUDIO_CLIP, "audio", { from: 5, to: 9 }), + [`${VIDEOS}/abc123/40.00-45.00`]: clip(UNCITED_CLIP, "video", { from: 40, to: 45 }), + [`${SOCIAL}/3kabc`]: { + kind: "post", + file: `posts/${SOCIAL}/3kabc/shot.png`, + bytes: 4, + sha256: "e".repeat(64), + width: null, + height: null, + durationSec: null, + media: [{ file: `posts/${SOCIAL}/3kabc/photo.jpg`, bytes: 5, sha256: "d".repeat(64) }], + }, + }; + for (const k of drop) delete moments[k]; + writeJson(reportMediaIndexFile(paths, siteId), { + format: REPORT_MEDIA_FORMAT, + version: REPORT_MEDIA_VERSION, + siteId, + preparedAt: "2026-10-05T00:00:00.000Z", + moments, + problems: [], + }); +} + +async function seedCorpus() { + writeJson(paths.settingsFile, {}); + writeJson(path.join(paths.channelsDir, VIDEOS, "config.json"), { + handling: "youtube", + name: "Demo Channel", + url: "https://www.youtube.com/@demo/videos", + }); + const meta = (id: string) => ({ + id, + title: `Demo stream ${id}`, + channel: VIDEOS, + upload_date: "20260110", + duration: 600, + webpage_url: `https://www.youtube.com/watch?v=${id}`, + extractor_key: "Youtube", + }); + const abc = path.join(paths.channelsDir, VIDEOS, "data", "abc123"); + writeJson(path.join(abc, "metadata.info.json"), meta("abc123")); + writeText( + path.join(abc, "transcript.en.vtt"), + vtt([ + [0, 6, "Okay so somebody asked about the bridge."], + [6, 10, "Let me be clear about this one."], + [10, 15, "The bridge opened in the spring,"], + [15, 20, "I was there for it."], + [20, 30, "Anyway, back to the mail."], + [30, 40, "Next letter."], + [40, 45, "never cited"], + [60, 70, "Far outside any window."], + ]), + ); + // The served `en` track parses to no cues; the original track has them. + const def = path.join(paths.channelsDir, VIDEOS, "data", "def456"); + writeJson(path.join(def, "metadata.info.json"), meta("def456")); + writeText(path.join(def, "transcript.en.vtt"), vtt([[5, 9, "lost words"]], false)); + writeText(path.join(def, "transcript.en-orig.vtt"), vtt([[5, 9, "Words only the original track has."]])); + + writeJson(path.join(paths.channelsDir, SOCIAL, "config.json"), { + handling: "youtube", + name: "Demo Social", + url: "https://bsky.app/profile/demo.example", + sourceKind: "social", + platform: "bluesky", + socialHandle: "demo.example", + }); + await writePosts(path.join(paths.channelsDir, SOCIAL), [ + { + id: "3kabc", + slug: `${SOCIAL}/3kabc`, + channelSlug: SOCIAL, + author: "demo.example", + authorName: "Demo", + createdAt: "2025-03-14T12:00:00.000Z", + uploadDate: "20250314", + text: "Posted to settle it. The bridge was open before I got there.", + url: "https://bsky.app/profile/demo.example/post/3kabc", + platform: "bluesky", + isReply: false, + isRepost: false, + links: [], + }, + ]); + + seedSite("cited", { publish: "cited", reports: [REPORT] }); + seedSite("full", { reports: [REPORT] }); + seedSite("plain", {}, false); + await buildIndex({ paths, onLog: () => {} }); +} + +async function compose(siteId: string, opts: { allowMissingMedia?: boolean } = {}) { + const log = console.log; + console.log = () => {}; + try { + await composeSite({ siteId, paths, ...opts }); + } finally { + console.log = log; + } +} + +await seedCorpus(); + +test("a cited site: exactly its reports, moments, cited media and contract — the corpus pruned", async () => { + // A full compose leaves the corpus in public/, as the shared public dir + // would hold it after another site's build; and some stray files besides. + await compose("plain"); + assert.ok(existsSync(pub("transcripts", VIDEOS))); + for (const f of ["sw.js", "tags.json", "duplicates.json", "hub-sites.json", "chart-templates.json"]) writeText(pub(f), "stale"); + writeText(pub("archives", "manifest.json"), "{}"); + writeText(pub("digests", VIDEOS, "manifest.json"), "{}"); + + await compose("cited"); + const abc = `${VIDEOS}/abc123/10.00-20.00`; + const def = `${VIDEOS}/def456/5.00-9.00`; + assert.deepEqual(filesUnder(paths.exportPublicDir), [ + "_headers", + "corpus.json", + "llms.txt", + `m/${VIDEOS}/abc123/10.00-20.00/moment.json`, + `m/${VIDEOS}/def456/5.00-9.00/moment.json`, + `m/${SOCIAL}/3kabc/moment.json`, + "m/index.json", + `media/clips/${abc}.mp4`, + `media/clips/${def}.m4a`, + `media/posts/${SOCIAL}/3kabc/photo.jpg`, + `media/posts/${SOCIAL}/3kabc/shot.png`, + `reports/${REPORT}/citations.csv`, + `reports/${REPORT}/citations.json`, + `reports/${REPORT}/page.json`, + `reports/${REPORT}/stills/a01.png`, + "reports/index.json", + "robots.txt", + "site.json", + "sitemap.xml", + ]); + assert.equal(readFileSync(pub("media", "clips", `${abc}.mp4`), "utf8"), `video-bytes-${CLIP}`); + assert.equal(readFileSync(pub("media", "clips", `${def}.m4a`), "utf8"), `audio-bytes-${AUDIO_CLIP}`); + // The uncited citation's clip, prepared, is not published; nor its page. + assert.ok(!JSON.stringify(readJson(pub("m", "index.json"))).includes("40.00")); +}); + +test("a cited site's contract: site.json without channels, corpus.json scope cited at spec 5, the cited llms.txt and sitemap", () => { + const site = readJson<{ siteId: string; channels: unknown[]; pwa: boolean }>(pub("site.json")); + assert.equal(site.siteId, "cited"); + assert.deepEqual(site.channels, []); + assert.equal(site.pwa, false); + const corpus = readJson<{ + spec: number; + site: { id: string; scope?: string; audience?: string }; + channels: unknown[]; + totals: { channels: number; videos: number }; + reports?: { index: string; count: number }; + }>(pub("corpus.json")); + assert.equal(corpus.spec, CONTRACT.corpusSpec); + assert.equal(corpus.spec, 5); + assert.equal(corpus.site.id, "cited"); + assert.equal(corpus.site.scope, "cited"); + assert.equal(corpus.site.audience, "public"); + assert.deepEqual(corpus.channels, []); + assert.deepEqual(corpus.totals, { channels: 0, videos: 0 }); + assert.equal(corpus.reports?.index, "https://cited.example.test/reports/index.json"); + assert.equal(corpus.reports?.count, 1); + + const llms = readFileSync(pub("llms.txt"), "utf8"); + assert.match(llms, /^# Site cited/); + assert.match(llms, /\[Checking a demo article\]\(https:\/\/cited\.example\.test\/reports\/demo-report\/\): Four claims, one stream/); + assert.match(llms, /there is no searchable corpus here/); + assert.doesNotMatch(llms, /## Channels|shardScheme|page-<NNNN>|transcripts\//); + + const sitemap = readFileSync(pub("sitemap.xml"), "utf8"); + const locs = [...sitemap.matchAll(/<loc>([^<]+)<\/loc>/g)].map((m) => m[1].replace("https://cited.example.test", "")); + assert.deepEqual(locs, [ + "/", + "/reports/", + "/reports/demo-report/", + `/m/${VIDEOS}/abc123/10.00-20.00/`, + `/m/${VIDEOS}/def456/5.00-9.00/`, + `/m/${SOCIAL}/3kabc/`, + ]); + // The bundle names itself and its scope: the deploy guards take it. + assert.equal(builtBundleProblem(paths.exportPublicDir, "cited"), null); + assert.equal(citedBuildProblem(paths.exportPublicDir), null); +}); + +type View = { citations: Record<string, { verification?: Record<string, unknown>; record?: Record<string, unknown>; shot?: string; text?: string; author?: string }> }; + +test("verification is computed at compose: a hand-typed block is overwritten, a page's dropped, en-orig read when en has no cues", () => { + const view = readJson<View>(pub("reports", REPORT, "page.json")); + const v = view.citations.c01.verification!; + assert.equal(v.quoteScore, 1); + assert.equal(v.method, QUOTE_CHECK_METHOD); + assert.ok((v.quoteCheckedAt as string) >= NOW_FLOOR, "checked now, not when the document says"); + assert.ok(!("voiceChecked" in v), "a voice is never vouched for by compose"); + assert.equal(view.citations.c02.verification?.quoteScore, 1, "checked against the en-orig track"); + assert.equal(view.citations.p01.verification?.quoteScore, 1, "a post's quote against its text"); + assert.equal(view.citations.w01.verification, undefined); + assert.equal(view.citations.a01.verification, undefined); + assert.equal(view.citations.u01, undefined, "a citation never cited is not in the view"); + // The record, resolved from the corpus; a cited site links no corpus. + assert.deepEqual(view.citations.c01.record, { + channel: VIDEOS, + channelTitle: "Demo Channel", + id: "abc123", + title: "Demo stream abc123", + date: "2026-01-10", + platform: "youtube", + originalUrl: "https://www.youtube.com/watch?v=abc123&t=10s", + }); + assert.equal(view.citations.p01.shot, `/media/posts/${SOCIAL}/3kabc/shot.png`); + assert.equal(view.citations.p01.author, "Demo (@demo.example)"); +}); + +test("moment pages: the clip with its pad, the bounded cue context, the post's capture, cited in", () => { + const span = readJson<Record<string, unknown> & { cues: { start: number; text: string; inSpan: boolean }[]; citedIn: { href: string }[] }>( + pub("m", VIDEOS, "abc123", "10.00-20.00", "moment.json"), + ); + assert.equal(span.kind, "video"); + assert.deepEqual(span.clip, { src: `/media/clips/${VIDEOS}/abc123/10.00-20.00.mp4`, start: 8, end: 23 }); + assert.equal(span.start, 10); + assert.equal(span.end, 20); + // ±15 s of context, never the whole record. + assert.deepEqual( + span.cues.map((q) => [q.start, q.inSpan]), + [ + [0, false], + [6, false], + [10, true], + [15, true], + [20, false], + [30, false], + ], + ); + assert.deepEqual( + span.citedIn.map((e) => e.href), + [`/reports/${REPORT}/#claim-1`], + ); + const audio = readJson<Record<string, unknown>>(pub("m", VIDEOS, "def456", "5.00-9.00", "moment.json")); + assert.equal(audio.kind, "audio"); + assert.deepEqual(audio.clip, { src: `/media/clips/${VIDEOS}/def456/5.00-9.00.m4a`, start: 5, end: 9 }); + + const post = readJson<Record<string, unknown> & { post: unknown; record: Record<string, unknown> }>( + pub("m", SOCIAL, "3kabc", "moment.json"), + ); + assert.equal(post.kind, "post"); + assert.equal(post.date, "2025-03-14"); + assert.deepEqual(post.post, { + author: "Demo (@demo.example)", + text: "Posted to settle it. The bridge was open before I got there.", + shot: `/media/posts/${SOCIAL}/3kabc/shot.png`, + media: [{ src: `/media/posts/${SOCIAL}/3kabc/photo.jpg`, kind: "image" }], + }); + assert.equal(post.record.originalUrl, "https://bsky.app/profile/demo.example/post/3kabc"); +}); + +test("the citations as files: a valid citation set without the saved copy, and one CSV row per citation", () => { + const set = readJson(pub("reports", REPORT, "citations.json")); + const parsed = parseCitationSet(set); + assert.ok(parsed.ok, JSON.stringify(parsed.problems)); + assert.deepEqual(parsed.problems, []); + assert.deepEqual(Object.keys((set as { citations: object }).citations), ["a01", "c01", "p01", "c02", "w01"]); + assert.ok(!JSON.stringify(set).includes("saved"), "a source's saved copy is never published"); + + const csv = readFileSync(pub("reports", REPORT, "citations.csv"), "utf8").trimEnd().split("\r\n"); + assert.equal(csv[0], CITATIONS_CSV_COLUMNS.join(",")); + assert.equal(csv.length, 6); + assert.equal( + csv[2], + `c01,2,video,${VIDEOS},abc123,10,20,"The bridge opened in the spring, I was there for it.",,2026-01-10,https://www.youtube.com/watch?v=abc123&t=10s,/m/${VIDEOS}/abc123/10.00-20.00/,1`, + ); + assert.match(csv[1], /^a01,1,source,,,,,He opened the bridge himself\.,,,https:\/\/example\.org\/article,,$/); +}); + +test("a quote that drifted from its cues fails compose, before anything is written", async () => { + const file = path.join(paths.sitesDir, "cited", "reports", REPORT, "report.json"); + writeJson(file, report({ c01Quote: "He said he cut the ribbon himself that morning." })); + try { + await assert.rejects(compose("cited"), (e: unknown) => { + assert.ok(e instanceof ComposeReportsError); + assert.deepEqual( + e.problems.map((p) => [p.kind, p.citation]), + [["quote-drift", `${REPORT}#c01`]], + ); + return true; + }); + assert.ok(!existsSync(pub("reports"))); + assert.ok(!existsSync(pub("site.json")), "a failed compose leaves public/ naming no site"); + } finally { + writeJson(file, report()); + } +}); + +test("a citation without prepared media fails compose with the list, unless --allow-missing-media", async () => { + seedMedia("cited", [`${VIDEOS}/abc123/10.00-20.00`]); + try { + await assert.rejects(compose("cited"), (e: unknown) => { + assert.ok(e instanceof ComposeReportsError); + assert.deepEqual( + e.problems.map((p) => [p.kind, p.moment]), + [["missing-media", `${VIDEOS}/abc123/10.00-20.00`]], + ); + assert.match(e.message, /cited by demo-report#c01/); + return true; + }); + await compose("cited", { allowMissingMedia: true }); + const m = readJson<Record<string, unknown>>(pub("m", VIDEOS, "abc123", "10.00-20.00", "moment.json")); + assert.equal(m.clip, undefined); + assert.ok(!existsSync(pub("media", "clips", VIDEOS, "abc123"))); + } finally { + seedMedia("cited"); + } +}); + +test("a clip prepared for another span is stale: the reports changed since prepare", async () => { + const sidecar = path.join(reportMediaDir(paths, "cited"), `${CLIP}.json`); + const saved = readFileSync(sidecar, "utf8"); + writeJson(sidecar, { ...JSON.parse(saved), span: { from: 10, to: 20 } }); + try { + await assert.rejects(compose("cited"), (e: unknown) => { + assert.ok(e instanceof ComposeReportsError); + assert.deepEqual(e.problems.map((p) => p.kind), ["stale-media"]); + return true; + }); + } finally { + writeFileSync(sidecar, saved); + } +}); + +test("a full site with reports keeps its corpus and links each moment into it", async () => { + await compose("full"); + assert.ok(existsSync(pub("transcripts", VIDEOS)), "the corpus is composed as ever"); + assert.ok(existsSync(pub("summaries", "manifest.json"))); + const corpus = readJson<{ spec: number; site: { scope?: string; audience?: string }; channels: unknown[]; reports?: { count: number } }>( + pub("corpus.json"), + ); + assert.equal(corpus.spec, 5); + assert.equal(corpus.site.scope, undefined); + assert.equal(corpus.site.audience, undefined); + assert.equal(corpus.channels.length, 2); + assert.equal(corpus.reports?.count, 1); + const m = readJson<{ record: { corpusUrl?: string } }>(pub("m", VIDEOS, "abc123", "10.00-20.00", "moment.json")); + assert.equal(m.record.corpusUrl, `/?v=${VIDEOS}%2Fabc123&t=10`); + const post = readJson<{ record: { corpusUrl?: string } }>(pub("m", SOCIAL, "3kabc", "moment.json")); + assert.equal(post.record.corpusUrl, `/?v=${SOCIAL}%2F3kabc&vm=post`); + const llms = readFileSync(pub("llms.txt"), "utf8"); + assert.match(llms, /## Reports[^]*Checking a demo article[^]*## Channels/); + assert.match(readFileSync(pub("sitemap.xml"), "utf8"), /\/reports\/demo-report\//); + assert.equal(citedBuildProblem(paths.exportPublicDir), null, "the audit leaves a full build alone"); +}); + +test("a site with no reports ships none of the last site's", async () => { + await compose("cited"); + await compose("plain"); + for (const entry of ["reports", "m", "media"]) assert.ok(!existsSync(pub(entry)), entry); + const corpus = readJson<{ reports?: unknown }>(pub("corpus.json")); + assert.equal(corpus.reports, undefined); +});