commit 23dd41318f263c99f8feb4e9f010996c5ec32e91
parent 30e05773e5139737ffb8dfa277c9f5e159898f40
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 03:30:39 -0400
common: compose's reports stage (publish/composeReports.ts) — resolve, verify and write a site's reports, moments, cited media and stills; a cited site composes only them, pruning the corpus; --allow-missing-media
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
6 files changed, 889 insertions(+), 14 deletions(-)
diff --git a/.gitignore b/.gitignore
@@ -77,6 +77,10 @@ yarn-error.log*
# per-site bulk-download archive zips + manifest (regenerate with `pnpm build`)
# (no trailing slash — may be a worktree symlink, as above)
/export/public/archives
+# a report site's reports, moment pages and cited media (compose-site's reports stage)
+/export/public/reports
+/export/public/m
+/export/public/media
# oversize archives staged for the deploy-time R2 upload
/export/.r2-staging/
/export/.export-index/
diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md
@@ -132,6 +132,7 @@ The publish pipeline sets these for a process it spawns. Listed so a reader know
| `SITE_ID` | — | Which site a compose or an export build is for. `archilyzer build site <id>` sets it; `compose site` and `build site` fall back to it when no id is given. | common/bin/compose-site.ts, export/app/lib/site.ts |
| `INSTANCE_MODE` | a site | `hub` makes the export build the hub. Set by `archilyzer build hub`. | export/app/lib/mode.ts, common/lib/archive/contract.ts |
| `BUILD_ARCHIVES` | on | `0` skips archive-zip generation for one build (`--skip-archives`). | common/bin/compose-site.ts, common/bin/build-archives.ts |
+| `REPORTS_ALLOW_MISSING_MEDIA` | off | `1` lets a report citation whose evidence media was not prepared through compose (`--allow-missing-media`): its moment page renders without a clip. Off, compose fails with the list. | common/bin/compose-site.ts |
| `ARCHIVES_READONLY` | off | `1` inside a docker-mode build container: materialize archives, never write the shared cache. | common/bin/compose-site.ts |
| `HOMEPAGE_PUBLIC_DIR` | `<repo>/homepage/public` | Where `compose homepage` and `source publish` write. | common/bin/compose-homepage.ts, common/publish/source.ts |
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -46,12 +46,17 @@ export const COMMANDS: Command[] = [
},
{
path: ["compose", "site"],
- usage: "<id> compose one site's export/public (default: SITE_ID)",
+ usage:
+ "<id> [--allow-missing-media] compose one site's export/public (default: SITE_ID); --allow-missing-media lets a report citation with no prepared media through",
+ flags: { "allow-missing-media": "boolean" },
maxPositionals: 1,
- run: async ({ positionals, env }) => {
+ run: async ({ positionals, flags, env }) => {
const siteId = siteIdFrom(positionals, env, "compose site");
if (!siteId) return 2;
- await (await import("./compose-site")).main({ siteId });
+ await (await import("./compose-site")).main({
+ siteId,
+ allowMissingMedia: flags["allow-missing-media"] === true,
+ });
return 0;
},
},
@@ -74,8 +79,8 @@ export const COMMANDS: Command[] = [
{
path: ["build", "site"],
usage:
- "<id> [--nodata] [--skip-archives] data phase + compose + next build into export/out (default id: SITE_ID)",
- flags: { nodata: "boolean", "skip-archives": "boolean" },
+ "<id> [--nodata] [--skip-archives] [--allow-missing-media] data phase + compose + next build into export/out (default id: SITE_ID)",
+ flags: { nodata: "boolean", "skip-archives": "boolean", "allow-missing-media": "boolean" },
maxPositionals: 1,
run: async ({ positionals, flags, env }) => {
const siteId = siteIdFrom(positionals, env, "build site");
@@ -95,6 +100,7 @@ export const COMMANDS: Command[] = [
signal: interrupted(),
skipData: flags.nodata === true,
skipArchives: flags["skip-archives"] === true,
+ allowMissingMedia: flags["allow-missing-media"] === true,
});
if (code !== 0) console.error(`build site ${siteId}: failed (exit ${code})`);
return code;
diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts
@@ -9,9 +9,17 @@
// public/transcripts/<slug>/ <- shared/transcripts/<slug>/ (site's members)
// public/stats/* <- sites/<id>/stats/ (per-site)
// public/chart-templates.json <- sites/<id>/chart-templates.json (per-site)
+// public/reports/, m/, media/ <- the site's reports, resolved against the
+// corpus (publish/composeReports.ts)
//
// Only the site's member channels are composed, so each deployed bundle holds
// only that site's data. Checked-in static assets in public/ are left intact.
+//
+// A CITED site (site.json `publish: "cited"`) publishes its reports and the
+// moments they cite, and nothing else: its compose removes every corpus-shaped
+// entry (CORPUS_PUBLIC_ENTRIES) instead of composing it, then writes the
+// reports, site.json (no channels) and corpus.json (`site.scope: "cited"`) —
+// see composeCitedSite.
import path from "node:path";
import { cp, link, mkdir, rm, readdir, access, readFile, writeFile, stat, rename } from "node:fs/promises";
@@ -63,7 +71,13 @@ import {
import { readChannelConfig } from "../controller/channels";
import { builtSiteIdIn } from "../lib/builtExport";
import { publishedMemberSlugs } from "../lib/postsVisibility";
-import { isPrivateSite } from "../lib/siteSchema";
+import { isCitedSite, isPrivateSite } from "../lib/siteSchema";
+import { MANIFEST_VERSION, SUMMARIES_PAGE_SIZE } from "../lib/manifest";
+import {
+ composeReports,
+ reportRoutes,
+ type ComposedReports,
+} from "../publish/composeReports";
import { runIfEntryPoint } from "./_cli";
import { copyPublicFile, ownDir, writePublicFile } from "./_publicFile";
@@ -84,11 +98,17 @@ async function emitFederationFiles(
renderHeadersFile("compose-site.ts"),
);
+ // A cited site has no summaries: its descriptor names no channel, and it is
+ // ALWAYS written — the deploy guards read the bundle's identity from it.
+ const cited = isCitedSite(site);
const manifestPath = path.join(paths.exportSummariesDir, "manifest.json");
- if (!(await exists(manifestPath))) return; // no composed data → no descriptor
- const manifest = JSON.parse(await readFile(manifestPath, "utf8")) as Manifest;
+ if (!cited && !(await exists(manifestPath))) return; // no composed data → no descriptor
+ const manifest = cited
+ ? emptyManifest(site)
+ : (JSON.parse(await readFile(manifestPath, "utf8")) as Manifest);
const descriptor = buildSiteDescriptor(site, manifest, resolveSocialLinks(site), {
- pwa: shipsPwa(site),
+ // A cited site is a handful of pages: nothing to install or cache offline.
+ pwa: !cited && shipsPwa(site),
hubUrl: resolveHubUrl(site),
});
await writePublicFile(
@@ -97,6 +117,89 @@ async function emitFederationFiles(
);
}
+// A cited site's stand-in for the summaries manifest its descriptor would be
+// built from: no channels, fresh as of this compose.
+function emptyManifest(site: Site): Manifest {
+ return {
+ version: MANIFEST_VERSION,
+ totalCount: 0,
+ pageSize: SUMMARIES_PAGE_SIZE,
+ pageCount: 0,
+ generatedAt: new Date().toISOString(),
+ channels: [],
+ groups: site.groups,
+ defaultGroupId: site.defaultGroupId,
+ siteId: site.siteId,
+ };
+}
+
+// Everything a full compose writes into public/ that is the CORPUS — what a
+// cited site must not ship, whichever site's compose left it there. Removed
+// by composeCitedSite before it writes anything; lib/builtExport.ts
+// citedBuildProblem audits the built out/ against the same promise. A new
+// corpus-shaped entry in this file's full compose belongs here too.
+export const CORPUS_PUBLIC_ENTRIES: readonly string[] = [
+ "summaries",
+ "stats",
+ "transcripts",
+ "subs",
+ "posts",
+ "digests",
+ "archives",
+ "chart-templates.json",
+ "search-aliases.json",
+ TAGS_FILENAME,
+ DUPLICATES_FILENAME,
+ "sw.js",
+ // The hub's, which a site never ships.
+ "hub-sites.json",
+ "hub-summary.json",
+ // Rewritten for the cited site below.
+ "corpus.json",
+ "llms.txt",
+ "robots.txt",
+ "sitemap.xml",
+];
+
+// Compose a CITED site: prune the corpus, then its reports and its contract.
+async function composeCitedSite(
+ site: Site,
+ paths: ReturnType<typeof getPaths>,
+ cachePath: string,
+ opts: { allowMissingMedia: boolean },
+): Promise<void> {
+ for (const entry of CORPUS_PUBLIC_ENTRIES) {
+ await rm(path.join(paths.exportPublicDir, entry), { recursive: true, force: true });
+ }
+ // …and an earlier full compose's staged oversize archives, which the deploy
+ // would otherwise upload under this site's name.
+ await rm(path.join(path.dirname(paths.exportPublicDir), ".r2-staging", site.siteId), {
+ recursive: true,
+ force: true,
+ });
+ console.log(`[compose] ${site.siteId} publishes only its reports: the corpus is not composed.`);
+
+ const composed = await composeReports({
+ paths,
+ site,
+ allowMissingMedia: opts.allowMissingMedia,
+ log: console.log,
+ });
+ await emitFederationFiles(site, paths);
+ await emitAiFiles(site, paths, composed);
+ // Nothing of the corpus is in public/ now: the next full compose of this
+ // site composes every stage afresh.
+ await writeComposeCache(cachePath, { transcripts: {}, subs: {}, posts: {}, digests: {} });
+ console.log(
+ `Composed cited site "${site.siteId}" into ${paths.exportPublicDir} ` +
+ `(${composed.reports.length} report(s), ${composed.moments.length} moment(s)).`,
+ );
+}
+
+// `--allow-missing-media` (archilyzer compose site / build site), or the
+// environment variable the build passes it to compose as.
+export const ALLOW_MISSING_MEDIA_ENV = "REPORTS_ALLOW_MISSING_MEDIA";
+
// Emit the AI-discovery surface — a small FIXED set of site-root files
// (llms.txt, corpus.json, robots.txt, sitemap.xml). These document how to
// navigate the already-served paginated shards; they never enumerate per-video
@@ -105,6 +208,7 @@ async function emitFederationFiles(
async function emitAiFiles(
site: Site,
paths: ReturnType<typeof getPaths>,
+ composed: ComposedReports,
): Promise<void> {
const sitePath = path.join(paths.exportPublicDir, "site.json");
if (!(await exists(sitePath))) return; // no composed data → nothing to describe
@@ -171,6 +275,8 @@ async function emitAiFiles(
// A private site says so in its own corpus.json — the bundle's word that
// the deploy guard (lib/builtExport.ts builtAudienceProblem) reads.
private: isPrivateSite(site),
+ reportCount: composed.reports.length,
+ cited: isCitedSite(site),
});
await writePublicFile(
path.join(paths.exportPublicDir, "corpus.json"),
@@ -178,7 +284,7 @@ async function emitAiFiles(
);
await writePublicFile(
path.join(paths.exportPublicDir, "llms.txt"),
- renderSiteLlmsTxt(corpus),
+ renderSiteLlmsTxt(corpus, { reports: composed.reports }),
);
await writePublicFile(
path.join(paths.exportPublicDir, "robots.txt"),
@@ -189,11 +295,14 @@ async function emitAiFiles(
// siteUrl; otherwise clear any stale copy from a previous build.
const sitemapPath = path.join(paths.exportPublicDir, "sitemap.xml");
if (descriptor.siteUrl) {
- const routes = ["/", "/changelog"];
+ // A cited site's home IS its report index; a full site's reports follow
+ // its own routes.
+ const routes = isCitedSite(site) ? ["/"] : ["/", "/changelog"];
if (hasArchives) routes.push("/downloads");
if (await exists(path.join(paths.exportPublicDir, DUPLICATES_FILENAME))) {
routes.push("/duplicates");
}
+ routes.push(...reportRoutes(composed).filter((r) => !routes.includes(r)));
await writePublicFile(
sitemapPath,
renderSitemapXml({ siteUrl: descriptor.siteUrl, routes }),
@@ -695,7 +804,7 @@ export async function reconcileChannelTree(
// `archilyzer compose site <id>` passes the id; the bare bin (export's
// compose:site script) reads SITE_ID, as it always has.
export async function main(
- opts: { siteId?: string; paths?: Paths } = {},
+ opts: { siteId?: string; paths?: Paths; allowMissingMedia?: boolean } = {},
): Promise<void> {
const siteId = opts.siteId?.trim() || process.env.SITE_ID;
if (!siteId) {
@@ -756,6 +865,13 @@ export async function main(
// short part-way leaves the next one nothing to trust.
await rm(path.join(paths.exportPublicDir, "site.json"), { force: true });
+ const allowMissingMedia =
+ opts.allowMissingMedia === true || process.env[ALLOW_MISSING_MEDIA_ENV] === "1";
+ if (isCitedSite(site)) {
+ await composeCitedSite(site, paths, cachePath, { allowMissingMedia });
+ return;
+ }
+
// --- per-site aggregates (whole-dir swaps), gated on the source signature ---
const summariesSrc = path.join(paths.exportSitesIndexDir, siteId, "summaries");
const summariesSig = await dirSignature(summariesSrc);
@@ -1037,6 +1153,11 @@ export async function main(
cache.duplicates = `${wrote ? "written" : "empty"}|${dupKey}`;
}
+ // --- reports: the site's published reports and the moments they cite ---
+ // (a site with none: whatever an earlier compose left is removed). Before
+ // site.json: a stage that fails leaves public/ naming no site.
+ const composed = await composeReports({ paths, site, allowMissingMedia, log: console.log });
+
// --- federation contract: /site.json descriptor + CORS _headers ---
await emitFederationFiles(site, paths);
// A site's bundle is not a hub's. export/public is shared with the hub build,
@@ -1054,8 +1175,8 @@ export async function main(
await composeArchives(site, memberSlugs, paths);
// --- AI discovery: llms.txt / corpus.json / robots.txt / sitemap.xml ---
- // (after site.json + archives — both feed into these fixed-count files)
- await emitAiFiles(site, paths);
+ // (after site.json + archives + reports — all feed into these fixed-count files)
+ await emitAiFiles(site, paths, composed);
// Persist the incremental-compose signatures for the next build.
await writeComposeCache(cachePath, cache);
diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts
@@ -141,6 +141,7 @@ const DECLARED: EnvVarDecl[] = [
{ name: "SITE_ID", audience: "internal", default: "—", readBy: "common/bin/compose-site.ts, export/app/lib/site.ts", doc: "Which site a compose or an export build is for. `archilyzer build site <id>` sets it; `compose site` and `build site` fall back to it when no id is given." },
{ name: "INSTANCE_MODE", audience: "internal", default: "a site", readBy: "export/app/lib/mode.ts, common/lib/archive/contract.ts", doc: "`hub` makes the export build the hub. Set by `archilyzer build hub`." },
{ name: "BUILD_ARCHIVES", audience: "internal", default: "on", readBy: "common/bin/compose-site.ts, common/bin/build-archives.ts", doc: "`0` skips archive-zip generation for one build (`--skip-archives`)." },
+ { name: "REPORTS_ALLOW_MISSING_MEDIA", audience: "internal", default: "off", readBy: "common/bin/compose-site.ts", doc: "`1` lets a report citation whose evidence media was not prepared through compose (`--allow-missing-media`): its moment page renders without a clip. Off, compose fails with the list." },
{ name: "ARCHIVES_READONLY", audience: "internal", default: "off", readBy: "common/bin/compose-site.ts", doc: "`1` inside a docker-mode build container: materialize archives, never write the shared cache." },
{ name: "HOMEPAGE_PUBLIC_DIR", audience: "internal", default: "`<repo>/homepage/public`", readBy: "common/bin/compose-homepage.ts, common/publish/source.ts", doc: "Where `compose homepage` and `source publish` write." },
diff --git a/common/publish/composeReports.ts b/common/publish/composeReports.ts
@@ -0,0 +1,742 @@
+// THE REPORTS STAGE OF COMPOSE — a site's published reports, resolved against
+// the corpus and written as the views the export's Reports pages read
+// (lib/report/views.ts names every file), with the media and stills they cite
+// (plans/report-sites.md, "Compose and the contract").
+//
+// Runs for EVERY site's compose, full or cited: it first removes what an
+// earlier compose — of this site or another, public/ is shared — left under
+// `reports/`, `m/` and `media/`, so a site with no reports ships none.
+//
+// What it reads:
+// - the site's `reports` (site.json, in order), each report.json parsed and
+// validated by the document's own checker (./reportMedia.ts
+// loadSiteReports); any problem fails the stage;
+// - per cited video/audio record: transcript.cues.json when it is fresh,
+// else metadata.info.json and the raw transcript; when those cues are
+// empty, the English VTT tracks in turn, `en-orig` first (a served `en`
+// track can parse to no cues);
+// - per cited post: the channel's posts archive (lib/posts-server.ts), and
+// the post visibility rule (lib/postsVisibility.ts);
+// - the media `archilyzer reports prepare` cut and copied for the site
+// (./reportMedia.ts — the manifest and the cache beside it). This stage
+// never cuts: the build has no ffmpeg.
+//
+// What it computes: every citation's VERIFICATION, overwriting whatever the
+// document carried (lib/citations/verify.ts): a span's quote against its cue
+// window, a post's against its text; a `source` or `page` citation keeps none.
+// A quote that drifted fails the stage.
+//
+// What it writes, under the public dir:
+// reports/index.json the report index
+// reports/<id>/page.json each report's page view
+// reports/<id>/citations.{json,csv} its citations as data (the JSON is
+// an `archilyzer-citations` set;
+// never a source's `saved` copy)
+// reports/<id>/<still> each cited source still
+// m/index.json, m/<key>/moment.json one moment view per cited moment,
+// "cited in" across every report
+// media/clips/<channel>/<id>/<s>-<e>.mp4 a span's prepared clip (.m4a for
+// a clip cut as audio)
+// media/posts/<channel>/<id>/<file> a cited post's prepared capture
+// — ONLY what the published reports cite. A citation whose media was not
+// prepared fails the stage with the list, unless `allowMissingMedia`: its page
+// then renders without a clip.
+//
+// THE STAGE FAILS BEFORE IT WRITES: every problem is collected first and
+// thrown together (ComposeReportsError), so a failed compose leaves no half
+// set of reports behind.
+
+import { copyFile, mkdir, readdir, readFile, rm, stat, writeFile } from "node:fs/promises";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import { siteChannelSlugs, type Site } from "../lib/site";
+import { isCitedSite } from "../lib/siteSchema";
+import { getSettings } from "../lib/settings";
+import { postsVisibleTo } from "../lib/postsVisibility";
+import { assertChannelTextReadable } from "../lib/channelMedia";
+import type { ChannelConfig } from "../lib/channelConfig";
+import { readChannelConfig } from "../controller/channels";
+import { isCuesJsonFresh, readNormalizedTranscript } from "../controller/normalizeTranscript";
+import { loadRawMetadataFromDir, summarize } from "../lib/transcripts-server";
+import type { TranscriptSummary } from "../lib/transcripts";
+import { parseVtt, type Cue } from "../lib/vtt";
+import { parseTranscriptJson } from "../lib/whisper";
+import { WHISPER_FILENAME, isEnglishVtt, resolvePrimaryVtt } from "../lib/videoStatus";
+import { platformMomentUrl } from "../lib/momentUrl";
+import type { Platform } from "../lib/platform";
+import { readAllPosts } from "../lib/posts-server";
+import type { Post } from "../lib/posts";
+import { readJsonFile } from "../lib/jsonFile-server";
+import { momentPath, parseMomentKey, type SpanMoment } from "../lib/citations/moments";
+import { CITATIONS_VERSION, type Citation, type PostCitation, type SpanCitation } from "../lib/citations/schema";
+import { cueWindowText, quoteDrifted, quoteVerification, QUOTE_DRIFT_THRESHOLD } from "../lib/citations/verify";
+import { buildCitedIn } from "../lib/report/citedIn";
+import type { Report } from "../lib/report/schema";
+import { reportCitationNumbers } from "../lib/report/uses";
+import {
+ MOMENT_INDEX_FORMAT,
+ MOMENT_PAGE_FORMAT,
+ MOMENTS_INDEX_PATH,
+ REPORT_INDEX_FORMAT,
+ REPORT_VIEWS_VERSION,
+ REPORTS_INDEX_PATH,
+ buildReportPageView,
+ citedInViews,
+ evidenceClipPath,
+ momentViewPath,
+ orderedCitations,
+ reportCitationsDownloadPath,
+ reportIndexEntry,
+ reportViewPath,
+ type CitationView,
+ type CueLineView,
+ type MomentPageView,
+ type MomentPostView,
+ type RecordView,
+ type ReportIndexEntry,
+ type ReportIndexView,
+ type ReportPageView,
+} from "../lib/report/views";
+import { evidenceSpan, isAudioOnlyPlatform, type EvidenceSpan } from "../lib/evidenceClip-server";
+import {
+ citedMoments,
+ loadSiteReports,
+ readReportMediaIndex,
+ reportMediaDir,
+ siteReportDir,
+ type ReportMediaEntry,
+} from "./reportMedia";
+
+// The public dir's entries this stage owns. Every compose removes them first.
+export const REPORT_PUBLIC_ENTRIES: readonly string[] = ["reports", "m", "media"];
+
+// The transcript lines a moment page shows either side of its span, and at
+// most how many: bounded context, never the record (plans/report-sites.md,
+// Risks 5).
+export const MOMENT_CUE_CONTEXT_SECONDS = 15;
+export const MOMENT_CUE_LINES_MAX = 80;
+
+export const CITATIONS_CSV_COLUMNS = [
+ "citation",
+ "number",
+ "kind",
+ "channel",
+ "id",
+ "start",
+ "end",
+ "quote",
+ "speaker",
+ "date",
+ "originalUrl",
+ "momentPath",
+ "quoteScore",
+] as const;
+
+export type ComposeReportsProblemKind =
+ | "missing-report"
+ | "invalid-report"
+ | "not-in-site"
+ | "not-visible"
+ | "unreadable"
+ | "missing-record"
+ | "no-cues"
+ | "quote-drift"
+ | "missing-post"
+ | "missing-still"
+ | "missing-media"
+ | "stale-media";
+
+export type ComposeReportsProblem = {
+ kind: ComposeReportsProblemKind;
+ message: string;
+ report?: string;
+ // `<reportId>#<citationId>`.
+ citation?: string;
+ moment?: string;
+ // A JSON path in the report (an invalid report's problems).
+ path?: string;
+};
+
+// The kinds `allowMissingMedia` lets through: the page renders without a clip.
+const MEDIA_PROBLEMS: ReadonlySet<ComposeReportsProblemKind> = new Set(["missing-media", "stale-media"]);
+
+export class ComposeReportsError extends Error {
+ constructor(readonly problems: ComposeReportsProblem[]) {
+ super(
+ `the site's reports cannot be composed (${problems.length} problem(s)):\n` +
+ formatComposeReportsProblems(problems).map((l) => ` ${l}`).join("\n"),
+ );
+ this.name = "ComposeReportsError";
+ }
+}
+
+export function formatComposeReportsProblems(problems: readonly ComposeReportsProblem[]): string[] {
+ return problems.map((p) => {
+ const where = p.citation ?? p.moment ?? `${p.report ?? "?"}${p.path ? ` at ${p.path}` : ""}`;
+ return `${p.kind}: ${where}: ${p.message}`;
+ });
+}
+
+export type ComposeReportsOptions = {
+ paths: Paths;
+ site: Site;
+ // Default: paths.exportPublicDir.
+ publicDir?: string;
+ // Let a citation without prepared media through (`--allow-missing-media`).
+ allowMissingMedia?: boolean;
+ // `social.x.visibility` and friends; default the live settings.
+ settings?: { social?: { x?: { visibility?: unknown } } };
+ now?: () => Date;
+ log?: (line: string) => void;
+};
+
+export type ComposedReports = {
+ // The index's entries, in the site's order.
+ reports: ReportIndexEntry[];
+ // Every moment page written.
+ moments: string[];
+ // Media problems let through by `allowMissingMedia`.
+ allowed: ComposeReportsProblem[];
+};
+
+// ─── Reading the corpus ───
+
+type CitedRecord = {
+ summary: Pick<TranscriptSummary, "id" | "slug" | "title" | "uploadDate" | "platform" | "webpageUrl" | "channel">;
+ cues: Cue[];
+};
+
+// The English VTT tracks of a video dir, `en-orig` first, then the order
+// resolvePrimaryVtt prefers.
+function englishVttsByPreference(entries: readonly string[]): string[] {
+ const vtts = entries.filter(isEnglishVtt);
+ const ordered: string[] = [];
+ const orig = vtts.find((n) => n === "transcript.en-orig.vtt");
+ if (orig) ordered.push(orig);
+ const rest = vtts.filter((n) => n !== orig);
+ while (rest.length > 0) {
+ const best = resolvePrimaryVtt(rest)!;
+ ordered.push(best);
+ rest.splice(rest.indexOf(best), 1);
+ }
+ return ordered;
+}
+
+async function readCues(file: string, kind: "vtt" | "whisper"): Promise<Cue[]> {
+ try {
+ const raw = await readFile(file, "utf8");
+ return kind === "vtt" ? parseVtt(raw) : parseTranscriptJson(raw);
+ } catch {
+ return [];
+ }
+}
+
+// A cited record: its summary and its cues, or null when the data dir holds
+// neither a normalized transcript nor metadata.
+export async function readCitedRecord(
+ channelsDir: string,
+ slug: string,
+ id: string,
+ channelName?: string,
+): Promise<CitedRecord | null> {
+ const dir = path.join(channelsDir, slug, "data", id);
+ const entries = await readdir(dir).catch(() => [] as string[]);
+ if (entries.length === 0) return null;
+ let summary: CitedRecord["summary"] | null = null;
+ let cues: Cue[] = [];
+ const fresh = await isCuesJsonFresh(dir);
+ if (fresh.fresh) {
+ const n = await readNormalizedTranscript(fresh.cuesPath);
+ if (n) {
+ summary = n;
+ cues = n.cues ?? [];
+ }
+ }
+ if (!summary) {
+ const meta = await loadRawMetadataFromDir(dir);
+ if (!meta) return null;
+ summary = summarize(slug, id, meta, channelName);
+ if (entries.includes(WHISPER_FILENAME)) cues = await readCues(path.join(dir, WHISPER_FILENAME), "whisper");
+ if (cues.length === 0) {
+ const primary = resolvePrimaryVtt(entries);
+ if (primary) cues = await readCues(path.join(dir, primary), "vtt");
+ }
+ }
+ if (cues.length === 0) {
+ for (const name of englishVttsByPreference(entries)) {
+ cues = await readCues(path.join(dir, name), "vtt");
+ if (cues.length > 0) break;
+ }
+ }
+ return { summary, cues };
+}
+
+const isoDay = (uploadDate: string | undefined): string | undefined =>
+ uploadDate && /^\d{8}$/.test(uploadDate)
+ ? `${uploadDate.slice(0, 4)}-${uploadDate.slice(4, 6)}-${uploadDate.slice(6, 8)}`
+ : undefined;
+
+// A record in this site's corpus — the viewer's `?v=` link (a FULL site only).
+function corpusLink(slug: string, params: Record<string, string>): string {
+ return `/?${new URLSearchParams({ v: slug, ...params }).toString()}`;
+}
+
+const cueLine = (text: string) => text.replace(/\s+/g, " ").trim();
+
+const IMAGE_EXTS = new Set([".png", ".jpg", ".jpeg", ".gif", ".webp", ".avif"]);
+
+function csvCell(v: unknown): string {
+ if (v === undefined || v === null) return "";
+ const s = String(v);
+ return /[",\r\n]/.test(s) ? `"${s.replace(/"/g, '""')}"` : s;
+}
+
+// A report's citations as CSV: one row per citation it cites, in number order.
+export function citationsCsv(view: ReportPageView): string {
+ const rows: string[] = [CITATIONS_CSV_COLUMNS.join(",")];
+ for (const c of orderedCitations(view)) {
+ const span = c.kind === "video" || c.kind === "audio" ? c : null;
+ const recorded = c.kind === "video" || c.kind === "audio" || c.kind === "post" ? c : null;
+ const row: Record<(typeof CITATIONS_CSV_COLUMNS)[number], unknown> = {
+ citation: c.id,
+ number: c.number,
+ kind: c.kind,
+ channel: recorded?.record.channel,
+ id: recorded?.record.id,
+ start: span?.start,
+ end: span?.end,
+ quote: c.quote,
+ speaker: c.speaker,
+ date: c.date ?? recorded?.record.date,
+ originalUrl: recorded ? recorded.record.originalUrl : c.href,
+ momentPath: recorded?.href,
+ quoteScore: c.verification?.quoteScore,
+ };
+ rows.push(CITATIONS_CSV_COLUMNS.map((k) => csvCell(row[k])).join(","));
+ }
+ return rows.join("\r\n") + "\r\n";
+}
+
+// A report's citations as an `archilyzer-citations` set (CITATIONS.md): the
+// citations it cites, verification computed, and the sources they quote —
+// never a source's `saved` copy.
+export function citationSet(report: Report, view: ReportPageView): unknown {
+ const cited = orderedCitations(view).map((c) => c.id);
+ const all = report.citations ?? {};
+ const citations = Object.fromEntries(cited.map((id) => [id, all[id]]));
+ const sourceIds = new Set<string>();
+ for (const id of cited) if (all[id].kind === "source") sourceIds.add((all[id] as { source: string }).source);
+ if (report.subject) sourceIds.add(report.subject.source);
+ const sources = Object.fromEntries(
+ [...sourceIds]
+ .filter((id) => report.sources?.[id])
+ .map((id) => {
+ const { saved: _saved, ...rest } = report.sources![id];
+ void _saved;
+ return [id, rest];
+ }),
+ );
+ return {
+ format: "archilyzer-citations",
+ version: CITATIONS_VERSION,
+ ...(Object.keys(sources).length > 0 ? { sources } : {}),
+ citations,
+ };
+}
+
+const json = (value: unknown) => `${JSON.stringify(value, null, 2)}\n`;
+
+async function writeOut(publicDir: string, urlPath: string, data: string): Promise<void> {
+ const file = path.join(publicDir, ...urlPath.split("/").filter(Boolean));
+ await mkdir(path.dirname(file), { recursive: true });
+ await writeFile(file, data);
+}
+
+async function copyOut(publicDir: string, src: string, urlPath: string): Promise<void> {
+ const file = path.join(publicDir, ...urlPath.split("/").filter(Boolean));
+ await mkdir(path.dirname(file), { recursive: true });
+ await copyFile(src, file);
+}
+
+const isFile = async (p: string) => (await stat(p).catch(() => null))?.isFile() === true;
+
+const sameSpan = (a: EvidenceSpan, b: EvidenceSpan) =>
+ Math.abs(a.from - b.from) < 0.001 && Math.abs(a.to - b.to) < 0.001;
+
+// ─── The stage ───
+
+export async function composeReports(opts: ComposeReportsOptions): Promise<ComposedReports> {
+ const { paths, site } = opts;
+ const publicDir = opts.publicDir ?? paths.exportPublicDir;
+ const log = opts.log ?? (() => {});
+ const now = (opts.now?.() ?? new Date()).toISOString();
+ const cited = isCitedSite(site);
+
+ // Whatever an earlier compose left. rm removes a link, never its target (a
+ // worktree's public/ entries may be links into the primary checkout).
+ for (const entry of REPORT_PUBLIC_ENTRIES) {
+ await rm(path.join(publicDir, entry), { recursive: true, force: true });
+ }
+ if ((site.reports ?? []).length === 0) {
+ log("[reports] none published.");
+ return { reports: [], moments: [], allowed: [] };
+ }
+
+ const problems: ComposeReportsProblem[] = [];
+ const loaded = await loadSiteReports(paths, site);
+ for (const p of loaded.problems) {
+ problems.push({
+ kind: p.kind === "missing-report" ? "missing-report" : "invalid-report",
+ message: p.message,
+ report: p.report,
+ path: p.path,
+ });
+ }
+ // Compose works on its own copy: verification is overwritten below.
+ const reports: Report[] = loaded.problems.length > 0 ? [] : structuredClone(loaded.reports);
+
+ const settings = opts.settings ?? getSettings();
+ const pool = siteChannelSlugs(site);
+ const configs = new Map<string, ChannelConfig | null>();
+ const configOf = async (slug: string) => {
+ if (!configs.has(slug)) configs.set(slug, await readChannelConfig(paths, slug).catch(() => null));
+ return configs.get(slug) ?? null;
+ };
+ // Per channel, whether its text can be read (a legacy or migrating channel
+ // cannot), as the problem's sentence or null.
+ const unreadable = new Map<string, string | null>();
+ const textProblem = async (slug: string): Promise<string | null> => {
+ if (!unreadable.has(slug)) {
+ try {
+ await assertChannelTextReadable(paths, slug, await configOf(slug));
+ unreadable.set(slug, null);
+ } catch (e) {
+ unreadable.set(slug, (e as Error).message);
+ }
+ }
+ return unreadable.get(slug) ?? null;
+ };
+ const records = new Map<string, CitedRecord | null>();
+ const recordOf = async (slug: string, id: string) => {
+ const k = `${slug}/${id}`;
+ if (!records.has(k)) records.set(k, await readCitedRecord(paths.channelsDir, slug, id, (await configOf(slug))?.name));
+ return records.get(k) ?? null;
+ };
+ const postsByChannel = new Map<string, Map<string, Post>>();
+ const postOf = async (slug: string, id: string) => {
+ if (!postsByChannel.has(slug)) {
+ const posts = await readAllPosts(path.join(paths.channelsDir, slug)).catch(() => [] as Post[]);
+ postsByChannel.set(slug, new Map(posts.map((p) => [p.id, p])));
+ }
+ return postsByChannel.get(slug)!.get(id) ?? null;
+ };
+
+ // Resolve and verify every citation the reports cite. A channel outside the
+ // site's pool, or a post the site may not carry, is refused before its
+ // record is read.
+ const refusedChannel = new Set<string>();
+ for (const report of reports) {
+ const all = report.citations ?? {};
+ // Only what the report cites: a citation it defines but never cites is
+ // not in its view, and is neither checked nor published.
+ const used = new Set(reportCitationNumbers(report).keys());
+ for (const [cid, c] of Object.entries(all)) {
+ const ref = `${report.id}#${cid}`;
+ if (!used.has(cid)) continue;
+ if (c.kind === "source" || c.kind === "page") {
+ delete c.verification;
+ if (c.kind === "source" && c.image) {
+ const src = path.join(siteReportDir(paths, site.siteId, report.id), c.image);
+ if (!(await isFile(src))) {
+ problems.push({ kind: "missing-still", citation: ref, report: report.id, message: `the still ${c.image} does not exist` });
+ }
+ }
+ continue;
+ }
+ if (!pool.has(c.channel)) {
+ problems.push({ kind: "not-in-site", citation: ref, report: report.id, message: `channel "${c.channel}" is not one of this site's channels` });
+ refusedChannel.add(c.channel);
+ continue;
+ }
+ const text = await textProblem(c.channel);
+ if (text) {
+ problems.push({ kind: "unreadable", citation: ref, report: report.id, message: text });
+ continue;
+ }
+ if (c.kind === "post") {
+ if (!postsVisibleTo(site, await configOf(c.channel), settings)) {
+ problems.push({ kind: "not-visible", citation: ref, report: report.id, message: `this site may not carry posts of "${c.channel}" (the post visibility rule)` });
+ continue;
+ }
+ const post = await postOf(c.channel, c.id);
+ if (!post) {
+ problems.push({ kind: "missing-post", citation: ref, report: report.id, message: `no post ${c.id} in the posts archive of "${c.channel}"` });
+ continue;
+ }
+ c.verification = quoteVerification(c.quote, post.text, now);
+ } else {
+ const record = await recordOf(c.channel, c.id);
+ if (!record) {
+ problems.push({ kind: "missing-record", citation: ref, report: report.id, message: `no record ${c.channel}/${c.id} (no metadata or transcript in its data dir)` });
+ continue;
+ }
+ if (record.cues.length === 0) {
+ problems.push({ kind: "no-cues", citation: ref, report: report.id, message: `${c.channel}/${c.id} has no transcript cues to check the quote against` });
+ continue;
+ }
+ c.verification = quoteVerification(c.quote, cueWindowText(record.cues, c.start, c.end), now);
+ }
+ if (quoteDrifted(c.verification)) {
+ problems.push({
+ kind: "quote-drift",
+ citation: ref,
+ report: report.id,
+ message:
+ `the quote matches ${Math.round((c.verification.quoteScore ?? 0) * 100)}% of what the record says there ` +
+ `(at least ${Math.round(QUOTE_DRIFT_THRESHOLD * 100)}% is required): quote it verbatim, or fix the span`,
+ });
+ }
+ }
+ }
+
+ // The prepared media, per moment.
+ const media = await readReportMediaIndex(paths, site.siteId);
+ const cacheDir = reportMediaDir(paths, site.siteId);
+ const citedIn = buildCitedIn(reports);
+ const momentInfo = new Map(citedMoments(reports).map((m) => [m.key, m]));
+ const mediaOf = new Map<string, ReportMediaEntry>();
+ for (const key of Object.keys(citedIn)) {
+ const m = momentInfo.get(key);
+ if (!m || refusedChannel.has(m.moment.channel)) continue;
+ const entry = media?.siteId === site.siteId ? media.moments[key] : undefined;
+ const citedBy = citedIn[key].map((e) => `${e.reportId}#${e.citationId}`);
+ const missing = (kind: ComposeReportsProblemKind, message: string) =>
+ problems.push({ kind, moment: key, message: `${message} (cited by ${[...new Set(citedBy)].join(", ")})` });
+ if (!entry) {
+ missing("missing-media", "no prepared media — run Prepare evidence media (archilyzer reports prepare)");
+ continue;
+ }
+ const files = entry.kind === "post" ? [entry.file, ...entry.media.map((f) => f.file)] : [entry.file];
+ const absent = [];
+ for (const f of files) if (!(await isFile(path.join(cacheDir, f)))) absent.push(f);
+ if (absent.length > 0) {
+ missing("missing-media", `the prepared media is gone from the cache (${absent.join(", ")}) — prepare again`);
+ continue;
+ }
+ if (entry.kind !== "post" && m.moment.kind === "span") {
+ // The clip was cut for a span and pad; a report changed since must not
+ // ship the old cut under the new moment's page.
+ const sidecar = await readJsonFile(path.join(cacheDir, entry.file.replace(/\.[^.]+$/, ".json")));
+ const cut = sidecar.ok ? (sidecar.value as { span?: EvidenceSpan }).span : undefined;
+ const want = evidenceSpan({ start: m.moment.start, end: m.moment.end, pad: m.pad });
+ if (cut && !sameSpan(cut, want)) {
+ missing("stale-media", `the clip was cut for ${cut.from}–${cut.to} s, the reports now cite ${want.from}–${want.to} s — prepare again`);
+ continue;
+ }
+ }
+ mediaOf.set(key, entry);
+ }
+
+ const allowed = opts.allowMissingMedia ? problems.filter((p) => MEDIA_PROBLEMS.has(p.kind)) : [];
+ const fatal = problems.filter((p) => !allowed.includes(p));
+ if (fatal.length > 0) throw new ComposeReportsError(fatal);
+ for (const line of formatComposeReportsProblems(allowed)) log(`[reports] allowed (--allow-missing-media): ${line}`);
+
+ // ─── The views ───
+
+ const recordView = async (c: SpanCitation | PostCitation): Promise<RecordView> => {
+ const config = await configOf(c.channel);
+ if (c.kind === "post") {
+ const post = (await postOf(c.channel, c.id))!;
+ return defined({
+ channel: c.channel,
+ channelTitle: config?.name ?? post.authorName,
+ id: c.id,
+ date: post.createdAt.slice(0, 10),
+ platform: post.platform,
+ originalUrl: post.url,
+ corpusUrl: cited ? undefined : corpusLink(`${c.channel}/${c.id}`, { vm: "post" }),
+ });
+ }
+ const { summary } = (await recordOf(c.channel, c.id))!;
+ const audioOnly = isAudioOnlyPlatform(config?.platform);
+ const seconds = Math.max(0, Math.floor(c.start));
+ return defined({
+ channel: c.channel,
+ channelTitle: config?.name ?? (summary.channel || undefined),
+ id: c.id,
+ title: summary.title,
+ date: isoDay(summary.uploadDate),
+ platform: summary.platform,
+ originalUrl: audioOnly
+ ? summary.webpageUrl || undefined
+ : (platformMomentUrl(summary.webpageUrl, summary.platform as Platform, c.start) ?? undefined),
+ // The viewer keys a record by its PUBLISHED slug, which on a platform
+ // with two ids is not the data dir's name.
+ corpusUrl: cited ? undefined : corpusLink(summary.slug ?? `${c.channel}/${summary.id}`, seconds > 0 ? { t: String(seconds) } : {}),
+ });
+ };
+ // buildReportPageView's resolver is synchronous: resolve every cited
+ // record first, by the citation it is resolved for.
+ const recordViews = new WeakMap<Citation, RecordView>();
+ for (const report of reports) {
+ const all = report.citations ?? {};
+ for (const id of reportCitationNumbers(report).keys()) {
+ const c = all[id];
+ if (c.kind === "video" || c.kind === "audio" || c.kind === "post") recordViews.set(c, await recordView(c));
+ }
+ }
+ const postShot = (c: { channel: string; id: string }): string | undefined => {
+ const entry = mediaOf.get(`${c.channel}/${c.id}`);
+ return entry?.kind === "post" ? `/media/${entry.file}` : undefined;
+ };
+ const views: ReportPageView[] = reports.map((report) =>
+ buildReportPageView(report, {
+ record: (c) => recordViews.get(c)!,
+ post: (c) => {
+ const post = postsByChannel.get(c.channel)?.get(c.id);
+ return post ? { author: postAuthor(post), text: post.text, shot: postShot(c) } : undefined;
+ },
+ downloads: {
+ json: reportCitationsDownloadPath(report.id, "json"),
+ csv: reportCitationsDownloadPath(report.id, "csv"),
+ },
+ }),
+ );
+
+ const index: ReportIndexView = {
+ format: REPORT_INDEX_FORMAT,
+ version: REPORT_VIEWS_VERSION,
+ reports: views.map(reportIndexEntry),
+ };
+
+ const moments: MomentPageView[] = [];
+ for (const [key, entries] of Object.entries(citedIn)) {
+ const first = entries[0];
+ const report = reports.find((r) => r.id === first.reportId)!;
+ const c = report.citations![first.citationId] as Citation;
+ if (c.kind !== "video" && c.kind !== "audio" && c.kind !== "post") continue;
+ const view = views.find((v) => v.id === report.id)!.citations[first.citationId] as Extract<
+ CitationView,
+ { kind: "video" | "audio" | "post" }
+ >;
+ const entry = mediaOf.get(key);
+ const common = {
+ format: MOMENT_PAGE_FORMAT,
+ version: REPORT_VIEWS_VERSION,
+ key,
+ record: view.record,
+ quote: c.quote,
+ ...(c.speaker ? { speaker: c.speaker } : {}),
+ ...(c.date ? { date: c.date } : {}),
+ ...(c.verification ? { verification: c.verification } : {}),
+ citedIn: citedInViews(entries, views),
+ } as const;
+ if (c.kind === "post") {
+ const post = postsByChannel.get(c.channel)!.get(c.id)!;
+ const postView: MomentPostView = {
+ author: postAuthor(post),
+ text: post.text,
+ ...(entry?.kind === "post"
+ ? {
+ shot: `/media/${entry.file}`,
+ ...(entry.media.length > 0
+ ? {
+ media: entry.media.map((f) => ({
+ src: `/media/${f.file}`,
+ kind: IMAGE_EXTS.has(path.extname(f.file).toLowerCase()) ? ("image" as const) : ("video" as const),
+ })),
+ }
+ : {}),
+ }
+ : {}),
+ };
+ moments.push({ ...common, kind: "post", post: postView });
+ continue;
+ }
+ const m = parseMomentKey(key) as SpanMoment;
+ const info = momentInfo.get(key)!;
+ const span = evidenceSpan({ start: m.start, end: m.end, pad: info.pad });
+ const { cues } = records.get(`${c.channel}/${c.id}`)!;
+ const from = m.start - MOMENT_CUE_CONTEXT_SECONDS;
+ const to = m.end + MOMENT_CUE_CONTEXT_SECONDS;
+ const lines: CueLineView[] = cues
+ .filter((q) => q.end > from && q.start < to)
+ .slice(0, MOMENT_CUE_LINES_MAX)
+ .map((q) => ({ start: q.start, end: q.end, text: cueLine(q.text), inSpan: q.end > m.start && q.start < m.end }));
+ const audio = entry?.kind === "audio";
+ moments.push({
+ ...common,
+ // A span whose clip had to be cut as sound is heard, not watched.
+ kind: audio ? "audio" : info.kind === "audio" ? "audio" : "video",
+ start: m.start,
+ end: m.end,
+ ...(entry && entry.kind !== "post" ? { clip: { src: clipPath(m, entry.kind), start: span.from, end: span.to } } : {}),
+ cues: lines,
+ });
+ }
+ moments.sort((a, b) => (a.key < b.key ? -1 : a.key > b.key ? 1 : 0));
+
+ // ─── Writing ───
+
+ await writeOut(publicDir, REPORTS_INDEX_PATH, json(index));
+ for (let i = 0; i < reports.length; i++) {
+ const report = reports[i];
+ const view = views[i];
+ await writeOut(publicDir, reportViewPath(report.id), json(view));
+ await writeOut(publicDir, reportCitationsDownloadPath(report.id, "json"), json(citationSet(report, view)));
+ await writeOut(publicDir, reportCitationsDownloadPath(report.id, "csv"), citationsCsv(view));
+ for (const c of orderedCitations(view)) {
+ if (c.kind !== "source" || !c.image) continue;
+ const rel = (report.citations![c.id] as { image: string }).image;
+ await copyOut(publicDir, path.join(siteReportDir(paths, site.siteId, report.id), rel), c.image);
+ }
+ }
+ await writeOut(
+ publicDir,
+ MOMENTS_INDEX_PATH,
+ json({ format: MOMENT_INDEX_FORMAT, version: REPORT_VIEWS_VERSION, moments: moments.map((m) => m.key) }),
+ );
+ let copied = 0;
+ for (const m of moments) {
+ await writeOut(publicDir, momentViewPath(m.key), json(m));
+ const entry = mediaOf.get(m.key);
+ if (!entry) continue;
+ if (entry.kind === "post") {
+ for (const f of [entry.file, ...entry.media.map((x) => x.file)]) {
+ await copyOut(publicDir, path.join(cacheDir, f), `/media/${f}`);
+ copied++;
+ }
+ } else {
+ await copyOut(publicDir, path.join(cacheDir, entry.file), clipPath(parseMomentKey(m.key) as SpanMoment, entry.kind));
+ copied++;
+ }
+ }
+ log(
+ `[reports] ${reports.length} report(s), ${moments.length} moment page(s), ${copied} media file(s)` +
+ `${allowed.length ? `, ${allowed.length} without media` : ""}.`,
+ );
+ return { reports: index.reports, moments: moments.map((m) => m.key), allowed };
+}
+
+// A clip's published path: the moment's (lib/report/views.ts), `.m4a` for a
+// clip cut as sound.
+function clipPath(m: SpanMoment, kind: "video" | "audio"): string {
+ const p = evidenceClipPath(m);
+ return kind === "audio" ? p.replace(/\.mp4$/, ".m4a") : p;
+}
+
+function postAuthor(post: Post): string {
+ const handle = post.author.startsWith("@") ? post.author : `@${post.author}`;
+ return post.authorName ? `${post.authorName} (${handle})` : handle;
+}
+
+const defined = <T extends object>(o: T): T =>
+ Object.fromEntries(Object.entries(o).filter(([, v]) => v !== undefined)) as T;
+
+// The site-root routes the reports add to a sitemap: the index, each report,
+// each moment page.
+export function reportRoutes(composed: Pick<ComposedReports, "reports" | "moments">): string[] {
+ if (composed.reports.length === 0) return [];
+ return ["/reports/", ...composed.reports.map((r) => r.href), ...composed.moments.map((k) => momentPath(k))];
+}