Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 23dd41318f263c99f8feb4e9f010996c5ec32e91
parent 30e05773e5139737ffb8dfa277c9f5e159898f40
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 03:30:39 -0400

common: compose's reports stage (publish/composeReports.ts) — resolve, verify and write a site's reports, moments, cited media and stills; a cited site composes only them, pruning the corpus; --allow-missing-media

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
M.gitignore | 4++++
MENVIRONMENT.md | 1+
Mcommon/bin/archilyzer.ts | 16+++++++++++-----
Mcommon/bin/compose-site.ts | 139+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Mcommon/lib/envVars.ts | 1+
Acommon/publish/composeReports.ts | 742+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
6 files changed, 889 insertions(+), 14 deletions(-)

diff --git a/.gitignore b/.gitignore @@ -77,6 +77,10 @@ yarn-error.log* # per-site bulk-download archive zips + manifest (regenerate with `pnpm build`) # (no trailing slash — may be a worktree symlink, as above) /export/public/archives +# a report site's reports, moment pages and cited media (compose-site's reports stage) +/export/public/reports +/export/public/m +/export/public/media # oversize archives staged for the deploy-time R2 upload /export/.r2-staging/ /export/.export-index/ diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md @@ -132,6 +132,7 @@ The publish pipeline sets these for a process it spawns. Listed so a reader know | `SITE_ID` | — | Which site a compose or an export build is for. `archilyzer build site <id>` sets it; `compose site` and `build site` fall back to it when no id is given. | common/bin/compose-site.ts, export/app/lib/site.ts | | `INSTANCE_MODE` | a site | `hub` makes the export build the hub. Set by `archilyzer build hub`. | export/app/lib/mode.ts, common/lib/archive/contract.ts | | `BUILD_ARCHIVES` | on | `0` skips archive-zip generation for one build (`--skip-archives`). | common/bin/compose-site.ts, common/bin/build-archives.ts | +| `REPORTS_ALLOW_MISSING_MEDIA` | off | `1` lets a report citation whose evidence media was not prepared through compose (`--allow-missing-media`): its moment page renders without a clip. Off, compose fails with the list. | common/bin/compose-site.ts | | `ARCHIVES_READONLY` | off | `1` inside a docker-mode build container: materialize archives, never write the shared cache. | common/bin/compose-site.ts | | `HOMEPAGE_PUBLIC_DIR` | `<repo>/homepage/public` | Where `compose homepage` and `source publish` write. | common/bin/compose-homepage.ts, common/publish/source.ts | diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts @@ -46,12 +46,17 @@ export const COMMANDS: Command[] = [ }, { path: ["compose", "site"], - usage: "<id> compose one site's export/public (default: SITE_ID)", + usage: + "<id> [--allow-missing-media] compose one site's export/public (default: SITE_ID); --allow-missing-media lets a report citation with no prepared media through", + flags: { "allow-missing-media": "boolean" }, maxPositionals: 1, - run: async ({ positionals, env }) => { + run: async ({ positionals, flags, env }) => { const siteId = siteIdFrom(positionals, env, "compose site"); if (!siteId) return 2; - await (await import("./compose-site")).main({ siteId }); + await (await import("./compose-site")).main({ + siteId, + allowMissingMedia: flags["allow-missing-media"] === true, + }); return 0; }, }, @@ -74,8 +79,8 @@ export const COMMANDS: Command[] = [ { path: ["build", "site"], usage: - "<id> [--nodata] [--skip-archives] data phase + compose + next build into export/out (default id: SITE_ID)", - flags: { nodata: "boolean", "skip-archives": "boolean" }, + "<id> [--nodata] [--skip-archives] [--allow-missing-media] data phase + compose + next build into export/out (default id: SITE_ID)", + flags: { nodata: "boolean", "skip-archives": "boolean", "allow-missing-media": "boolean" }, maxPositionals: 1, run: async ({ positionals, flags, env }) => { const siteId = siteIdFrom(positionals, env, "build site"); @@ -95,6 +100,7 @@ export const COMMANDS: Command[] = [ signal: interrupted(), skipData: flags.nodata === true, skipArchives: flags["skip-archives"] === true, + allowMissingMedia: flags["allow-missing-media"] === true, }); if (code !== 0) console.error(`build site ${siteId}: failed (exit ${code})`); return code; diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts @@ -9,9 +9,17 @@ // public/transcripts/<slug>/ <- shared/transcripts/<slug>/ (site's members) // public/stats/* <- sites/<id>/stats/ (per-site) // public/chart-templates.json <- sites/<id>/chart-templates.json (per-site) +// public/reports/, m/, media/ <- the site's reports, resolved against the +// corpus (publish/composeReports.ts) // // Only the site's member channels are composed, so each deployed bundle holds // only that site's data. Checked-in static assets in public/ are left intact. +// +// A CITED site (site.json `publish: "cited"`) publishes its reports and the +// moments they cite, and nothing else: its compose removes every corpus-shaped +// entry (CORPUS_PUBLIC_ENTRIES) instead of composing it, then writes the +// reports, site.json (no channels) and corpus.json (`site.scope: "cited"`) — +// see composeCitedSite. import path from "node:path"; import { cp, link, mkdir, rm, readdir, access, readFile, writeFile, stat, rename } from "node:fs/promises"; @@ -63,7 +71,13 @@ import { import { readChannelConfig } from "../controller/channels"; import { builtSiteIdIn } from "../lib/builtExport"; import { publishedMemberSlugs } from "../lib/postsVisibility"; -import { isPrivateSite } from "../lib/siteSchema"; +import { isCitedSite, isPrivateSite } from "../lib/siteSchema"; +import { MANIFEST_VERSION, SUMMARIES_PAGE_SIZE } from "../lib/manifest"; +import { + composeReports, + reportRoutes, + type ComposedReports, +} from "../publish/composeReports"; import { runIfEntryPoint } from "./_cli"; import { copyPublicFile, ownDir, writePublicFile } from "./_publicFile"; @@ -84,11 +98,17 @@ async function emitFederationFiles( renderHeadersFile("compose-site.ts"), ); + // A cited site has no summaries: its descriptor names no channel, and it is + // ALWAYS written — the deploy guards read the bundle's identity from it. + const cited = isCitedSite(site); const manifestPath = path.join(paths.exportSummariesDir, "manifest.json"); - if (!(await exists(manifestPath))) return; // no composed data → no descriptor - const manifest = JSON.parse(await readFile(manifestPath, "utf8")) as Manifest; + if (!cited && !(await exists(manifestPath))) return; // no composed data → no descriptor + const manifest = cited + ? emptyManifest(site) + : (JSON.parse(await readFile(manifestPath, "utf8")) as Manifest); const descriptor = buildSiteDescriptor(site, manifest, resolveSocialLinks(site), { - pwa: shipsPwa(site), + // A cited site is a handful of pages: nothing to install or cache offline. + pwa: !cited && shipsPwa(site), hubUrl: resolveHubUrl(site), }); await writePublicFile( @@ -97,6 +117,89 @@ async function emitFederationFiles( ); } +// A cited site's stand-in for the summaries manifest its descriptor would be +// built from: no channels, fresh as of this compose. +function emptyManifest(site: Site): Manifest { + return { + version: MANIFEST_VERSION, + totalCount: 0, + pageSize: SUMMARIES_PAGE_SIZE, + pageCount: 0, + generatedAt: new Date().toISOString(), + channels: [], + groups: site.groups, + defaultGroupId: site.defaultGroupId, + siteId: site.siteId, + }; +} + +// Everything a full compose writes into public/ that is the CORPUS — what a +// cited site must not ship, whichever site's compose left it there. Removed +// by composeCitedSite before it writes anything; lib/builtExport.ts +// citedBuildProblem audits the built out/ against the same promise. A new +// corpus-shaped entry in this file's full compose belongs here too. +export const CORPUS_PUBLIC_ENTRIES: readonly string[] = [ + "summaries", + "stats", + "transcripts", + "subs", + "posts", + "digests", + "archives", + "chart-templates.json", + "search-aliases.json", + TAGS_FILENAME, + DUPLICATES_FILENAME, + "sw.js", + // The hub's, which a site never ships. + "hub-sites.json", + "hub-summary.json", + // Rewritten for the cited site below. + "corpus.json", + "llms.txt", + "robots.txt", + "sitemap.xml", +]; + +// Compose a CITED site: prune the corpus, then its reports and its contract. +async function composeCitedSite( + site: Site, + paths: ReturnType<typeof getPaths>, + cachePath: string, + opts: { allowMissingMedia: boolean }, +): Promise<void> { + for (const entry of CORPUS_PUBLIC_ENTRIES) { + await rm(path.join(paths.exportPublicDir, entry), { recursive: true, force: true }); + } + // …and an earlier full compose's staged oversize archives, which the deploy + // would otherwise upload under this site's name. + await rm(path.join(path.dirname(paths.exportPublicDir), ".r2-staging", site.siteId), { + recursive: true, + force: true, + }); + console.log(`[compose] ${site.siteId} publishes only its reports: the corpus is not composed.`); + + const composed = await composeReports({ + paths, + site, + allowMissingMedia: opts.allowMissingMedia, + log: console.log, + }); + await emitFederationFiles(site, paths); + await emitAiFiles(site, paths, composed); + // Nothing of the corpus is in public/ now: the next full compose of this + // site composes every stage afresh. + await writeComposeCache(cachePath, { transcripts: {}, subs: {}, posts: {}, digests: {} }); + console.log( + `Composed cited site "${site.siteId}" into ${paths.exportPublicDir} ` + + `(${composed.reports.length} report(s), ${composed.moments.length} moment(s)).`, + ); +} + +// `--allow-missing-media` (archilyzer compose site / build site), or the +// environment variable the build passes it to compose as. +export const ALLOW_MISSING_MEDIA_ENV = "REPORTS_ALLOW_MISSING_MEDIA"; + // Emit the AI-discovery surface — a small FIXED set of site-root files // (llms.txt, corpus.json, robots.txt, sitemap.xml). These document how to // navigate the already-served paginated shards; they never enumerate per-video @@ -105,6 +208,7 @@ async function emitFederationFiles( async function emitAiFiles( site: Site, paths: ReturnType<typeof getPaths>, + composed: ComposedReports, ): Promise<void> { const sitePath = path.join(paths.exportPublicDir, "site.json"); if (!(await exists(sitePath))) return; // no composed data → nothing to describe @@ -171,6 +275,8 @@ async function emitAiFiles( // A private site says so in its own corpus.json — the bundle's word that // the deploy guard (lib/builtExport.ts builtAudienceProblem) reads. private: isPrivateSite(site), + reportCount: composed.reports.length, + cited: isCitedSite(site), }); await writePublicFile( path.join(paths.exportPublicDir, "corpus.json"), @@ -178,7 +284,7 @@ async function emitAiFiles( ); await writePublicFile( path.join(paths.exportPublicDir, "llms.txt"), - renderSiteLlmsTxt(corpus), + renderSiteLlmsTxt(corpus, { reports: composed.reports }), ); await writePublicFile( path.join(paths.exportPublicDir, "robots.txt"), @@ -189,11 +295,14 @@ async function emitAiFiles( // siteUrl; otherwise clear any stale copy from a previous build. const sitemapPath = path.join(paths.exportPublicDir, "sitemap.xml"); if (descriptor.siteUrl) { - const routes = ["/", "/changelog"]; + // A cited site's home IS its report index; a full site's reports follow + // its own routes. + const routes = isCitedSite(site) ? ["/"] : ["/", "/changelog"]; if (hasArchives) routes.push("/downloads"); if (await exists(path.join(paths.exportPublicDir, DUPLICATES_FILENAME))) { routes.push("/duplicates"); } + routes.push(...reportRoutes(composed).filter((r) => !routes.includes(r))); await writePublicFile( sitemapPath, renderSitemapXml({ siteUrl: descriptor.siteUrl, routes }), @@ -695,7 +804,7 @@ export async function reconcileChannelTree( // `archilyzer compose site <id>` passes the id; the bare bin (export's // compose:site script) reads SITE_ID, as it always has. export async function main( - opts: { siteId?: string; paths?: Paths } = {}, + opts: { siteId?: string; paths?: Paths; allowMissingMedia?: boolean } = {}, ): Promise<void> { const siteId = opts.siteId?.trim() || process.env.SITE_ID; if (!siteId) { @@ -756,6 +865,13 @@ export async function main( // short part-way leaves the next one nothing to trust. await rm(path.join(paths.exportPublicDir, "site.json"), { force: true }); + const allowMissingMedia = + opts.allowMissingMedia === true || process.env[ALLOW_MISSING_MEDIA_ENV] === "1"; + if (isCitedSite(site)) { + await composeCitedSite(site, paths, cachePath, { allowMissingMedia }); + return; + } + // --- per-site aggregates (whole-dir swaps), gated on the source signature --- const summariesSrc = path.join(paths.exportSitesIndexDir, siteId, "summaries"); const summariesSig = await dirSignature(summariesSrc); @@ -1037,6 +1153,11 @@ export async function main( cache.duplicates = `${wrote ? "written" : "empty"}|${dupKey}`; } + // --- reports: the site's published reports and the moments they cite --- + // (a site with none: whatever an earlier compose left is removed). Before + // site.json: a stage that fails leaves public/ naming no site. + const composed = await composeReports({ paths, site, allowMissingMedia, log: console.log }); + // --- federation contract: /site.json descriptor + CORS _headers --- await emitFederationFiles(site, paths); // A site's bundle is not a hub's. export/public is shared with the hub build, @@ -1054,8 +1175,8 @@ export async function main( await composeArchives(site, memberSlugs, paths); // --- AI discovery: llms.txt / corpus.json / robots.txt / sitemap.xml --- - // (after site.json + archives — both feed into these fixed-count files) - await emitAiFiles(site, paths); + // (after site.json + archives + reports — all feed into these fixed-count files) + await emitAiFiles(site, paths, composed); // Persist the incremental-compose signatures for the next build. await writeComposeCache(cachePath, cache); diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts @@ -141,6 +141,7 @@ const DECLARED: EnvVarDecl[] = [ { name: "SITE_ID", audience: "internal", default: "—", readBy: "common/bin/compose-site.ts, export/app/lib/site.ts", doc: "Which site a compose or an export build is for. `archilyzer build site <id>` sets it; `compose site` and `build site` fall back to it when no id is given." }, { name: "INSTANCE_MODE", audience: "internal", default: "a site", readBy: "export/app/lib/mode.ts, common/lib/archive/contract.ts", doc: "`hub` makes the export build the hub. Set by `archilyzer build hub`." }, { name: "BUILD_ARCHIVES", audience: "internal", default: "on", readBy: "common/bin/compose-site.ts, common/bin/build-archives.ts", doc: "`0` skips archive-zip generation for one build (`--skip-archives`)." }, + { name: "REPORTS_ALLOW_MISSING_MEDIA", audience: "internal", default: "off", readBy: "common/bin/compose-site.ts", doc: "`1` lets a report citation whose evidence media was not prepared through compose (`--allow-missing-media`): its moment page renders without a clip. Off, compose fails with the list." }, { name: "ARCHIVES_READONLY", audience: "internal", default: "off", readBy: "common/bin/compose-site.ts", doc: "`1` inside a docker-mode build container: materialize archives, never write the shared cache." }, { name: "HOMEPAGE_PUBLIC_DIR", audience: "internal", default: "`<repo>/homepage/public`", readBy: "common/bin/compose-homepage.ts, common/publish/source.ts", doc: "Where `compose homepage` and `source publish` write." }, diff --git a/common/publish/composeReports.ts b/common/publish/composeReports.ts @@ -0,0 +1,742 @@ +// THE REPORTS STAGE OF COMPOSE — a site's published reports, resolved against +// the corpus and written as the views the export's Reports pages read +// (lib/report/views.ts names every file), with the media and stills they cite +// (plans/report-sites.md, "Compose and the contract"). +// +// Runs for EVERY site's compose, full or cited: it first removes what an +// earlier compose — of this site or another, public/ is shared — left under +// `reports/`, `m/` and `media/`, so a site with no reports ships none. +// +// What it reads: +// - the site's `reports` (site.json, in order), each report.json parsed and +// validated by the document's own checker (./reportMedia.ts +// loadSiteReports); any problem fails the stage; +// - per cited video/audio record: transcript.cues.json when it is fresh, +// else metadata.info.json and the raw transcript; when those cues are +// empty, the English VTT tracks in turn, `en-orig` first (a served `en` +// track can parse to no cues); +// - per cited post: the channel's posts archive (lib/posts-server.ts), and +// the post visibility rule (lib/postsVisibility.ts); +// - the media `archilyzer reports prepare` cut and copied for the site +// (./reportMedia.ts — the manifest and the cache beside it). This stage +// never cuts: the build has no ffmpeg. +// +// What it computes: every citation's VERIFICATION, overwriting whatever the +// document carried (lib/citations/verify.ts): a span's quote against its cue +// window, a post's against its text; a `source` or `page` citation keeps none. +// A quote that drifted fails the stage. +// +// What it writes, under the public dir: +// reports/index.json the report index +// reports/<id>/page.json each report's page view +// reports/<id>/citations.{json,csv} its citations as data (the JSON is +// an `archilyzer-citations` set; +// never a source's `saved` copy) +// reports/<id>/<still> each cited source still +// m/index.json, m/<key>/moment.json one moment view per cited moment, +// "cited in" across every report +// media/clips/<channel>/<id>/<s>-<e>.mp4 a span's prepared clip (.m4a for +// a clip cut as audio) +// media/posts/<channel>/<id>/<file> a cited post's prepared capture +// — ONLY what the published reports cite. A citation whose media was not +// prepared fails the stage with the list, unless `allowMissingMedia`: its page +// then renders without a clip. +// +// THE STAGE FAILS BEFORE IT WRITES: every problem is collected first and +// thrown together (ComposeReportsError), so a failed compose leaves no half +// set of reports behind. + +import { copyFile, mkdir, readdir, readFile, rm, stat, writeFile } from "node:fs/promises"; +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import { siteChannelSlugs, type Site } from "../lib/site"; +import { isCitedSite } from "../lib/siteSchema"; +import { getSettings } from "../lib/settings"; +import { postsVisibleTo } from "../lib/postsVisibility"; +import { assertChannelTextReadable } from "../lib/channelMedia"; +import type { ChannelConfig } from "../lib/channelConfig"; +import { readChannelConfig } from "../controller/channels"; +import { isCuesJsonFresh, readNormalizedTranscript } from "../controller/normalizeTranscript"; +import { loadRawMetadataFromDir, summarize } from "../lib/transcripts-server"; +import type { TranscriptSummary } from "../lib/transcripts"; +import { parseVtt, type Cue } from "../lib/vtt"; +import { parseTranscriptJson } from "../lib/whisper"; +import { WHISPER_FILENAME, isEnglishVtt, resolvePrimaryVtt } from "../lib/videoStatus"; +import { platformMomentUrl } from "../lib/momentUrl"; +import type { Platform } from "../lib/platform"; +import { readAllPosts } from "../lib/posts-server"; +import type { Post } from "../lib/posts"; +import { readJsonFile } from "../lib/jsonFile-server"; +import { momentPath, parseMomentKey, type SpanMoment } from "../lib/citations/moments"; +import { CITATIONS_VERSION, type Citation, type PostCitation, type SpanCitation } from "../lib/citations/schema"; +import { cueWindowText, quoteDrifted, quoteVerification, QUOTE_DRIFT_THRESHOLD } from "../lib/citations/verify"; +import { buildCitedIn } from "../lib/report/citedIn"; +import type { Report } from "../lib/report/schema"; +import { reportCitationNumbers } from "../lib/report/uses"; +import { + MOMENT_INDEX_FORMAT, + MOMENT_PAGE_FORMAT, + MOMENTS_INDEX_PATH, + REPORT_INDEX_FORMAT, + REPORT_VIEWS_VERSION, + REPORTS_INDEX_PATH, + buildReportPageView, + citedInViews, + evidenceClipPath, + momentViewPath, + orderedCitations, + reportCitationsDownloadPath, + reportIndexEntry, + reportViewPath, + type CitationView, + type CueLineView, + type MomentPageView, + type MomentPostView, + type RecordView, + type ReportIndexEntry, + type ReportIndexView, + type ReportPageView, +} from "../lib/report/views"; +import { evidenceSpan, isAudioOnlyPlatform, type EvidenceSpan } from "../lib/evidenceClip-server"; +import { + citedMoments, + loadSiteReports, + readReportMediaIndex, + reportMediaDir, + siteReportDir, + type ReportMediaEntry, +} from "./reportMedia"; + +// The public dir's entries this stage owns. Every compose removes them first. +export const REPORT_PUBLIC_ENTRIES: readonly string[] = ["reports", "m", "media"]; + +// The transcript lines a moment page shows either side of its span, and at +// most how many: bounded context, never the record (plans/report-sites.md, +// Risks 5). +export const MOMENT_CUE_CONTEXT_SECONDS = 15; +export const MOMENT_CUE_LINES_MAX = 80; + +export const CITATIONS_CSV_COLUMNS = [ + "citation", + "number", + "kind", + "channel", + "id", + "start", + "end", + "quote", + "speaker", + "date", + "originalUrl", + "momentPath", + "quoteScore", +] as const; + +export type ComposeReportsProblemKind = + | "missing-report" + | "invalid-report" + | "not-in-site" + | "not-visible" + | "unreadable" + | "missing-record" + | "no-cues" + | "quote-drift" + | "missing-post" + | "missing-still" + | "missing-media" + | "stale-media"; + +export type ComposeReportsProblem = { + kind: ComposeReportsProblemKind; + message: string; + report?: string; + // `<reportId>#<citationId>`. + citation?: string; + moment?: string; + // A JSON path in the report (an invalid report's problems). + path?: string; +}; + +// The kinds `allowMissingMedia` lets through: the page renders without a clip. +const MEDIA_PROBLEMS: ReadonlySet<ComposeReportsProblemKind> = new Set(["missing-media", "stale-media"]); + +export class ComposeReportsError extends Error { + constructor(readonly problems: ComposeReportsProblem[]) { + super( + `the site's reports cannot be composed (${problems.length} problem(s)):\n` + + formatComposeReportsProblems(problems).map((l) => ` ${l}`).join("\n"), + ); + this.name = "ComposeReportsError"; + } +} + +export function formatComposeReportsProblems(problems: readonly ComposeReportsProblem[]): string[] { + return problems.map((p) => { + const where = p.citation ?? p.moment ?? `${p.report ?? "?"}${p.path ? ` at ${p.path}` : ""}`; + return `${p.kind}: ${where}: ${p.message}`; + }); +} + +export type ComposeReportsOptions = { + paths: Paths; + site: Site; + // Default: paths.exportPublicDir. + publicDir?: string; + // Let a citation without prepared media through (`--allow-missing-media`). + allowMissingMedia?: boolean; + // `social.x.visibility` and friends; default the live settings. + settings?: { social?: { x?: { visibility?: unknown } } }; + now?: () => Date; + log?: (line: string) => void; +}; + +export type ComposedReports = { + // The index's entries, in the site's order. + reports: ReportIndexEntry[]; + // Every moment page written. + moments: string[]; + // Media problems let through by `allowMissingMedia`. + allowed: ComposeReportsProblem[]; +}; + +// ─── Reading the corpus ─── + +type CitedRecord = { + summary: Pick<TranscriptSummary, "id" | "slug" | "title" | "uploadDate" | "platform" | "webpageUrl" | "channel">; + cues: Cue[]; +}; + +// The English VTT tracks of a video dir, `en-orig` first, then the order +// resolvePrimaryVtt prefers. +function englishVttsByPreference(entries: readonly string[]): string[] { + const vtts = entries.filter(isEnglishVtt); + const ordered: string[] = []; + const orig = vtts.find((n) => n === "transcript.en-orig.vtt"); + if (orig) ordered.push(orig); + const rest = vtts.filter((n) => n !== orig); + while (rest.length > 0) { + const best = resolvePrimaryVtt(rest)!; + ordered.push(best); + rest.splice(rest.indexOf(best), 1); + } + return ordered; +} + +async function readCues(file: string, kind: "vtt" | "whisper"): Promise<Cue[]> { + try { + const raw = await readFile(file, "utf8"); + return kind === "vtt" ? parseVtt(raw) : parseTranscriptJson(raw); + } catch { + return []; + } +} + +// A cited record: its summary and its cues, or null when the data dir holds +// neither a normalized transcript nor metadata. +export async function readCitedRecord( + channelsDir: string, + slug: string, + id: string, + channelName?: string, +): Promise<CitedRecord | null> { + const dir = path.join(channelsDir, slug, "data", id); + const entries = await readdir(dir).catch(() => [] as string[]); + if (entries.length === 0) return null; + let summary: CitedRecord["summary"] | null = null; + let cues: Cue[] = []; + const fresh = await isCuesJsonFresh(dir); + if (fresh.fresh) { + const n = await readNormalizedTranscript(fresh.cuesPath); + if (n) { + summary = n; + cues = n.cues ?? []; + } + } + if (!summary) { + const meta = await loadRawMetadataFromDir(dir); + if (!meta) return null; + summary = summarize(slug, id, meta, channelName); + if (entries.includes(WHISPER_FILENAME)) cues = await readCues(path.join(dir, WHISPER_FILENAME), "whisper"); + if (cues.length === 0) { + const primary = resolvePrimaryVtt(entries); + if (primary) cues = await readCues(path.join(dir, primary), "vtt"); + } + } + if (cues.length === 0) { + for (const name of englishVttsByPreference(entries)) { + cues = await readCues(path.join(dir, name), "vtt"); + if (cues.length > 0) break; + } + } + return { summary, cues }; +} + +const isoDay = (uploadDate: string | undefined): string | undefined => + uploadDate && /^\d{8}$/.test(uploadDate) + ? `${uploadDate.slice(0, 4)}-${uploadDate.slice(4, 6)}-${uploadDate.slice(6, 8)}` + : undefined; + +// A record in this site's corpus — the viewer's `?v=` link (a FULL site only). +function corpusLink(slug: string, params: Record<string, string>): string { + return `/?${new URLSearchParams({ v: slug, ...params }).toString()}`; +} + +const cueLine = (text: string) => text.replace(/\s+/g, " ").trim(); + +const IMAGE_EXTS = new Set([".png", ".jpg", ".jpeg", ".gif", ".webp", ".avif"]); + +function csvCell(v: unknown): string { + if (v === undefined || v === null) return ""; + const s = String(v); + return /[",\r\n]/.test(s) ? `"${s.replace(/"/g, '""')}"` : s; +} + +// A report's citations as CSV: one row per citation it cites, in number order. +export function citationsCsv(view: ReportPageView): string { + const rows: string[] = [CITATIONS_CSV_COLUMNS.join(",")]; + for (const c of orderedCitations(view)) { + const span = c.kind === "video" || c.kind === "audio" ? c : null; + const recorded = c.kind === "video" || c.kind === "audio" || c.kind === "post" ? c : null; + const row: Record<(typeof CITATIONS_CSV_COLUMNS)[number], unknown> = { + citation: c.id, + number: c.number, + kind: c.kind, + channel: recorded?.record.channel, + id: recorded?.record.id, + start: span?.start, + end: span?.end, + quote: c.quote, + speaker: c.speaker, + date: c.date ?? recorded?.record.date, + originalUrl: recorded ? recorded.record.originalUrl : c.href, + momentPath: recorded?.href, + quoteScore: c.verification?.quoteScore, + }; + rows.push(CITATIONS_CSV_COLUMNS.map((k) => csvCell(row[k])).join(",")); + } + return rows.join("\r\n") + "\r\n"; +} + +// A report's citations as an `archilyzer-citations` set (CITATIONS.md): the +// citations it cites, verification computed, and the sources they quote — +// never a source's `saved` copy. +export function citationSet(report: Report, view: ReportPageView): unknown { + const cited = orderedCitations(view).map((c) => c.id); + const all = report.citations ?? {}; + const citations = Object.fromEntries(cited.map((id) => [id, all[id]])); + const sourceIds = new Set<string>(); + for (const id of cited) if (all[id].kind === "source") sourceIds.add((all[id] as { source: string }).source); + if (report.subject) sourceIds.add(report.subject.source); + const sources = Object.fromEntries( + [...sourceIds] + .filter((id) => report.sources?.[id]) + .map((id) => { + const { saved: _saved, ...rest } = report.sources![id]; + void _saved; + return [id, rest]; + }), + ); + return { + format: "archilyzer-citations", + version: CITATIONS_VERSION, + ...(Object.keys(sources).length > 0 ? { sources } : {}), + citations, + }; +} + +const json = (value: unknown) => `${JSON.stringify(value, null, 2)}\n`; + +async function writeOut(publicDir: string, urlPath: string, data: string): Promise<void> { + const file = path.join(publicDir, ...urlPath.split("/").filter(Boolean)); + await mkdir(path.dirname(file), { recursive: true }); + await writeFile(file, data); +} + +async function copyOut(publicDir: string, src: string, urlPath: string): Promise<void> { + const file = path.join(publicDir, ...urlPath.split("/").filter(Boolean)); + await mkdir(path.dirname(file), { recursive: true }); + await copyFile(src, file); +} + +const isFile = async (p: string) => (await stat(p).catch(() => null))?.isFile() === true; + +const sameSpan = (a: EvidenceSpan, b: EvidenceSpan) => + Math.abs(a.from - b.from) < 0.001 && Math.abs(a.to - b.to) < 0.001; + +// ─── The stage ─── + +export async function composeReports(opts: ComposeReportsOptions): Promise<ComposedReports> { + const { paths, site } = opts; + const publicDir = opts.publicDir ?? paths.exportPublicDir; + const log = opts.log ?? (() => {}); + const now = (opts.now?.() ?? new Date()).toISOString(); + const cited = isCitedSite(site); + + // Whatever an earlier compose left. rm removes a link, never its target (a + // worktree's public/ entries may be links into the primary checkout). + for (const entry of REPORT_PUBLIC_ENTRIES) { + await rm(path.join(publicDir, entry), { recursive: true, force: true }); + } + if ((site.reports ?? []).length === 0) { + log("[reports] none published."); + return { reports: [], moments: [], allowed: [] }; + } + + const problems: ComposeReportsProblem[] = []; + const loaded = await loadSiteReports(paths, site); + for (const p of loaded.problems) { + problems.push({ + kind: p.kind === "missing-report" ? "missing-report" : "invalid-report", + message: p.message, + report: p.report, + path: p.path, + }); + } + // Compose works on its own copy: verification is overwritten below. + const reports: Report[] = loaded.problems.length > 0 ? [] : structuredClone(loaded.reports); + + const settings = opts.settings ?? getSettings(); + const pool = siteChannelSlugs(site); + const configs = new Map<string, ChannelConfig | null>(); + const configOf = async (slug: string) => { + if (!configs.has(slug)) configs.set(slug, await readChannelConfig(paths, slug).catch(() => null)); + return configs.get(slug) ?? null; + }; + // Per channel, whether its text can be read (a legacy or migrating channel + // cannot), as the problem's sentence or null. + const unreadable = new Map<string, string | null>(); + const textProblem = async (slug: string): Promise<string | null> => { + if (!unreadable.has(slug)) { + try { + await assertChannelTextReadable(paths, slug, await configOf(slug)); + unreadable.set(slug, null); + } catch (e) { + unreadable.set(slug, (e as Error).message); + } + } + return unreadable.get(slug) ?? null; + }; + const records = new Map<string, CitedRecord | null>(); + const recordOf = async (slug: string, id: string) => { + const k = `${slug}/${id}`; + if (!records.has(k)) records.set(k, await readCitedRecord(paths.channelsDir, slug, id, (await configOf(slug))?.name)); + return records.get(k) ?? null; + }; + const postsByChannel = new Map<string, Map<string, Post>>(); + const postOf = async (slug: string, id: string) => { + if (!postsByChannel.has(slug)) { + const posts = await readAllPosts(path.join(paths.channelsDir, slug)).catch(() => [] as Post[]); + postsByChannel.set(slug, new Map(posts.map((p) => [p.id, p]))); + } + return postsByChannel.get(slug)!.get(id) ?? null; + }; + + // Resolve and verify every citation the reports cite. A channel outside the + // site's pool, or a post the site may not carry, is refused before its + // record is read. + const refusedChannel = new Set<string>(); + for (const report of reports) { + const all = report.citations ?? {}; + // Only what the report cites: a citation it defines but never cites is + // not in its view, and is neither checked nor published. + const used = new Set(reportCitationNumbers(report).keys()); + for (const [cid, c] of Object.entries(all)) { + const ref = `${report.id}#${cid}`; + if (!used.has(cid)) continue; + if (c.kind === "source" || c.kind === "page") { + delete c.verification; + if (c.kind === "source" && c.image) { + const src = path.join(siteReportDir(paths, site.siteId, report.id), c.image); + if (!(await isFile(src))) { + problems.push({ kind: "missing-still", citation: ref, report: report.id, message: `the still ${c.image} does not exist` }); + } + } + continue; + } + if (!pool.has(c.channel)) { + problems.push({ kind: "not-in-site", citation: ref, report: report.id, message: `channel "${c.channel}" is not one of this site's channels` }); + refusedChannel.add(c.channel); + continue; + } + const text = await textProblem(c.channel); + if (text) { + problems.push({ kind: "unreadable", citation: ref, report: report.id, message: text }); + continue; + } + if (c.kind === "post") { + if (!postsVisibleTo(site, await configOf(c.channel), settings)) { + problems.push({ kind: "not-visible", citation: ref, report: report.id, message: `this site may not carry posts of "${c.channel}" (the post visibility rule)` }); + continue; + } + const post = await postOf(c.channel, c.id); + if (!post) { + problems.push({ kind: "missing-post", citation: ref, report: report.id, message: `no post ${c.id} in the posts archive of "${c.channel}"` }); + continue; + } + c.verification = quoteVerification(c.quote, post.text, now); + } else { + const record = await recordOf(c.channel, c.id); + if (!record) { + problems.push({ kind: "missing-record", citation: ref, report: report.id, message: `no record ${c.channel}/${c.id} (no metadata or transcript in its data dir)` }); + continue; + } + if (record.cues.length === 0) { + problems.push({ kind: "no-cues", citation: ref, report: report.id, message: `${c.channel}/${c.id} has no transcript cues to check the quote against` }); + continue; + } + c.verification = quoteVerification(c.quote, cueWindowText(record.cues, c.start, c.end), now); + } + if (quoteDrifted(c.verification)) { + problems.push({ + kind: "quote-drift", + citation: ref, + report: report.id, + message: + `the quote matches ${Math.round((c.verification.quoteScore ?? 0) * 100)}% of what the record says there ` + + `(at least ${Math.round(QUOTE_DRIFT_THRESHOLD * 100)}% is required): quote it verbatim, or fix the span`, + }); + } + } + } + + // The prepared media, per moment. + const media = await readReportMediaIndex(paths, site.siteId); + const cacheDir = reportMediaDir(paths, site.siteId); + const citedIn = buildCitedIn(reports); + const momentInfo = new Map(citedMoments(reports).map((m) => [m.key, m])); + const mediaOf = new Map<string, ReportMediaEntry>(); + for (const key of Object.keys(citedIn)) { + const m = momentInfo.get(key); + if (!m || refusedChannel.has(m.moment.channel)) continue; + const entry = media?.siteId === site.siteId ? media.moments[key] : undefined; + const citedBy = citedIn[key].map((e) => `${e.reportId}#${e.citationId}`); + const missing = (kind: ComposeReportsProblemKind, message: string) => + problems.push({ kind, moment: key, message: `${message} (cited by ${[...new Set(citedBy)].join(", ")})` }); + if (!entry) { + missing("missing-media", "no prepared media — run Prepare evidence media (archilyzer reports prepare)"); + continue; + } + const files = entry.kind === "post" ? [entry.file, ...entry.media.map((f) => f.file)] : [entry.file]; + const absent = []; + for (const f of files) if (!(await isFile(path.join(cacheDir, f)))) absent.push(f); + if (absent.length > 0) { + missing("missing-media", `the prepared media is gone from the cache (${absent.join(", ")}) — prepare again`); + continue; + } + if (entry.kind !== "post" && m.moment.kind === "span") { + // The clip was cut for a span and pad; a report changed since must not + // ship the old cut under the new moment's page. + const sidecar = await readJsonFile(path.join(cacheDir, entry.file.replace(/\.[^.]+$/, ".json"))); + const cut = sidecar.ok ? (sidecar.value as { span?: EvidenceSpan }).span : undefined; + const want = evidenceSpan({ start: m.moment.start, end: m.moment.end, pad: m.pad }); + if (cut && !sameSpan(cut, want)) { + missing("stale-media", `the clip was cut for ${cut.from}–${cut.to} s, the reports now cite ${want.from}–${want.to} s — prepare again`); + continue; + } + } + mediaOf.set(key, entry); + } + + const allowed = opts.allowMissingMedia ? problems.filter((p) => MEDIA_PROBLEMS.has(p.kind)) : []; + const fatal = problems.filter((p) => !allowed.includes(p)); + if (fatal.length > 0) throw new ComposeReportsError(fatal); + for (const line of formatComposeReportsProblems(allowed)) log(`[reports] allowed (--allow-missing-media): ${line}`); + + // ─── The views ─── + + const recordView = async (c: SpanCitation | PostCitation): Promise<RecordView> => { + const config = await configOf(c.channel); + if (c.kind === "post") { + const post = (await postOf(c.channel, c.id))!; + return defined({ + channel: c.channel, + channelTitle: config?.name ?? post.authorName, + id: c.id, + date: post.createdAt.slice(0, 10), + platform: post.platform, + originalUrl: post.url, + corpusUrl: cited ? undefined : corpusLink(`${c.channel}/${c.id}`, { vm: "post" }), + }); + } + const { summary } = (await recordOf(c.channel, c.id))!; + const audioOnly = isAudioOnlyPlatform(config?.platform); + const seconds = Math.max(0, Math.floor(c.start)); + return defined({ + channel: c.channel, + channelTitle: config?.name ?? (summary.channel || undefined), + id: c.id, + title: summary.title, + date: isoDay(summary.uploadDate), + platform: summary.platform, + originalUrl: audioOnly + ? summary.webpageUrl || undefined + : (platformMomentUrl(summary.webpageUrl, summary.platform as Platform, c.start) ?? undefined), + // The viewer keys a record by its PUBLISHED slug, which on a platform + // with two ids is not the data dir's name. + corpusUrl: cited ? undefined : corpusLink(summary.slug ?? `${c.channel}/${summary.id}`, seconds > 0 ? { t: String(seconds) } : {}), + }); + }; + // buildReportPageView's resolver is synchronous: resolve every cited + // record first, by the citation it is resolved for. + const recordViews = new WeakMap<Citation, RecordView>(); + for (const report of reports) { + const all = report.citations ?? {}; + for (const id of reportCitationNumbers(report).keys()) { + const c = all[id]; + if (c.kind === "video" || c.kind === "audio" || c.kind === "post") recordViews.set(c, await recordView(c)); + } + } + const postShot = (c: { channel: string; id: string }): string | undefined => { + const entry = mediaOf.get(`${c.channel}/${c.id}`); + return entry?.kind === "post" ? `/media/${entry.file}` : undefined; + }; + const views: ReportPageView[] = reports.map((report) => + buildReportPageView(report, { + record: (c) => recordViews.get(c)!, + post: (c) => { + const post = postsByChannel.get(c.channel)?.get(c.id); + return post ? { author: postAuthor(post), text: post.text, shot: postShot(c) } : undefined; + }, + downloads: { + json: reportCitationsDownloadPath(report.id, "json"), + csv: reportCitationsDownloadPath(report.id, "csv"), + }, + }), + ); + + const index: ReportIndexView = { + format: REPORT_INDEX_FORMAT, + version: REPORT_VIEWS_VERSION, + reports: views.map(reportIndexEntry), + }; + + const moments: MomentPageView[] = []; + for (const [key, entries] of Object.entries(citedIn)) { + const first = entries[0]; + const report = reports.find((r) => r.id === first.reportId)!; + const c = report.citations![first.citationId] as Citation; + if (c.kind !== "video" && c.kind !== "audio" && c.kind !== "post") continue; + const view = views.find((v) => v.id === report.id)!.citations[first.citationId] as Extract< + CitationView, + { kind: "video" | "audio" | "post" } + >; + const entry = mediaOf.get(key); + const common = { + format: MOMENT_PAGE_FORMAT, + version: REPORT_VIEWS_VERSION, + key, + record: view.record, + quote: c.quote, + ...(c.speaker ? { speaker: c.speaker } : {}), + ...(c.date ? { date: c.date } : {}), + ...(c.verification ? { verification: c.verification } : {}), + citedIn: citedInViews(entries, views), + } as const; + if (c.kind === "post") { + const post = postsByChannel.get(c.channel)!.get(c.id)!; + const postView: MomentPostView = { + author: postAuthor(post), + text: post.text, + ...(entry?.kind === "post" + ? { + shot: `/media/${entry.file}`, + ...(entry.media.length > 0 + ? { + media: entry.media.map((f) => ({ + src: `/media/${f.file}`, + kind: IMAGE_EXTS.has(path.extname(f.file).toLowerCase()) ? ("image" as const) : ("video" as const), + })), + } + : {}), + } + : {}), + }; + moments.push({ ...common, kind: "post", post: postView }); + continue; + } + const m = parseMomentKey(key) as SpanMoment; + const info = momentInfo.get(key)!; + const span = evidenceSpan({ start: m.start, end: m.end, pad: info.pad }); + const { cues } = records.get(`${c.channel}/${c.id}`)!; + const from = m.start - MOMENT_CUE_CONTEXT_SECONDS; + const to = m.end + MOMENT_CUE_CONTEXT_SECONDS; + const lines: CueLineView[] = cues + .filter((q) => q.end > from && q.start < to) + .slice(0, MOMENT_CUE_LINES_MAX) + .map((q) => ({ start: q.start, end: q.end, text: cueLine(q.text), inSpan: q.end > m.start && q.start < m.end })); + const audio = entry?.kind === "audio"; + moments.push({ + ...common, + // A span whose clip had to be cut as sound is heard, not watched. + kind: audio ? "audio" : info.kind === "audio" ? "audio" : "video", + start: m.start, + end: m.end, + ...(entry && entry.kind !== "post" ? { clip: { src: clipPath(m, entry.kind), start: span.from, end: span.to } } : {}), + cues: lines, + }); + } + moments.sort((a, b) => (a.key < b.key ? -1 : a.key > b.key ? 1 : 0)); + + // ─── Writing ─── + + await writeOut(publicDir, REPORTS_INDEX_PATH, json(index)); + for (let i = 0; i < reports.length; i++) { + const report = reports[i]; + const view = views[i]; + await writeOut(publicDir, reportViewPath(report.id), json(view)); + await writeOut(publicDir, reportCitationsDownloadPath(report.id, "json"), json(citationSet(report, view))); + await writeOut(publicDir, reportCitationsDownloadPath(report.id, "csv"), citationsCsv(view)); + for (const c of orderedCitations(view)) { + if (c.kind !== "source" || !c.image) continue; + const rel = (report.citations![c.id] as { image: string }).image; + await copyOut(publicDir, path.join(siteReportDir(paths, site.siteId, report.id), rel), c.image); + } + } + await writeOut( + publicDir, + MOMENTS_INDEX_PATH, + json({ format: MOMENT_INDEX_FORMAT, version: REPORT_VIEWS_VERSION, moments: moments.map((m) => m.key) }), + ); + let copied = 0; + for (const m of moments) { + await writeOut(publicDir, momentViewPath(m.key), json(m)); + const entry = mediaOf.get(m.key); + if (!entry) continue; + if (entry.kind === "post") { + for (const f of [entry.file, ...entry.media.map((x) => x.file)]) { + await copyOut(publicDir, path.join(cacheDir, f), `/media/${f}`); + copied++; + } + } else { + await copyOut(publicDir, path.join(cacheDir, entry.file), clipPath(parseMomentKey(m.key) as SpanMoment, entry.kind)); + copied++; + } + } + log( + `[reports] ${reports.length} report(s), ${moments.length} moment page(s), ${copied} media file(s)` + + `${allowed.length ? `, ${allowed.length} without media` : ""}.`, + ); + return { reports: index.reports, moments: moments.map((m) => m.key), allowed }; +} + +// A clip's published path: the moment's (lib/report/views.ts), `.m4a` for a +// clip cut as sound. +function clipPath(m: SpanMoment, kind: "video" | "audio"): string { + const p = evidenceClipPath(m); + return kind === "audio" ? p.replace(/\.mp4$/, ".m4a") : p; +} + +function postAuthor(post: Post): string { + const handle = post.author.startsWith("@") ? post.author : `@${post.author}`; + return post.authorName ? `${post.authorName} (${handle})` : handle; +} + +const defined = <T extends object>(o: T): T => + Object.fromEntries(Object.entries(o).filter(([, v]) => v !== undefined)) as T; + +// The site-root routes the reports add to a sitemap: the index, each report, +// each moment page. +export function reportRoutes(composed: Pick<ComposedReports, "reports" | "moments">): string[] { + if (composed.reports.length === 0) return []; + return ["/reports/", ...composed.reports.map((r) => r.href), ...composed.moments.map((k) => momentPath(k))]; +}