#!/usr/bin/env tsx // Composes the served export public dir for ONE site (SITE_ID) ahead of // `next build`. Next.js serves a fixed `public/` dir, so per-site bundles are // assembled into it one at a time: // // public/summaries/* <- sites//summaries/ (per-site) // public/subs/manifest.json <- sites//subs/manifest.json (per-site) // public/subs// <- shared/subs// (site's members) // public/transcripts// <- shared/transcripts// (site's members) // public/stats/* <- sites//stats/ (per-site) // public/chart-templates.json <- sites//chart-templates.json (per-site) // public/reports/, m/, media/ <- the site's reports, resolved against the // corpus (publish/composeReports.ts) // // Only the site's member channels are composed, so each deployed bundle holds // only that site's data. Checked-in static assets in public/ are left intact. // // A CITED site (site.json `search: false`) publishes its reports and the // moments they cite, and nothing else: its compose removes every corpus-shaped // entry (CORPUS_PUBLIC_ENTRIES) instead of composing it, then writes the // reports, site.json (no channels) and corpus.json (`site.scope: "cited"`) — // see composeCitedSite. import path from "node:path"; import { cp, link, mkdir, rm, readdir, access, readFile, writeFile, stat, rename } from "node:fs/promises"; import type { Dirent } from "node:fs"; import { getPaths, type Paths } from "../lib/paths"; import { dirSignature } from "../lib/dirSignature"; import { getSite, resolveSocialLinks, resolveHubUrl, type Site } from "../lib/site"; import { getSettings } from "../lib/settings"; import { DUPLICATES_FILENAME, DUPLICATE_OVERRIDES_FILENAME, clusterIsPublishable, filterClusterToChannels, sanitizeDuplicateOverrides, type DuplicateReport, } from "../lib/duplicates"; import type { Manifest, SubsManifest } from "../lib/manifest"; import type { PostsManifest } from "../lib/posts"; import type { DigestsManifest } from "../lib/digests"; import { buildSiteDescriptor, type PublicSiteDescriptor } from "../lib/siteDescriptor"; import { shipsPwa } from "../lib/archive/contract"; import { SITE_CORS_PATHS, SITE_TYPED_PATHS, renderHeadersFile } from "../lib/archive/headers"; import { effectiveSiteAliases } from "../lib/aliasesStore"; import { effectiveSiteTags } from "../lib/curatedTagsStore"; import { TAGS_FILENAME } from "../lib/curatedTags"; import { TAG_COUNTS_FILENAME, publishedTagsFrom, type TagCountsFile, } from "../controller/curatedTagsIndex"; import { buildSiteCorpus, renderSiteLlmsTxt, renderRobotsTxt, renderSitemapXml, } from "../lib/corpus"; import { archiveCacheDir, archiveTranscripts, } from "../controller/archiveTranscripts"; import { archiveLiveChat } from "../controller/archiveLiveChat"; import { openChannelSigner } from "../lib/channelSignature"; import { ARCHIVE_MANIFEST_FILENAME, DEFAULT_ARCHIVE_MAX_BYTES, type ArchiveManifest, type ArchiveManifestEntry, } from "../lib/archiveOptions"; import { readChannelConfig } from "../controller/channels"; import { builtSiteIdIn } from "../lib/builtExport"; import { publishedMemberSlugs } from "../lib/postsVisibility"; import { isCitedSite, isPrivateSite } from "../lib/siteSchema"; import { MANIFEST_VERSION, SUMMARIES_PAGE_SIZE } from "../lib/manifest"; import { composeReports, reportRoutes, type ComposedReports, } from "../publish/composeReports"; import { emptyPostsManifest, tombstoneNoStoreForSite, writePostsTombstones, } from "../publish/tombstones"; import { runIfEntryPoint } from "./_cli"; import { copyPublicFile, ownDir, writePublicFile } from "./_publicFile"; // Emit the public federation contract: /site.json (branding + channels + // freshness) and the CORS _headers file. Emitted for EVERY site regardless of // whether it ships a PWA — a dumb instance is still federatable. // // Every file this compose puts in public/ is written with writePublicFile / // copyPublicFile, and every tree it writes into is ownDir'd first, so nothing // is written THROUGH a link: in a git worktree public/'s entries are links into // the primary checkout (_publicFile.ts). // // Exported, with emitAiFiles, for the cited fixture site's e2e staging // (export/e2e-report/stage.ts), which writes a cited site's contract around // fixture report views exactly as this compose would. // // `noStore` names the paths _headers serves uncached: the tombstones of X // channels this site withheld (publish/tombstones.ts). export async function emitFederationFiles( site: Site, paths: ReturnType, opts: { noStore?: readonly string[] } = {}, ): Promise { await writePublicFile( path.join(paths.exportPublicDir, "_headers"), renderHeadersFile("compose-site.ts", SITE_CORS_PATHS, { noStore: opts.noStore, types: SITE_TYPED_PATHS }), ); // A cited site has no summaries: its descriptor names no channel, and it is // ALWAYS written — the deploy guards read the bundle's identity from it. const cited = isCitedSite(site); const manifestPath = path.join(paths.exportSummariesDir, "manifest.json"); if (!cited && !(await exists(manifestPath))) return; // no composed data → no descriptor const manifest = cited ? emptyManifest(site) : (JSON.parse(await readFile(manifestPath, "utf8")) as Manifest); const descriptor = buildSiteDescriptor(site, manifest, resolveSocialLinks(site), { // A cited site is a handful of pages: nothing to install or cache offline. pwa: !cited && shipsPwa(site), hubUrl: resolveHubUrl(site), }); await writePublicFile( path.join(paths.exportPublicDir, "site.json"), JSON.stringify(descriptor), ); } // A cited site's stand-in for the summaries manifest its descriptor would be // built from: no channels, fresh as of this compose. function emptyManifest(site: Site): Manifest { return { version: MANIFEST_VERSION, totalCount: 0, pageSize: SUMMARIES_PAGE_SIZE, pageCount: 0, generatedAt: new Date().toISOString(), channels: [], groups: site.groups, defaultGroupId: site.defaultGroupId, siteId: site.siteId, }; } // Everything a full compose writes into public/ that is the CORPUS — what a // cited site must not ship, whichever site's compose left it there. Removed // by composeCitedSite before it writes anything; lib/builtExport.ts // citedBuildProblem audits the built out/ against the same promise. A new // corpus-shaped entry in this file's full compose belongs here too. export const CORPUS_PUBLIC_ENTRIES: readonly string[] = [ "summaries", "stats", "transcripts", "subs", "posts", "digests", "archives", "chart-templates.json", "search-aliases.json", TAGS_FILENAME, DUPLICATES_FILENAME, "sw.js", // The hub's, which a site never ships. "hub-sites.json", "hub-summary.json", // Rewritten for the cited site below. "corpus.json", "llms.txt", "robots.txt", "sitemap.xml", ]; // Compose a CITED site: prune the corpus, then its reports and its contract. async function composeCitedSite( site: Site, paths: ReturnType, cachePath: string, opts: { allowMissingMedia: boolean }, ): Promise { for (const entry of CORPUS_PUBLIC_ENTRIES) { await rm(path.join(paths.exportPublicDir, entry), { recursive: true, force: true }); } // …and an earlier full compose's staged oversize archives, which the deploy // would otherwise upload under this site's name. await rm(path.join(path.dirname(paths.exportPublicDir), ".r2-staging", site.siteId), { recursive: true, force: true, }); console.log(`[compose] ${site.siteId} publishes only its reports: the corpus is not composed.`); const composed = await composeReports({ paths, site, allowMissingMedia: opts.allowMissingMedia, log: console.log, }); await emitFederationFiles(site, paths); await emitAiFiles(site, paths, composed); // Nothing of the corpus is in public/ now: the next full compose of this // site composes every stage afresh. await writeComposeCache(cachePath, { transcripts: {}, subs: {}, posts: {}, digests: {} }); console.log( `Composed cited site "${site.siteId}" into ${paths.exportPublicDir} ` + `(${composed.reports.length} report(s), ${composed.moments.length} moment(s)).`, ); } // `--allow-missing-media` (archilyzer compose site / build site), or the // environment variable the build passes it to compose as. export const ALLOW_MISSING_MEDIA_ENV = "REPORTS_ALLOW_MISSING_MEDIA"; // Emit the AI-discovery surface — a small FIXED set of site-root files // (llms.txt, corpus.json, robots.txt, sitemap.xml). These document how to // navigate the already-served paginated shards; they never enumerate per-video // files, so the count is constant regardless of corpus size. Runs after // site.json and the archives are composed (both feed into these files). export async function emitAiFiles( site: Site, paths: ReturnType, composed: ComposedReports, ): Promise { const sitePath = path.join(paths.exportPublicDir, "site.json"); if (!(await exists(sitePath))) return; // no composed data → nothing to describe const descriptor = JSON.parse( await readFile(sitePath, "utf8"), ) as PublicSiteDescriptor; // Bulk archives present? (mirror of export/app/lib/archives.ts hasArchives) let hasArchives = false; const amPath = path.join( paths.exportPublicDir, "archives", ARCHIVE_MANIFEST_FILENAME, ); if (await exists(amPath)) { try { const am = JSON.parse(await readFile(amPath, "utf8")) as ArchiveManifest; hasArchives = Array.isArray(am.entries) && am.entries.some((e) => !e.oversize || e.url); } catch { // A malformed archive manifest just means we omit the archives link. } } // Per-channel post counts come from the site posts manifest just composed // above; absent (no social channels) leaves corpus.json shaped as before. const postCounts: Record = {}; try { const raw = await readFile( path.join(paths.exportPostsDir, "manifest.json"), "utf8", ); const pm = JSON.parse(raw) as PostsManifest; for (const ch of pm.channels ?? []) postCounts[ch.slug] = ch.postCount; } catch { /* no posts manifest for this site */ } // Per-channel digest counts, from the site digests manifest composed above. // Absent (no digested channels) leaves corpus.json without a digest scheme. const digestCounts: Record = {}; try { const raw = await readFile( path.join(paths.exportDigestsDir, "manifest.json"), "utf8", ); const dm = JSON.parse(raw) as DigestsManifest; for (const ch of dm.channels ?? []) digestCounts[ch.slug] = ch.digestCount; } catch { /* no digests manifest for this site */ } // Whether this site published a curated vocabulary. Presence of the file is // the fact — it was written (or removed) just above, by the same rule that // decides whether any tag means anything on this site. const hasTags = await exists(path.join(paths.exportPublicDir, TAGS_FILENAME)); const corpus = buildSiteCorpus(descriptor, { hasArchives, postCounts, digestCounts, hasTags, // A private site says so in its own corpus.json — the bundle's word that // the deploy guard (lib/builtExport.ts builtAudienceProblem) reads. private: isPrivateSite(site), reportCount: composed.reports.length, cited: isCitedSite(site), }); await writePublicFile( path.join(paths.exportPublicDir, "corpus.json"), JSON.stringify(corpus), ); await writePublicFile( path.join(paths.exportPublicDir, "llms.txt"), renderSiteLlmsTxt(corpus, { reports: composed.reports }), ); await writePublicFile( path.join(paths.exportPublicDir, "robots.txt"), renderRobotsTxt({ siteUrl: descriptor.siteUrl }), ); // A sitemap of relative paths is useless, so only emit one with an absolute // siteUrl; otherwise clear any stale copy from a previous build. const sitemapPath = path.join(paths.exportPublicDir, "sitemap.xml"); if (descriptor.siteUrl) { // A cited site's home IS its report index; a full site's reports follow // its own routes. const routes = isCitedSite(site) ? ["/"] : ["/", "/changelog"]; if (hasArchives) routes.push("/downloads"); if (await exists(path.join(paths.exportPublicDir, DUPLICATES_FILENAME))) { routes.push("/duplicates"); } routes.push(...reportRoutes(composed).filter((r) => !routes.includes(r))); await writePublicFile( sitemapPath, renderSitemapXml({ siteUrl: descriptor.siteUrl, routes }), ); } else { await rm(sitemapPath, { force: true }); } } // Compose the service worker into the served public dir ONLY when this instance // ships a PWA. The SW source lives outside public/ (export/service-worker/) so // a dumb instance emits no /sw.js at all — not just an unregistered one. Mirrors // how site.json/_headers are generated rather than checked in. async function composeServiceWorker( site: Site, paths: ReturnType, ): Promise { const dest = path.join(paths.exportPublicDir, "sw.js"); await rm(dest, { force: true }); if (!shipsPwa(site)) return; const src = path.join(paths.monorepoRoot, "export", "service-worker", "site-sw.js"); if (await exists(src)) await cp(src, dest); } // Whether this build generates the downloadable archive zips. Opt-OUT at three // levels, all defaulting on: the global setting, the per-site `archives` flag, // and a per-build BUILD_ARCHIVES=0 env (set by the editor's "Skip archive zips" // control). Effective = AND of all three. function archivesEnabled(site: Site): boolean { const globalOn = getSettings().buildArchives !== false; const siteOn = site.archives !== false; const buildOn = process.env.BUILD_ARCHIVES !== "0"; return globalOn && siteOn && buildOn; } // The served-file size cap for this site: per-site override, else MAX_ARCHIVE_BYTES // env, else the Cloudflare-safe default. 0 = no cap. function archiveMaxBytes(site: Site): number { if (typeof site.archiveMaxBytes === "number" && site.archiveMaxBytes >= 0) { return site.archiveMaxBytes; } const env = Number(process.env.MAX_ARCHIVE_BYTES); if (Number.isFinite(env) && env >= 0) return env; return DEFAULT_ARCHIVE_MAX_BYTES; } function humanBytes(n: number): string { if (n < 1024) return `${n} B`; const units = ["KB", "MB", "GB"]; let v = n / 1024; let i = 0; while (v >= 1024 && i < units.length - 1) { v /= 1024; i++; } return `${v.toFixed(v >= 10 || i === 0 ? 0 : 1)} ${units[i]}`; } // Channel display names keyed by slug, read from the already-composed per-site // subs manifest. Slugs are stable; display names are the human-facing titles. async function channelTitles( paths: ReturnType, ): Promise> { const titles = new Map(); const manifestPath = path.join(paths.exportSubsDir, "manifest.json"); if (!(await exists(manifestPath))) return titles; try { const sm = JSON.parse(await readFile(manifestPath, "utf8")) as SubsManifest; for (const c of sm.channels ?? []) { if (c.slug && c.name) titles.set(c.slug, c.name); } } catch { // A malformed subs manifest just means we fall back to slugs. } return titles; } // Overflow object-storage (Cloudflare R2) target for archives that exceed the // Pages per-file cap: where to stage the file for the deploy-time upload, and the // public URL base + key prefix the manifest points visitors at. Absent = no // overflow storage configured, so oversize files are dropped instead. type OverflowTarget = { stagingDir: string; // export/.r2-staging//archives publicBaseUrl: string; // e.g. https://archives.example.com (no trailing slash) keyPrefix: string; // /archives }; // Stat a produced archive and enforce the size cap. An oversize file can't ship // as a Pages asset, so either: (a) overflow storage is configured — move it to // the staging dir for the deploy-time R2 upload and record its public `url`; or // (b) it isn't — drop it and mark the entry `oversize`, which the Downloads page // renders as unavailable. Under-cap files are served locally from /archives/. async function describeArchive( kind: ArchiveManifestEntry["kind"], scope: string, outDir: string, filename: string, videoCount: number, maxBytes: number, overflow: OverflowTarget | null, channelTitle?: string, ): Promise { const filePath = path.join(outDir, filename); let bytes = 0; try { bytes = (await stat(filePath)).size; } catch { bytes = 0; } const entry: ArchiveManifestEntry = { kind, scope, filename, bytes, videoCount }; if (channelTitle) entry.channelTitle = channelTitle; const oversize = maxBytes > 0 && bytes > maxBytes; if (oversize && overflow) { // Move it out of the served tree and into the upload staging area; the deploy // step pushes it to R2 and this URL resolves. await mkdir(overflow.stagingDir, { recursive: true }); await rename(filePath, path.join(overflow.stagingDir, filename)); entry.url = `${overflow.publicBaseUrl}/${overflow.keyPrefix}/${filename}`; console.log( `[archives] ${filename} is ${humanBytes(bytes)} (over ${humanBytes(maxBytes)} cap) — staged for R2 upload`, ); } else if (oversize) { await rm(filePath, { force: true }); entry.oversize = true; console.warn( `[archives] ${filename} is ${humanBytes(bytes)} (over ${humanBytes(maxBytes)} cap) — not served (no overflow storage configured)`, ); } return entry; } // Generate this site's bulk-download archive zips into public/archives and write // a manifest.json the Downloads page reads. Scoped to the site's member channels, // mirroring how every other per-site asset is composed. Runs on every build so // the zips always match the current corpus. async function composeArchives( site: Site, memberSlugs: string[], paths: ReturnType, ): Promise { const outDir = path.join(paths.exportPublicDir, "archives"); // The deploy-time R2 upload reads staged oversize files from here. Kept OUTSIDE // public/ so it's never deployed as Pages assets; namespaced by site so // concurrent per-site builds don't clobber each other. const stagingDir = path.join( path.dirname(paths.exportPublicDir), ".r2-staging", site.siteId, "archives", ); // Always clear both first so a disabled site (or one that lost the feature) // never serves — or uploads — a stale bundle from a previous build. await rm(outDir, { recursive: true, force: true }); await rm(stagingDir, { recursive: true, force: true }); if (!archivesEnabled(site)) { console.log("[archives] disabled for this build — skipping."); return; } await mkdir(outDir, { recursive: true }); // Overflow object storage (R2) for oversize archives, if configured globally. // Blank/absent config → oversize archives are dropped (today's behavior). const storage = getSettings().archiveStorage; const overflow: OverflowTarget | null = storage && storage.bucket && storage.publicBaseUrl ? { stagingDir, publicBaseUrl: storage.publicBaseUrl.replace(/\/+$/, ""), keyPrefix: `${site.siteId}/archives`, } : null; const maxBytes = archiveMaxBytes(site); const log = (m: string) => { if (m) console.log(`[archives] ${m}`); }; const build = { format: "zip" as const }; // Build (or reuse) archives in the persistent SHARED cache — not straight into // public/ — so a channel unchanged since the last build (or already built for // another site this run) is not re-zipped. One signer view is shared across the // transcripts + live-chat passes. compose then copies this site's member subset // out of the cache into public/archives below. // // ARCHIVES_READONLY (set by the docker per-site build container): the cache was // already warmed on the host by `build:archives`, and here it's a read-only // mount — so never generate or open the index, just materialize what's cached. // This is what keeps parallel per-site containers from racing writes to the one // shared cache. const cacheDir = archiveCacheDir(paths); const readOnly = process.env.ARCHIVES_READONLY === "1"; if (readOnly) log("read-only cache mode — materializing pre-built archives"); const signer = readOnly ? undefined : openChannelSigner(paths); let transcripts: Awaited>; let liveChat: Awaited>; try { const common = { paths, channelSlugs: memberSlugs, outDir: cacheDir, build, onLog: log, readOnly, signer, }; // Per-channel archives only — the combined "whole-site" bundles were dropped // (they duplicated the per-channel content and were always the first to blow // past the size cap). transcripts = await archiveTranscripts(common); liveChat = await archiveLiveChat(common); } finally { if (signer) await signer.close(); } // Materialize each cached archive into this site's served tree. Hard-link when // possible (same filesystem, read-only served asset) so it's near-free; fall // back to a byte copy across devices. describeArchive() below then stats the // public copy and applies the size cap / R2 overflow exactly as before, // leaving the shared cache original untouched. const materialize = async (srcPath: string): Promise => { const dest = path.join(outDir, path.basename(srcPath)); try { await link(srcPath, dest); } catch { await cp(srcPath, dest); } }; for (const a of transcripts.archives) await materialize(a.archivePath); for (const a of liveChat.archives) await materialize(a.archivePath); const titles = await channelTitles(paths); const bySlug = (a: { slug: string }, b: { slug: string }) => a.slug.localeCompare(b.slug); const entries: ArchiveManifestEntry[] = []; // Transcripts per channel, ordered by slug for a reproducible manifest. for (const a of [...transcripts.archives].sort(bySlug)) { entries.push( await describeArchive( "transcripts", a.slug, outDir, path.basename(a.archivePath), a.transcriptCount, maxBytes, overflow, titles.get(a.slug), ), ); } // Live chat per channel. Channels with no chat produce nothing to list. for (const a of [...liveChat.archives].sort(bySlug)) { entries.push( await describeArchive( "live-chat", a.slug, outDir, path.basename(a.archivePath), a.liveChatCount, maxBytes, overflow, titles.get(a.slug), ), ); } const manifest: ArchiveManifest = { version: 1, thresholdBytes: maxBytes, entries, }; await writeFile( path.join(outDir, ARCHIVE_MANIFEST_FILENAME), JSON.stringify(manifest, null, 2) + "\n", ); const available = entries.filter((e) => !e.oversize).length; const staged = entries.filter((e) => e.url).length; console.log( `[archives] wrote ${available}/${entries.length} archive(s) (${staged} staged for R2) to public/archives.`, ); } async function exists(p: string): Promise { try { await access(p); return true; } catch { return false; } } async function replaceDir(src: string, dest: string): Promise { await rm(dest, { recursive: true, force: true }); if (await exists(src)) { await mkdir(path.dirname(dest), { recursive: true }); await cp(src, dest, { recursive: true }); } } // Rewrite a served manifest whose `channels` name a slug outside `members`, // keeping the rest of it. A missing or unreadable manifest is left alone. // Answers whether it rewrote the file. async function narrowManifestChannels(file: string, members: Set): Promise { let m: { channels?: { slug?: string }[] }; try { m = JSON.parse(await readFile(file, "utf8")); } catch { return false; } const channels = m.channels ?? []; const kept = channels.filter((c) => typeof c.slug !== "string" || members.has(c.slug)); if (kept.length === channels.length) return false; await writePublicFile(file, JSON.stringify({ ...m, channels: kept })); return true; } // The channel slugs a site posts manifest lists, or null when there is no // readable manifest. async function readPostsManifestSlugs(file: string): Promise | null> { try { const pm = JSON.parse(await readFile(file, "utf8")) as PostsManifest; return new Set((pm.channels ?? []).map((c) => c.slug)); } catch { return null; } } // --- Incremental compose cache ------------------------------------------------ // Per-site record of what we last materialized into public/, keyed by a cheap // content signature of each source. When the signature is unchanged and the // destination is still present, the copy/serialize is skipped. Kept outside // public/ (a sibling of .export-index / .r2-staging) so it's never deployed. type ComposeCache = { transcripts: Record; subs: Record; // Optional for backwards compat: a cache written before the posts corpus // existed simply has no entry, so every social channel composes once. posts?: Record; // Same, for the AI-digest corpus: an existing cache composes digests once // rather than erroring on a missing key. digests?: Record; summaries?: string; stats?: string; duplicates?: string; }; function composeCachePath( paths: ReturnType, siteId: string, ): string { return path.join( path.dirname(paths.exportPublicDir), ".compose-cache", `${siteId}.json`, ); } async function readComposeCache(p: string): Promise { try { const parsed = JSON.parse(await readFile(p, "utf8")); return { transcripts: parsed?.transcripts ?? {}, subs: parsed?.subs ?? {}, posts: parsed?.posts ?? {}, digests: parsed?.digests ?? {}, summaries: parsed?.summaries, stats: parsed?.stats, duplicates: parsed?.duplicates, }; } catch { return { transcripts: {}, subs: {}, posts: {} }; } } async function writeComposeCache(p: string, cache: ComposeCache): Promise { await mkdir(path.dirname(p), { recursive: true }); await writeFile(p, JSON.stringify(cache)); } // Materialize the member subset of a shared per-channel tree into public/, // skipping channels whose source is unchanged since the last compose. Replaces // the previous "rm -rf the whole tree then cp every member" with an in-place // reconcile: only changed channels are re-copied, and channels no longer members // are pruned. Returns the new per-slug signature map for the compose cache. // // A MANIFEST-ONLY TREE IS A CHANNEL WITH NOTHING IN IT, NOT A MISSING CHANNEL. // The signature ignores manifest.json (it churns), so a member whose source is // JUST the manifest — a social channel's transcripts tree, which buildIndex // writes with `pageCount: 0` because such a channel never enters the video scan // — used to sign as "" and be deleted, while corpus.json still advertised its // transcripts manifest to every reader: a 404 (jeralyzer's thequartering-X). // The manifest is exactly what "0 transcripts" means, so that tree is copied // under a constant signature and the contract stays unconditional. export const MANIFEST_ONLY_SIGNATURE = "manifest-only"; export async function reconcileChannelTree( kind: string, srcRoot: string, destRoot: string, memberSlugs: string[], prev: Record, log: (m: string) => void, ): Promise> { // A linked tree (a worktree's public/ → the primary's) becomes this // checkout's own before anything is copied into or pruned from it. await ownDir(destRoot); const next: Record = {}; const memberSet = new Set(memberSlugs); let copied = 0; let skipped = 0; for (const slug of memberSlugs) { const src = path.join(srcRoot, slug); const dest = path.join(destRoot, slug); // Exclude the per-channel manifest.json — its `generatedAt` churns every // mutation build; the page files capture real content changes. let sig = await dirSignature(src, "manifest.json"); if (sig === "" && (await exists(path.join(src, "manifest.json")))) { sig = MANIFEST_ONLY_SIGNATURE; } if (sig === "") { // No source for this member — ensure no stale dest survives. await rm(dest, { recursive: true, force: true }); continue; } if (prev[slug] === sig && (await exists(dest))) { next[slug] = sig; skipped++; continue; } await rm(dest, { recursive: true, force: true }); await cp(src, dest, { recursive: true }); next[slug] = sig; copied++; } // Prune channel dirs that are no longer members (leaves non-dir siblings such // as the per-site subs manifest.json untouched). let entries: Dirent[]; try { entries = await readdir(destRoot, { withFileTypes: true }); } catch { entries = []; } for (const e of entries) { if (e.isDirectory() && !memberSet.has(e.name)) { await rm(path.join(destRoot, e.name), { recursive: true, force: true }); } } log(`[compose] ${kind}: ${copied} copied, ${skipped} unchanged.`); return next; } // `archilyzer compose site ` passes the id; the bare bin (export's // compose:site script) reads SITE_ID, as it always has. export async function main( opts: { siteId?: string; paths?: Paths; allowMissingMedia?: boolean } = {}, ): Promise { const siteId = opts.siteId?.trim() || process.env.SITE_ID; if (!siteId) { throw new Error("compose-site: a site id is required (argument or SITE_ID env var)"); } const paths = opts.paths ?? getPaths(); const site = getSite(siteId, paths); // The members this site may publish (lib/postsVisibility.ts): every member, // less an X channel while `social.x.visibility` is "private" and the site is // public. The index build's per-site manifests are narrowed by the same rule; // narrowing the trees here prunes an X channel a public site shipped before. const configs = new Map( await Promise.all( site.channels.map( async (c) => [c.slug, await readChannelConfig(paths, c.slug).catch(() => null)] as const, ), ), ); const memberSlugs = publishedMemberSlugs( site, (slug) => configs.get(slug), getSettings(), ); const withheld = site.channels.length - memberSlugs.length; if (withheld > 0) { console.log( `[compose] ${withheld} X channel(s) left out: X posts are private (social.x.visibility) and this site is public.`, ); } // Incremental compose: skip stages whose source is unchanged since last build. // // ONLY OVER THIS SITE'S OWN LAST COMPOSE. The cache is per site but public/ // is one directory every site composes into in turn (the basic build), so // after another site's compose a skipped stage would ship THAT site's files — // its summaries, its whole channel list — under this site's name: composing // a private site and then a public one shipped the private site's summaries // as the public site's. public/site.json names the site composed into it // last (it is written below, every compose); another name, or none, and the // site's own stages (summaries, stats, duplicates) are composed afresh. // // The per-channel tree signatures stay trusted: those trees are copies of // the SHARED trees, the same bytes whichever site copied them, and a // channel another site pruned is re-copied because its directory is gone. const cachePath = composeCachePath(paths, siteId); const lastComposed = builtSiteIdIn(paths.exportPublicDir); const cached = await readComposeCache(cachePath); const cache: ComposeCache = lastComposed === siteId ? cached : { ...cached, summaries: undefined, stats: undefined, duplicates: undefined }; if (lastComposed !== siteId && lastComposed !== null) { console.log( `[compose] public/ was last composed for "${lastComposed}": composing ${siteId}'s summaries, stats and duplicates afresh.`, ); } // Until this compose writes its own, public/ names no site: a compose cut // short part-way leaves the next one nothing to trust. await rm(path.join(paths.exportPublicDir, "site.json"), { force: true }); const allowMissingMedia = opts.allowMissingMedia === true || process.env[ALLOW_MISSING_MEDIA_ENV] === "1"; if (isCitedSite(site)) { await composeCitedSite(site, paths, cachePath, { allowMissingMedia }); return; } // --- per-site aggregates (whole-dir swaps), gated on the source signature --- const summariesSrc = path.join(paths.exportSitesIndexDir, siteId, "summaries"); const summariesSig = await dirSignature(summariesSrc); if ( cache.summaries !== summariesSig || !(await exists(paths.exportSummariesDir)) ) { await replaceDir(summariesSrc, paths.exportSummariesDir); cache.summaries = summariesSig; } else { console.log("[compose] summaries: unchanged."); } // The served summaries manifest lists only the members this compose // publishes (lib/postsVisibility.ts), even over an index built before // `social.x.visibility` flipped: site.json and corpus.json are built from it. // A narrowed copy is no longer the source's copy: the next compose copies // the summaries again rather than trusting it. if ( await narrowManifestChannels( path.join(paths.exportSummariesDir, "manifest.json"), new Set(memberSlugs), ) ) { cache.summaries = undefined; } const statsSrc = path.join(paths.exportSitesIndexDir, siteId, "stats"); const statsSig = await dirSignature(statsSrc); if (cache.stats !== statsSig || !(await exists(paths.exportStatsDir))) { await replaceDir(statsSrc, paths.exportStatsDir); cache.stats = statsSig; } else { console.log("[compose] stats: unchanged."); } // --- shared per-channel trees, filtered to this site's members --- // Reconcile the member subset in place (only changed channels are re-copied). cache.transcripts = await reconcileChannelTree( "transcripts", paths.exportSharedTranscriptsDir, paths.exportTranscriptsDir, memberSlugs, cache.transcripts, console.log, ); cache.subs = await reconcileChannelTree( "subs", paths.exportSharedSubsDir, paths.exportSubsDir, memberSlugs, cache.subs, console.log, ); // The social-post corpus: same shared-tree shape, same incremental reconcile. // Only social member channels have a source dir; reconcileChannelTree treats a // missing one as "nothing to copy", so passing every member slug is correct. // // NEVER A TREE THE SITE'S POSTS MANIFEST DOES NOT LIST. The index build wrote // that manifest by the same visibility rule as memberSlugs above, so the two // agree — unless a channel's config could not be read here (it then reads as // visible): the manifest is the index build's word, and a tree it withheld is // not shipped on a failed read. A site with no manifest yet keeps the old rule. const postsListed = await readPostsManifestSlugs( path.join(paths.exportSitesIndexDir, siteId, "posts", "manifest.json"), ); cache.posts = await reconcileChannelTree( "posts", paths.exportSharedPostsDir, paths.exportPostsDir, postsListed ? memberSlugs.filter((slug) => postsListed.has(slug)) : memberSlugs, cache.posts ?? {}, console.log, ); // The AI-digest corpus: same shared-tree shape again. Sparse — only channels // with at least one digest have a source dir, and reconcileChannelTree treats // a missing one as "nothing to copy", so passing every member slug is right. cache.digests = await reconcileChannelTree( "digests", paths.exportSharedDigestsDir, paths.exportDigestsDir, memberSlugs, cache.digests ?? {}, console.log, ); // Subs also carries a per-site manifest.json (tiny — copied every build). const subsManifestSrc = path.join( paths.exportSitesIndexDir, siteId, "subs", "manifest.json", ); if (await exists(subsManifestSrc)) { await ownDir(paths.exportSubsDir); await copyPublicFile(subsManifestSrc, path.join(paths.exportSubsDir, "manifest.json")); } // Same for the per-site posts manifest (which channels carry posts). const postsManifestSrc = path.join( paths.exportSitesIndexDir, siteId, "posts", "manifest.json", ); if (await exists(postsManifestSrc)) { await ownDir(paths.exportPostsDir); // Narrowed to the members this compose publishes, so a compose run over an // index built before `social.x.visibility` flipped (`archilyzer compose // site` alone, `build site --nodata`) does not list a withheld X channel's // name and count beside the pruned tree. const pm = JSON.parse(await readFile(postsManifestSrc, "utf8")) as PostsManifest; const members = new Set(memberSlugs); const channels = (pm.channels ?? []).filter((c) => members.has(c.slug)); await writePublicFile( path.join(paths.exportPostsDir, "manifest.json"), JSON.stringify({ ...pm, channels, totalCount: channels.reduce((n, c) => n + (c.postCount ?? 0), 0), }), ); } // --- tombstones for the X channels this site withheld (release 18) --- // A withheld channel's posts were served from this site's paths before (or // may still be cached at the edge from a build when X was public): every // such path is REPLACED with an empty object of the same shape and served // no-store, never left out (publish/tombstones.ts). The site posts manifest // above already lists no withheld channel; a site with no posts manifest from // the index (its only posts were X posts) ships an empty one — replacing // whatever an earlier compose left at that path (another site's, listing its // channels). const memberSet = new Set(memberSlugs); const tombstones = await writePostsTombstones({ postsDir: paths.exportPostsDir, sharedPostsDir: paths.exportSharedPostsDir, slugs: site.channels.map((c) => c.slug).filter((slug) => !memberSet.has(slug)), }); if (tombstones.length > 0) { if (!(await exists(postsManifestSrc))) { await writePublicFile( path.join(paths.exportPostsDir, "manifest.json"), JSON.stringify(emptyPostsManifest(new Date().toISOString(), siteId)), ); } console.log( `[compose] posts: ${tombstones.length} withheld X channel(s) shipped as tombstones ` + `(${tombstones.reduce((n, t) => n + t.pages, 0)} empty page(s)), served no-store.`, ); } // Same for the per-site digests manifest (which channels carry digests). const digestsManifestSrc = path.join( paths.exportSitesIndexDir, siteId, "digests", "manifest.json", ); if (await exists(digestsManifestSrc)) { await ownDir(paths.exportDigestsDir); await copyPublicFile( digestsManifestSrc, path.join(paths.exportDigestsDir, "manifest.json"), ); } // --- charts dashboard --- const templatesSrc = path.join( paths.exportSitesIndexDir, siteId, "chart-templates.json", ); const templatesDest = path.join(paths.exportPublicDir, "chart-templates.json"); await rm(templatesDest, { force: true }); if (await exists(templatesSrc)) { await cp(templatesSrc, templatesDest); } // --- search aliases (global merged with per-site overrides) --- // Read from source at compose time (like duplicates.json) rather than a // staging step. Always emitted — a fresh install ships the seeded defaults so // the viewer's suggestion chip works out of the box. const aliases = effectiveSiteAliases(paths, siteId); await writePublicFile( path.join(paths.exportPublicDir, "search-aliases.json"), JSON.stringify({ aliases }), ); // --- curated tags (corpus vocabulary + this site's overlay + its counts) --- // The vocabulary is read from source at compose time, like the aliases; the // counts come from the per-site tag-counts.json build:index wrote while it // was already streaming this site's summaries. // // What ships is only what means something HERE: hidden tags are dropped, and // so is any tag with no video on this site — a corpus-wide tag that matched // nothing here gets no chip here. Nothing left → no file at all, and the 404 // is the empty state (the same contract duplicates.json has). The stale copy // is removed in that case, so a site that loses its last tagged video stops // advertising tags on the very next build. const tagDefs = effectiveSiteTags(paths, siteId); let tagCounts: TagCountsFile | null = null; try { tagCounts = JSON.parse( await readFile( path.join(paths.exportSitesIndexDir, siteId, TAG_COUNTS_FILENAME), "utf8", ), ) as TagCountsFile; } catch { /* no counts staged for this site yet — treated as "nothing tagged" */ } const publishedTags = publishedTagsFrom(tagDefs, tagCounts); const tagsDest = path.join(paths.exportPublicDir, TAGS_FILENAME); if (publishedTags.tags.length > 0) { await writePublicFile(tagsDest, JSON.stringify(publishedTags)); } else { await rm(tagsDest, { force: true }); } // --- duplicate-shorts report (global → site-filtered) --- // The detector writes one global duplicates.json over the whole channel pool; // each site only serves its own channels, so filter clusters to the site's // members (dropping out-of-site refs and clusters that fall below 2 members) // before serving. The file is written only when the feature is enabled for this // site (per-site `duplicates` opt-out) AND there's at least one in-scope // cluster — so its mere presence is what hasDuplicates() keys off to show the // nav link. No file → the page shows its empty state and the link self-hides. // // UNCONFIRMED SUSPECTS ARE NOT SHIPPED. A `needsReview` cluster is evidence // that two videos share a title and a runtime — nothing compared their // content — so it is an internal review queue, not something to assert to a // viewer. It reaches the public site only once a human records `confirmed` in // duplicates.overrides.json, which is why that file (never read at compose // time before) is read here. const dupSrc = path.join(paths.transcriptsDir, DUPLICATES_FILENAME); const dupDest = path.join(paths.exportPublicDir, DUPLICATES_FILENAME); const dupOverridesSrc = path.join( paths.transcriptsDir, DUPLICATE_OVERRIDES_FILENAME, ); // Gate the re-filter/re-serialize on the source's mtime+size, the OVERRIDES' // mtime+size (a confirmation changes what ships without touching the report), // the member set, and the feature flag. The cache value is prefixed // written|/empty| so a skip can self-heal if public/ was wiped out of band // (presence must match). let dupStat: { mtimeMs: number; size: number } | null = null; try { const s = await stat(dupSrc); dupStat = { mtimeMs: s.mtimeMs, size: s.size }; } catch { dupStat = null; } let dupOvStat: { mtimeMs: number; size: number } | null = null; try { const s = await stat(dupOverridesSrc); dupOvStat = { mtimeMs: s.mtimeMs, size: s.size }; } catch { dupOvStat = null; } const dupEnabled = site.duplicates !== false && dupStat !== null; const dupKey = `${dupEnabled}|${dupStat?.mtimeMs ?? ""}|${dupStat?.size ?? ""}` + `|${dupOvStat?.mtimeMs ?? ""}|${dupOvStat?.size ?? ""}` + `|${[...memberSlugs].sort().join(",")}`; const dupDestPresent = await exists(dupDest); const dupCachedWritten = cache.duplicates?.startsWith("written|") ?? false; const dupCacheHit = cache.duplicates?.endsWith(`|${dupKey}`) && dupCachedWritten === dupDestPresent; if (dupCacheHit) { console.log("[compose] duplicates: unchanged."); } else { await rm(dupDest, { force: true }); let wrote = false; if (dupEnabled) { const report = JSON.parse( await readFile(dupSrc, "utf8"), ) as DuplicateReport; const memberSet = new Set(memberSlugs); // Never throws: an unreadable or malformed overrides file reads as "no // decisions recorded", which fails CLOSED — every suspect stays internal. let dupOverrides = sanitizeDuplicateOverrides(null); try { dupOverrides = sanitizeDuplicateOverrides( JSON.parse(await readFile(dupOverridesSrc, "utf8")), ); } catch { // no decisions recorded } const clusters = report.clusters .filter((c) => clusterIsPublishable(c, dupOverrides)) .map((c) => filterClusterToChannels(c, memberSet)) .filter((c): c is NonNullable => c !== null); if (clusters.length > 0) { const filtered: DuplicateReport = { ...report, totals: { ...report.totals, clusters: clusters.length, videosInClusters: clusters.reduce( (n, c) => n + c.videoRefs.length, 0, ), }, clusters, }; await writePublicFile(dupDest, JSON.stringify(filtered)); wrote = true; } } cache.duplicates = `${wrote ? "written" : "empty"}|${dupKey}`; } // --- reports: the site's published reports and the moments they cite --- // (a site with none: whatever an earlier compose left is removed). Before // site.json: a stage that fails leaves public/ naming no site. const composed = await composeReports({ paths, site, allowMissingMedia, log: console.log }); // --- federation contract: /site.json descriptor + CORS _headers --- await emitFederationFiles(site, paths, { noStore: tombstoneNoStoreForSite(tombstones) }); // A site's bundle is not a hub's. export/public is shared with the hub build, // whose compose writes hub-sites.json; left in place it ships in this site's // out/ and makes the bundle ambiguous to builtHubProblem (and to a site's // own registry, which treats the file as a hub's trusted pool). await rm(path.join(paths.exportPublicDir, "hub-sites.json"), { force: true }); // …and neither is the hub's copy of the official instances' numbers. await rm(path.join(paths.exportPublicDir, "hub-summary.json"), { force: true }); // --- service worker (only when this instance ships a PWA) --- await composeServiceWorker(site, paths); // --- bulk-download archive zips (on by default; see archivesEnabled) --- await composeArchives(site, memberSlugs, paths); // --- AI discovery: llms.txt / corpus.json / robots.txt / sitemap.xml --- // (after site.json + archives + reports — all feed into these fixed-count files) await emitAiFiles(site, paths, composed); // Persist the incremental-compose signatures for the next build. await writeComposeCache(cachePath, cache); const channelDirs = (await readdir(paths.exportTranscriptsDir).catch( () => [] as string[], )).length; console.log( `Composed site "${siteId}" into ${paths.exportPublicDir} (${channelDirs} channels).`, ); } runIfEntryPoint(import.meta.url, () => main());