// THE PUBLISHED CONTRACT, as code. // // A published archive is a pile of static JSON whose *shape* is a promise: // `/corpus.json` names every channel and how to walk manifest -> slugToPage -> // `page-.json`. Phase 0 collapsed the version constants into one // `CONTRACT` object; this module is the next step — the URL SHAPES that go with // them, so a reader and the writer that produced the files agree by // construction rather than by two people editing the same string twice. // // Browser-safe on purpose: every consumer of the contract (the viewer's caches, // the offline cache, the MCP reader, a project's cue resolver) imports from // here, and half of them run in a browser. Nothing in this file touches // `node:*`. // // WHY THIS FILE OWNS `CONTRACT` AND `pageFileName` RATHER THAN RE-EXPORTING // THEM: `lib/corpus.ts` must CALL the URL builders (one definition of the shape // `buildSiteCorpus` emits), and `lib/manifest.ts` reads `CONTRACT.pagePad` at // module scope. Importing `CONTRACT` from `corpus.ts` here would close the loop // corpus -> archive/contract -> manifest -> corpus, and the TDZ read of // `CONTRACT.manifest` at `manifest.ts`'s top level makes that cycle a hard // ReferenceError, not a warning. So the contract module sits at the BOTTOM of // the stack and `corpus.ts` / `manifest.ts` re-export from it: every existing // import site (`from "./corpus"`, `from "./manifest"`) is unchanged, and the // dependency graph is a DAG. import { DUPLICATES_FILENAME } from "../duplicates"; import { TAGS_FILENAME } from "../curatedTags"; // The machine-readable versions and constants of the published contract. Every // value here appears on the wire, so a change is a wire change — see the // per-field notes in lib/corpus.ts for what a bump means to a client. export const CONTRACT = { // /corpus.json's own `spec`. `generator` is deliberately unversioned; see the // note in lib/corpus.ts for why a credit line did not bump the spec. 5: a // site may publish reports (`reports`), and a CITED site (`site.scope: // "cited"`) publishes only them — no channels, no shards. corpusSpec: 5, // /site.json's `contract` (siteDescriptor.ts). siteDescriptor: 1, // The four manifest versions (manifest.ts). Each is the version field of one // served document; they move independently and always have. manifest: 3, transcriptsManifest: 1, subsManifest: 4, subsChannelManifest: 1, // Records per /summaries/page-NNNN.json. summariesPageSize: 1000, // The zero-padding on every page shard's file name. `page-0.json` is a 404 — // this is the whole reason the constant exists in one place. pagePad: 4, // The served trees that follow the manifest -> slugToPage -> page-NNNN walk. // "summaries" is flat (one manifest, pages, no per-channel level). layers: ["transcripts", "subs", "posts", "digests", "summaries"], } as const; export type ContractLayer = (typeof CONTRACT.layers)[number]; // `stats/` follows the same manifest -> page walk but is NOT a contract layer: // corpus.json's shardScheme does not document it, so a client that only knows // the contract must not be told to expect it. It still needs URLs, so the URL // builders take the wider ArchiveTree and CONTRACT.layers stays frozen. export type ArchiveTree = ContractLayer | "stats"; // Every served tree, contract layers plus the undocumented `stats`. This is the // list a cache walker enumerates (the offline cache, the two service workers); // CONTRACT.layers stays the frozen PUBLISHED list. export const ARCHIVE_TREES: readonly ArchiveTree[] = [ ...CONTRACT.layers, "stats", ]; // The trees with no per-channel level: one manifest at the tree root and pages // beside it. Everything else is ///…. const FLAT_TREES: ReadonlySet = new Set(["summaries", "stats"]); export function isFlatTree(tree: ArchiveTree): boolean { return FLAT_TREES.has(tree); } // The per-channel trees: ///…. What a per-channel cache eviction // has to sweep, and the half of ARCHIVE_TREES that takes a slug. export const PER_CHANNEL_TREES: readonly ArchiveTree[] = ARCHIVE_TREES.filter( (t) => !isFlatTree(t), ); // A tree's ROOT manifest, //manifest.json. // // Every tree ships one, and for a per-channel tree it is a DIFFERENT document // from the per-channel manifest: /subs/manifest.json is the site-level index of // which channels ship live chat (SubsManifest), while /subs//manifest.json // is that channel's slugToPage. manifestUrl() below builds the second; this // builds the first, and for a flat tree the two are the same file. export function treeManifestUrl(tree: ArchiveTree, base?: string): string { return archiveUrl(base, `/${tree}/manifest.json`); } // THE page-shard file name, for every layer. `page-0.json` is a 404 on every // published archive, so a copy that lost the padding would 404 silently against // a real site and pass every unit test. lib/manifest.ts re-exports this (and // the per-layer aliases with it), so the six historical copies stay one. export function pageFileName(index: number): string { return `page-${String(index).padStart(CONTRACT.pagePad, "0")}.json`; } // Join an origin base with a root-relative path. When no base is known (a site // built without a configured siteUrl) the path is left root-relative — still // correct for a same-origin fetch, just not portable cross-origin. // // This is the `join` buildSiteCorpus has always used; every URL builder below // goes through it, which is what makes "absolute when siteUrl is set, // root-relative otherwise" one rule instead of a dozen call sites. export function archiveUrl(base: string | undefined, p: string): string { if (!base) return p; return `${base.replace(/\/+$/, "")}${p}`; } // The root-level documents a reader fetches by name. Deliberately only the JSON // an ArchiveReader (or the viewer's offline cache) actually reads — llms.txt, // robots.txt and sitemap.xml are human/crawler surfaces with no reader. // // duplicates.json is the one that is legitimately absent: compose-site only // writes it when there is at least one publishable cluster, and corpus.json // does not declare it, so a 404 here is data, not an error. // // tags.json is absent the same way — the curated vocabulary, with per-site // counts, written only when this site has at least one visible tag with a // non-zero count. Unlike duplicates.json it IS declared in corpus.json (as // `tags`) when present, which is what the spec-4 bump announces: a reader that // does not know about it misses a document it could have fetched. An archive // built before spec 4 simply 404s here, and a tag filter over it must say so // rather than silently return nothing. export const ROOT_FILES = [ "corpus.json", "site.json", "search-aliases.json", DUPLICATES_FILENAME, TAGS_FILENAME, ] as const; export type RootFile = (typeof ROOT_FILES)[number]; export function rootFileUrl(file: RootFile, base?: string): string { return archiveUrl(base, `/${file}`); } export function corpusUrl(base?: string): string { return rootFileUrl("corpus.json", base); } // A tree's manifest URL. `slug` is required for the per-channel trees and // ignored for the flat ones (summaries, stats). export function manifestUrl( tree: ArchiveTree, slug?: string, base?: string, ): string { if (isFlatTree(tree)) return treeManifestUrl(tree, base); if (!slug) throw new Error(`manifestUrl(${tree}) needs a channel slug`); return archiveUrl(base, `/${tree}/${slug}/manifest.json`); } // One page shard of a tree. Same slug rule as manifestUrl. export function pageUrl( tree: ArchiveTree, slug: string | undefined, page: number, base?: string, ): string { const file = pageFileName(page); if (isFlatTree(tree)) return archiveUrl(base, `/${tree}/${file}`); if (!slug) throw new Error(`pageUrl(${tree}) needs a channel slug`); return archiveUrl(base, `/${tree}/${slug}/${file}`); } // Whether a build ships an installable PWA — the "dangerous permissions" // surface: a service worker, a web manifest, and installability. An axis // INDEPENDENT of the shell: // - hub mode always ships the PWA (Archilyzer IS the installable app); // - site mode is a dumb instance by default (federatable JSON only, not // installable) and opts in per site via the `pwa` config flag. // // Was two copies with "keep in sync" comments on each (compose-site.ts and // export/app/lib/mode.ts); S2c deleted both, and this is the one. // // THE `typeof process` GUARD STOPS A CRASH. IT DOES NOT MAKE HUB DETECTION // WORK CLIENT-SIDE, and that distinction is the whole comment. // // `INSTANCE_MODE` is neither `NEXT_PUBLIC_` nor listed in a next.config `env:` // block, so Next does not inline it and a client bundle reads `undefined` here. // (An earlier draft of this comment claimed the opposite; the S1 review // checked.) Hub detection is therefore SERVER-ONLY, guard or no guard, until // somebody ships the value to the client deliberately — which would need an env // var the client can actually see, not a change to this line. // // Today that costs nothing: the export app reaches this only through // export/app/lib/mode.ts ← export/app/layout.tsx, a SERVER component, and the // other caller is compose-site.ts, a build script. What the guard buys is the // failure mode when that stops being true. An unguarded `process.env` in a // module some client component pulls in throws `ReferenceError: process is not // defined` at import time and takes the page down with it; guarded, the same // import yields `false` — no PWA, which is the right answer for every site but // a hub and a recoverable one for a hub. A wrong answer nobody notices beats a // white screen. export function shipsPwa(site: { pwa?: boolean }): boolean { return ( site.pwa === true || (typeof process !== "undefined" && process.env.INSTANCE_MODE === "hub") ); } // ─── The hub's member entry, once ─── // // A hub member travelled under four names: `HubSiteEntry` (compose-hub.ts, the // built-in pool it writes), `HubMemberInput` (buildHubCorpus's input), // `HubCorpusSite` (the entry it emits into the hub corpus.json) and `HubSite` // (what mcp's HubSource reads back out of that same file). // // They collapse to TWO types, not one, and the reason is on the wire: the input // spells a member `siteTitle`/`siteUrl` and the emitted document spells it // `title`/`url` with two derived pointers beside it. corpus.json is frozen, so // the emitted shape cannot be renamed to match the input shape. What does // collapse is each DIRECTION: one input type (was two) and one emitted type, // with the read-back spelling now a Pick of the emitted one so the two can // never drift. // A member site as the hub is TOLD about it: the pool entry compose-hub writes // and the input buildHubCorpus maps. `accent`, `hubUrl` and `contract` are only // carried by the pool entry (the client's coerceBuiltin reads them); a member // without a `siteUrl` cannot be federated and is dropped by both consumers. export type HubMemberInput = { siteId: string; siteTitle: string; siteUrl: string; pwa?: boolean; accent?: string; hubUrl?: string; contract?: number; }; // A member site as the hub PUBLISHES it, in the `sites` array of a hub // corpus.json. Every field is on the wire. export type HubCorpusSite = { siteId: string; title: string; url: string; corpus: string; siteJson: string; pwa: boolean; }; // What a reader needs off that entry to reach a member: its identity and its // origin. A Pick rather than a restatement, so adding a field to the published // entry cannot leave a second copy of its name behind. export type HubSite = Pick;