import type { PublicSiteDescriptor } from "./siteDescriptor"; import { AI_DOC_URL, PROJECT_GENERATOR } from "./project"; import { TAGS_FILENAME } from "./curatedTags"; import { REPORTS_INDEX_PATH, reportFullTitle } from "./report/views"; import { CONTRACT, archiveUrl, corpusUrl, manifestUrl, rootFileUrl, type ContractLayer, type HubCorpusSite, type HubMemberInput, } from "./archive/contract"; // The machine-readable corpus index emitted at `/corpus.json` on every export // bundle (and an aggregate variant on a hub). It does NOT contain transcripts — // the corpus is far too large to enumerate per-video without blowing the // Cloudflare Pages file-count limit. Instead it *documents how to navigate the // existing paginated JSON shards*, turning the already-served // `subs/`, `transcripts/`, and `summaries/` trees into a self-describing API for // LLM tools (Claude Code `WebFetch`, the repo MCP server, the in-browser chat). // // Pure module: builders take already-loaded data and return plain objects / // strings. All file I/O lives in compose-site.ts / compose-hub.ts. // v2: the corpus now also describes the parallel social-post layer // (postScheme + per-channel posts manifest pointers). // v3: …and the DERIVED layer — AI digests (chapters + topic tags) served under // the same shard scheme (digestScheme + per-channel digests manifest pointers). // v4: curated tags — an operator-authored, cross-channel vocabulary published // at /tags.json, with the per-record key `curatedTags` on summaries and // transcript shards. A new FETCHABLE DOCUMENT, which is exactly what a bump // announces: a reader that skips it cannot narrow a corpus to "every stream // where X collabs". The per-record key itself is additive and bumps no manifest // version — an older reader ignores an unknown field, as it always has. // // v5: reports — a site may publish cited reports, announced as `reports` // ({ index: "/reports/index.json" }, the report index the export's Reports // pages read); and a CITED site (site.json `search: false`) publishes ONLY // them, saying so as `site.scope: "cited"` with no channels and zero totals. // An older reader sees an empty corpus there, which is the safe reading: there // is no shard to fetch. // // NOT a v4: the `generator` field added below is deliberately unversioned. Every // prior bump announced a new FETCHABLE LAYER — a reader that ignored it would // miss data it could otherwise have retrieved. `generator` is an informational // string that points at the software, not at any content; no reader's behaviour // changes by not knowing about it, and the only in-repo consumer // (the archive reader) casts the parsed corpus loosely. Bumping the spec would // force every client to re-evaluate compatibility for a credit line. // // CONTRACT ITSELF NOW LIVES IN `lib/archive/contract.ts`, beside the URL // builders this file calls, and is re-exported here so every `from "./corpus"` // import site is unchanged. It had to move rather than be imported: `manifest.ts` // reads `CONTRACT.pagePad` at module scope, so corpus -> archive/contract -> // manifest -> corpus would be a TDZ ReferenceError, not a lint warning. The // contract module imports nothing from either, which makes the graph a DAG. export { CONTRACT }; export type { ContractLayer }; // The hub member types moved with it (they are contract shapes, not corpus // builders) and are re-exported for the same reason. export type { HubCorpusSite, HubMemberInput }; // Every version constant below is a field of CONTRACT, re-exported under the // name its callers already use. Nothing on the wire changes by moving one. export const CORPUS_SPEC_VERSION = CONTRACT.corpusSpec; // How to resolve a single transcript from the paginated shards, described once // and embedded in every corpus.json so any HTTP client can navigate without // reading our source. const SHARD_SCHEME = { description: "Transcripts are served as paginated JSON shards — there is no per-video " + "file. To read one video's transcript: (1) GET the channel's transcripts " + "manifest; (2) look up the video id in its `slugToPage` map to get a page " + "number N; (3) GET page-.json (N zero-padded to 4 digits) and take " + "the record whose `id` matches.", transcriptsManifest: " -> { pageCount, slugToPage: { : } }", transcriptPage: "/transcripts//page-.json -> array of { id, title, uploadDate, " + "duration, channel, description, tags, webpageUrl, platform, cues: [{ start, end, text }] }; " + "a record with other English caption tracks whose words differ from its transcript " + "also carries `track` (the transcript's track id, e.g. \"en-orig\") and " + "`altTracks: [{ track, cues }]` (e.g. \"en\", the uploaded captions) — absent otherwise", subsManifest: " -> lighter list-view records under the same slugToPage scheme", summariesIndex: "/summaries/manifest.json + /summaries/page-.json -> cross-channel browse index", pageNumberFormat: "zero-padded to 4 digits, e.g. page 0 -> page-0000.json", } as const; // The social-post corpus: a parallel dataset to video transcripts, served under // the same paginated-shard scheme. A post has no timeline, so it carries a // `createdAt` instant instead of cue timings and its permalink needs no // timestamp fragment. const POST_SCHEME = { description: "Social posts (X/Twitter, Bluesky, forum threads) are archived as a PARALLEL corpus to " + "video transcripts and are served as paginated JSON shards under the same " + "scheme: (1) GET the channel's posts manifest; (2) look up the post id in " + "its `slugToPage` map to get a page number N; (3) GET page-.json and " + "take the record whose `id` matches. Only channels whose source is a social " + "account have a posts manifest.", postsManifest: " -> { pageCount, slugToPage: { : } }", postPage: "/posts//page-.json -> array of { id, slug, channelSlug, author, " + "authorName, createdAt, uploadDate, text, url, platform, threadId, replyTo, " + "quoted, repostOf, isReply, isRepost, links, mediaCount, media, engagement, " + "forum }; `forum` (platform \"xenforo\": one channel is one forum thread) " + "carries { host, threadId, threadTitle, page, position, authorId, editedAt, " + "quotes: [{ postId, author }] }, and quoted text in a forum post's `text` " + "is marked with leading \"> \" lines", ordering: "newest first, by `createdAt` (ISO-8601, ms precision where the source provides it)", dateFilter: "`uploadDate` (YYYYMMDD, derived from createdAt) is carried on every post so " + "the same date filters work across transcripts and posts", permalink: "each post carries its own canonical `url`; no timestamp fragment applies", } as const; // The AI-digest corpus: a DERIVED layer over transcripts, not a parallel source // like posts. Same paginated-shard scheme, but sparse — a video absent from a // digests manifest simply has not been digested, which is the normal case. const DIGEST_SCHEME = { description: "AI digests are chapters and topic tags DERIVED from a video's transcript " + "by a local model, composed with any human corrections before publication. " + "They are served as paginated JSON shards under the same scheme: (1) GET " + "the channel's digests manifest; (2) look up the video id in its " + "`slugToPage` map to get a page number N; (3) GET page-.json and take " + "the record whose `id` matches. THE LAYER IS SPARSE: a video id absent from " + "`slugToPage` has no digest, and a channel with no digests has no manifest " + "at all. Absence means 'not yet generated', never 'nothing to say'.", digestsManifest: " -> { pageCount, slugToPage: { : } }", digestPage: "/digests//page-.json -> array of { id, slug, generatedAt, " + "chapters: [{ id, start, clock, title, decidedBy }], tags: [{ id, tag, " + "decidedBy }], provenance, derivedFrom? }", chapterTiming: "`start` is SECONDS, already snapped to a real transcript cue boundary — use " + "it to seek. `clock` is the raw HH:MM:SS string the model emitted, kept for " + "auditing; do not parse it for timing.", attribution: "`decidedBy` is \"ai\" or \"human\" per item, so machine output and human " + "corrections stay distinguishable after composition.", derivedFrom: "When present, this digest was generated for a DIFFERENT video (the canonical " + "member of a duplicate cluster) and shared onto this one; it names that video " + "and the measured cue-timing offset in seconds. Treat its chapter titles as " + "describing the canonical upload.", } as const; export type CorpusChannel = { slug: string; name: string; videoCount: number; groupId?: string; manifests: { transcripts: string; subs: string; // Present only for social channels (the posts corpus). posts?: string; // Present only for channels with at least one digested video. digests?: string; }; // Present only for social channels. postCount?: number; // Number of this channel's videos that carry a digest. Present only when // non-zero, and deliberately reported ALONGSIDE videoCount rather than // instead of it: the ratio is the coverage of the derived layer, which is // what tells a client whether to expect a digest for an arbitrary video. digestCount?: number; }; export type SiteCorpus = { spec: number; kind: "site"; generatedAt: string; // The software that produced this bundle, e.g. "Archilyzer (https://…)". // Informational: it tells a machine reader where to find the tool that built // the archive it's navigating. Not versioned — see CORPUS_SPEC_VERSION. generator: string; site: { id: string; title: string; description: string; url?: string; hubUrl?: string; // Present only on a PRIVATE site's build (site.json `audience`, release 17 // slice XP): the operator's own reading copy, which no deploy path ships. // A CITED site's build always says who it is for, "public" included. audience?: "private" | "public"; // Present only on a CITED site's build: it publishes its reports and the // moments they cite, and nothing else — `channels` is empty, there are no // shards (spec 5). scope?: "cited"; }; totals: { channels: number; videos: number }; channels: CorpusChannel[]; shardScheme: typeof SHARD_SCHEME; // Present when this site includes at least one social channel. postScheme?: typeof POST_SCHEME; // Present when this site ships at least one digested video. digestScheme?: typeof DIGEST_SCHEME; // Present when this site publishes at least one curated tag. `url` is // /tags.json (the vocabulary + this site's per-tag counts) and `videoField` // names the per-record key those ids appear in — deliberately NOT `tags`, // which on a transcript record is the platform's own keywords. tags?: { url: string; videoField: "curatedTags"; description: string }; // Present when this build ships bulk-download archives (whole-channel zips). bulkArchives?: { manifest: string; note: string }; // Present when this site publishes at least one report (spec 5): the report // index, each entry linking its page; every report's page view and // citations sit beside it under /reports//. reports?: { index: string; count: number; description: string }; // Pointer to the human page on using the archive with AI: the homepage's AI // and MCP doc (AI_DOC_URL; the site's own /use-with-ai page until release // 16). The BYO-key chat is the site's /ask/. useWithAi: string; }; export type HubCorpus = { spec: number; kind: "hub"; generatedAt: string; // See SiteCorpus.generator. generator: string; hub: { title: string; url?: string }; sites: HubCorpusSite[]; federation: { description: string }; useWithAi: string; }; // The join every URL below goes through — "absolute when siteUrl is set, // root-relative otherwise" — now defined once in archive/contract.ts, where the // readers that must reproduce these URLs can reach it. const join = archiveUrl; // Build the per-site corpus index from the public site descriptor (`/site.json`) // plus whether this build emitted bulk archives. export function buildSiteCorpus( descriptor: PublicSiteDescriptor, opts: { hasArchives: boolean; // slug -> archived post count, for the social channels in this site. Absent // / empty means the site has no posts corpus and postScheme is omitted. postCounts?: Record; // slug -> digested video count. Absent / empty means the site ships no // derived layer and digestScheme is omitted. digestCounts?: Record; // Whether this site published a /tags.json (compose writes one only when a // visible tag has a non-zero count here). Absent/false leaves corpus.json // shaped as before apart from the spec bump. hasTags?: boolean; // A private site's build (site.json `audience: "private"`): corpus.json's // `site.audience` says so. Absent/false leaves corpus.json as before. private?: boolean; // How many reports this build published (compose's reports stage). Absent // or 0 leaves corpus.json without a `reports` pointer. reportCount?: number; // A CITED site's build: `site.scope: "cited"` and `site.audience` always. // Its descriptor carries no channels, so the corpus has none. cited?: boolean; }, ): SiteCorpus { const base = descriptor.siteUrl; const postCounts = opts.postCounts ?? {}; const digestCounts = opts.digestCounts ?? {}; const channels: CorpusChannel[] = descriptor.channels.map((c) => { const postCount = postCounts[c.slug]; const digestCount = digestCounts[c.slug]; return { slug: c.slug, name: c.name, videoCount: c.count, ...(c.groupId ? { groupId: c.groupId } : {}), ...(postCount ? { postCount } : {}), ...(digestCount ? { digestCount } : {}), // One definition of the manifest URL shape, shared with every reader. // The conditional spreads stay here: whether a layer is ADVERTISED is a // property of this site's build, not of the contract. manifests: { transcripts: manifestUrl("transcripts", c.slug, base), subs: manifestUrl("subs", c.slug, base), ...(postCount ? { posts: manifestUrl("posts", c.slug, base) } : {}), ...(digestCount ? { digests: manifestUrl("digests", c.slug, base) } : {}), }, }; }); const videos = channels.reduce((n, c) => n + c.videoCount, 0); const corpus: SiteCorpus = { spec: CORPUS_SPEC_VERSION, kind: "site", generatedAt: descriptor.generatedAt, generator: PROJECT_GENERATOR, site: { id: descriptor.siteId, title: descriptor.siteTitle, description: descriptor.siteDescription, ...(descriptor.siteUrl ? { url: descriptor.siteUrl } : {}), ...(descriptor.hubUrl ? { hubUrl: descriptor.hubUrl } : {}), ...(opts.private ? { audience: "private" as const } : opts.cited ? { audience: "public" as const } : {}), ...(opts.cited ? { scope: "cited" as const } : {}), }, totals: { channels: channels.length, videos }, channels, shardScheme: SHARD_SCHEME, useWithAi: AI_DOC_URL, }; // Only advertise the post scheme when this site actually ships posts, so a // pure-video site's corpus.json is unchanged apart from the spec bump. if (Object.values(postCounts).some((n) => n > 0)) { corpus.postScheme = POST_SCHEME; } // Likewise for the derived layer: a site with no digests is unchanged apart // from the spec bump. if (Object.values(digestCounts).some((n) => n > 0)) { corpus.digestScheme = DIGEST_SCHEME; } // The curated vocabulary, when this site ships one. if (opts.hasTags) { corpus.tags = { url: rootFileUrl(TAGS_FILENAME, base), videoField: "curatedTags", description: "Curated cross-channel tags, applied per video by the archive's " + "operator. Fetch tags.json for the vocabulary (id, label, group and " + "the number of videos carrying it on this site), then filter records " + "by their `curatedTags` array — which is NOT the same field as `tags`, " + "the platform's own keywords. Absent on archives built before spec 4.", }; } if (opts.reportCount) { corpus.reports = { index: join(base, REPORTS_INDEX_PATH), count: opts.reportCount, description: "Cited reports. The index lists each report (title, kind, counts, " + "`href` of its page); /reports//page.json is one report with every " + "citation resolved, and /reports//citations.json (an " + "archilyzer-citations set) and citations.csv are its citations as " + "data. A video or audio citation's moment page is /m///" + "-/ (moment.json beside it), a post's /m///.", }; } if (opts.hasArchives) { corpus.bulkArchives = { manifest: join(base, "/archives/manifest.json"), note: "Whole-channel transcript and live-chat archives for offline bulk ingestion.", }; } return corpus; } // Build the aggregate hub corpus: a directory of member origins, each pointing // at its own corpus.json. No shard data — the hub reads members cross-origin. export function buildHubCorpus( members: HubMemberInput[], opts: { hubTitle: string; hubUrl?: string; generatedAt: string }, ): HubCorpus { const sites: HubCorpusSite[] = members .filter((m) => m.siteUrl) .map((m) => { const url = m.siteUrl.replace(/\/+$/, ""); return { siteId: m.siteId, title: m.siteTitle, url, corpus: corpusUrl(url), siteJson: rootFileUrl("site.json", url), pwa: m.pwa === true, }; }); return { spec: CORPUS_SPEC_VERSION, kind: "hub", generatedAt: opts.generatedAt, generator: PROJECT_GENERATOR, hub: { title: opts.hubTitle, ...(opts.hubUrl ? { url: opts.hubUrl } : {}) }, sites, federation: { description: "This is a federation hub. Each site below is an independent origin " + "serving its own corpus.json and CORS-enabled JSON shards. To search " + "the whole federation, fetch each site's corpus.json and follow its " + "shardScheme; results can be merged client-side.", }, useWithAi: AI_DOC_URL, }; } // Cap on how many channels/sites we inline into llms.txt. The full set is always // in corpus.json; llms.txt is a human/LLM-readable overview, not an exhaustive // index. (Bounded either way — this is per-channel, never per-video.) const LLMS_INLINE_LIMIT = 100; // A published report as llms.txt and the sitemap list it: its title and page. export type LlmsReport = { title: string; href: string; subtitle?: string }; function pushReports(out: string[], base: string | undefined, reports: readonly LlmsReport[]): void { for (const r of reports.slice(0, LLMS_INLINE_LIMIT)) { out.push(`- [${reportFullTitle(r)}](${join(base, r.href)})${r.subtitle ? `: ${r.subtitle}` : ""}`); } if (reports.length > LLMS_INLINE_LIMIT) { out.push(`- …and ${reports.length - LLMS_INLINE_LIMIT} more — see the report index.`); } } // Render the per-site llms.txt (llmstxt.org convention: H1 + blockquote summary // + linked sections). A CITED site gets its own variant: its reports, and no // corpus layer, since it publishes none. `reports` lists the site's published // reports, in order (compose's reports stage). export function renderSiteLlmsTxt( corpus: SiteCorpus, opts: { reports?: readonly LlmsReport[] } = {}, ): string { if (corpus.site.scope === "cited") return renderCitedLlmsTxt(corpus, opts.reports ?? []); const base = corpus.site.url; const out: string[] = []; out.push(`# ${corpus.site.title}`); out.push(""); const summary = (corpus.site.description ? corpus.site.description.trim() + " " : "") + `A machine-navigable archive of ${corpus.totals.videos.toLocaleString()} ` + `transcripts across ${corpus.totals.channels} channel(s). ` + `See corpus.json for the shard API.`; out.push(`> ${summary}`); out.push(""); out.push("## Ask AI"); out.push( `- [Ask AI](${join(base, "/ask/")}): in-browser chat (bring your own API key).`, ); out.push( `- [Use with AI](${AI_DOC_URL}): MCP-server setup for Claude Code, Cursor, ` + `and other tools.`, ); out.push(""); out.push("## Corpus"); out.push( `- [corpus.json](${corpusUrl(base)}): machine-readable index — ` + `channels and how to fetch any transcript from the paginated JSON shards.`, ); if (corpus.digestScheme) { const digested = corpus.channels.reduce( (n, c) => n + (c.digestCount ?? 0), 0, ); out.push( `- AI digests: ${digested.toLocaleString()} of these transcripts also carry ` + `machine-generated chapters and topic tags, served under /digests/ — ` + `see corpus.json's digestScheme. Coverage is partial and growing.`, ); } if (corpus.tags) { out.push( `- [tags.json](${corpus.tags.url}): curated cross-channel tags applied ` + `per video by this archive's operator — the vocabulary plus how many ` + `videos carry each one here. Records name them in \`curatedTags\` ` + `(not \`tags\`, which is the platform's own keywords).`, ); } if (corpus.bulkArchives) { out.push( `- [Bulk archives](${join(base, "/downloads")}): whole-channel transcript ` + `and live-chat zips for offline ingestion.`, ); } if (corpus.reports && opts.reports?.length) { out.push(""); out.push("## Reports"); out.push( `- [Report index](${corpus.reports.index}): cited reports — each citation ` + `opens a moment page with the quote, its evidence and a link into the corpus.`, ); pushReports(out, base, opts.reports); } out.push(""); out.push("## Channels"); const shown = corpus.channels.slice(0, LLMS_INLINE_LIMIT); for (const c of shown) { out.push(`- ${c.name} — ${c.videoCount} transcripts: ${c.manifests.transcripts}`); } if (corpus.channels.length > shown.length) { out.push( `- …and ${corpus.channels.length - shown.length} more — see corpus.json for the full list.`, ); } out.push(""); out.push(`Generated by ${corpus.generator}`); return out.join("\n") + "\n"; } // The cited variant: a site that publishes reports and the moments they cite, // and nothing else — no search, no transcripts, no shards to describe. function renderCitedLlmsTxt(corpus: SiteCorpus, reports: readonly LlmsReport[]): string { const base = corpus.site.url; const out: string[] = []; out.push(`# ${corpus.site.title}`); out.push(""); out.push( `> ${corpus.site.description ? corpus.site.description.trim() + " " : ""}` + `${reports.length} cited report(s). This site publishes only its reports and ` + `the moments they cite — there is no searchable corpus here.`, ); out.push(""); out.push("## Reports"); if (corpus.reports) { out.push( `- [Report index](${corpus.reports.index}): every report, as JSON; ` + `/reports//page.json is one report with its citations resolved.`, ); } pushReports(out, base, reports); out.push(""); out.push("## Citations"); out.push( `- Each report's citations as data: /reports//citations.json (an ` + `archilyzer-citations set) and /reports//citations.csv.`, ); out.push( `- A cited video or audio span has a moment page, /m///-/ ` + `(moment.json beside it): the quote, the evidence clip, the transcript lines ` + `around it, the record, a link to the original and every report citing it. ` + `A cited post's is /m///.`, ); out.push( `- [corpus.json](${corpusUrl(base)}): this site's contract — \`site.scope\` is ` + `"cited", and there are no channels or shards.`, ); out.push(""); out.push(`Generated by ${corpus.generator}`); return out.join("\n") + "\n"; } // Render the aggregate hub llms.txt. export function renderHubLlmsTxt(corpus: HubCorpus): string { const base = corpus.hub.url; const out: string[] = []; out.push(`# ${corpus.hub.title}`); out.push(""); out.push( `> A federated hub across ${corpus.sites.length} transcript archive(s). ` + `Each site is an independent origin with its own machine-readable corpus. ` + `See corpus.json for the federation directory.`, ); out.push(""); out.push("## Ask AI"); out.push( `- [Ask AI](${join(base, "/ask/")}): ask across the whole federation ` + `(bring your own API key).`, ); out.push(`- [Use with AI](${AI_DOC_URL}): wire up the MCP server.`); out.push(""); out.push("## Federation"); out.push( `- [corpus.json](${corpusUrl(base)}): machine-readable directory ` + `of every member site and its corpus endpoint.`, ); out.push(""); out.push("## Sites"); for (const s of corpus.sites.slice(0, LLMS_INLINE_LIMIT)) { out.push(`- ${s.title}: ${s.url} (corpus: ${s.corpus})`); } out.push(""); out.push(`Generated by ${corpus.generator}`); return out.join("\n") + "\n"; } // robots.txt — allow crawling and advertise the LLM index + sitemap. export function renderRobotsTxt(opts: { siteUrl?: string }): string { const out: string[] = ["User-agent: *", "Allow: /", ""]; out.push("# AI / LLM corpus index: /llms.txt and /corpus.json"); if (opts.siteUrl) { out.push(`Sitemap: ${opts.siteUrl.replace(/\/+$/, "")}/sitemap.xml`); } return out.join("\n") + "\n"; } // sitemap.xml of the app's static routes. Only meaningful with an absolute // siteUrl; callers skip emission otherwise. One file regardless of corpus size. export function renderSitemapXml(opts: { siteUrl: string; routes: string[]; }): string { const base = opts.siteUrl.replace(/\/+$/, ""); const urls = opts.routes .map((r) => ` ${base}${r === "/" ? "/" : r}`) .join("\n"); return ( `\n` + `\n` + `${urls}\n` + `\n` ); }