commit 066f7e7cea04c3bd50969796a5477bbc25d449e9
parent 3189cc259ad00aca07f2f3974318d0f88b573813
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 03:30:39 -0400
common: corpus spec 5 — corpus.json announces reports, a cited site says site.scope cited with its audience; llms.txt lists reports, and a cited site gets its own variant
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
5 files changed, 121 insertions(+), 12 deletions(-)
diff --git a/common/controller/curatedTagsBuild.test.ts b/common/controller/curatedTagsBuild.test.ts
@@ -217,7 +217,7 @@ test("compose 1: the site publishes /tags.json and corpus.json points at it", as
const corpus = JSON.parse(
readFileSync(path.join(paths.exportPublicDir, "corpus.json"), "utf8"),
);
- assert.equal(corpus.spec, 4);
+ assert.equal(corpus.spec, 5);
assert.equal(corpus.tags.videoField, "curatedTags");
assert.match(corpus.tags.url, /\/tags\.json$/);
assert.match(
diff --git a/common/lib/archive/contract.test.ts b/common/lib/archive/contract.test.ts
@@ -164,7 +164,7 @@ test("shipsPwa: the site flag, or hub mode", () => {
test("CONTRACT is frozen where it is published", () => {
// These are on the wire. A change here is a change to every deployed archive's
// machine contract, so it belongs in a slice that says so, not in a refactor.
- assert.equal(CONTRACT.corpusSpec, 4);
+ assert.equal(CONTRACT.corpusSpec, 5);
assert.equal(CONTRACT.siteDescriptor, 1);
assert.equal(CONTRACT.manifest, 3);
assert.equal(CONTRACT.transcriptsManifest, 1);
diff --git a/common/lib/archive/contract.ts b/common/lib/archive/contract.ts
@@ -31,8 +31,10 @@ import { TAGS_FILENAME } from "../curatedTags";
// per-field notes in lib/corpus.ts for what a bump means to a client.
export const CONTRACT = {
// /corpus.json's own `spec`. `generator` is deliberately unversioned; see the
- // note in lib/corpus.ts for why a credit line did not bump the spec.
- corpusSpec: 4,
+ // note in lib/corpus.ts for why a credit line did not bump the spec. 5: a
+ // site may publish reports (`reports`), and a CITED site (`site.scope:
+ // "cited"`) publishes only them — no channels, no shards.
+ corpusSpec: 5,
// /site.json's `contract` (siteDescriptor.ts).
siteDescriptor: 1,
// The four manifest versions (manifest.ts). Each is the version field of one
diff --git a/common/lib/corpus.test.ts b/common/lib/corpus.test.ts
@@ -190,14 +190,15 @@ test("Use with AI is the homepage's AI and MCP doc; the chat is the instance's /
}
});
-test("the spec is 4, and `generator` is still not why", () => {
+test("the spec is 5, and `generator` is still not why", () => {
// Guard on the reasoning, not just the number: a bump announces a new
- // FETCHABLE document. spec 4 is /tags.json. The informational credit string
+ // FETCHABLE document. spec 4 is /tags.json, spec 5 the reports index (and a
+ // cited site that publishes only reports). The informational credit string
// added at spec 3's time breaks no reader and did not bump anything — if
// either assertion is ever updated, the version-history comment in corpus.ts
// must justify why.
- assert.equal(CORPUS_SPEC_VERSION, 4);
- assert.equal(buildSiteCorpus(descriptor(), { hasArchives: false }).spec, 4);
+ assert.equal(CORPUS_SPEC_VERSION, 5);
+ assert.equal(buildSiteCorpus(descriptor(), { hasArchives: false }).spec, 5);
});
test("the tags pointer is present only when the site published one", () => {
diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts
@@ -1,6 +1,7 @@
import type { PublicSiteDescriptor } from "./siteDescriptor";
import { AI_DOC_URL, PROJECT_GENERATOR } from "./project";
import { TAGS_FILENAME } from "./curatedTags";
+import { REPORTS_INDEX_PATH } from "./report/views";
import {
CONTRACT,
archiveUrl,
@@ -34,6 +35,13 @@ import {
// where X collabs". The per-record key itself is additive and bumps no manifest
// version — an older reader ignores an unknown field, as it always has.
//
+// v5: reports — a site may publish cited reports, announced as `reports`
+// ({ index: "/reports/index.json" }, the report index the export's Reports
+// pages read); and a CITED site (site.json `publish: "cited"`) publishes ONLY
+// them, saying so as `site.scope: "cited"` with no channels and zero totals.
+// An older reader sees an empty corpus there, which is the safe reading: there
+// is no shard to fetch.
+//
// NOT a v4: the `generator` field added below is deliberately unversioned. Every
// prior bump announced a new FETCHABLE LAYER — a reader that ignored it would
// miss data it could otherwise have retrieved. `generator` is an informational
@@ -177,7 +185,12 @@ export type SiteCorpus = {
hubUrl?: string;
// Present only on a PRIVATE site's build (site.json `audience`, release 17
// slice XP): the operator's own reading copy, which no deploy path ships.
- audience?: "private";
+ // A CITED site's build always says who it is for, "public" included.
+ audience?: "private" | "public";
+ // Present only on a CITED site's build: it publishes its reports and the
+ // moments they cite, and nothing else — `channels` is empty, there are no
+ // shards (spec 5).
+ scope?: "cited";
};
totals: { channels: number; videos: number };
channels: CorpusChannel[];
@@ -193,6 +206,10 @@ export type SiteCorpus = {
tags?: { url: string; videoField: "curatedTags"; description: string };
// Present when this build ships bulk-download archives (whole-channel zips).
bulkArchives?: { manifest: string; note: string };
+ // Present when this site publishes at least one report (spec 5): the report
+ // index, each entry linking its page; every report's page view and
+ // citations sit beside it under /reports/<id>/.
+ reports?: { index: string; count: number; description: string };
// Pointer to the human page on using the archive with AI: the homepage's AI
// and MCP doc (AI_DOC_URL; the site's own /use-with-ai page until release
// 16). The BYO-key chat is the site's /ask/.
@@ -235,6 +252,12 @@ export function buildSiteCorpus(
// A private site's build (site.json `audience: "private"`): corpus.json's
// `site.audience` says so. Absent/false leaves corpus.json as before.
private?: boolean;
+ // How many reports this build published (compose's reports stage). Absent
+ // or 0 leaves corpus.json without a `reports` pointer.
+ reportCount?: number;
+ // A CITED site's build: `site.scope: "cited"` and `site.audience` always.
+ // Its descriptor carries no channels, so the corpus has none.
+ cited?: boolean;
},
): SiteCorpus {
const base = descriptor.siteUrl;
@@ -274,7 +297,8 @@ export function buildSiteCorpus(
description: descriptor.siteDescription,
...(descriptor.siteUrl ? { url: descriptor.siteUrl } : {}),
...(descriptor.hubUrl ? { hubUrl: descriptor.hubUrl } : {}),
- ...(opts.private ? { audience: "private" as const } : {}),
+ ...(opts.private ? { audience: "private" as const } : opts.cited ? { audience: "public" as const } : {}),
+ ...(opts.cited ? { scope: "cited" as const } : {}),
},
totals: { channels: channels.length, videos },
channels,
@@ -304,6 +328,19 @@ export function buildSiteCorpus(
"the platform's own keywords. Absent on archives built before spec 4.",
};
}
+ if (opts.reportCount) {
+ corpus.reports = {
+ index: join(base, REPORTS_INDEX_PATH),
+ count: opts.reportCount,
+ description:
+ "Cited reports. The index lists each report (title, kind, counts, " +
+ "`href` of its page); /reports/<id>/page.json is one report with every " +
+ "citation resolved, and /reports/<id>/citations.json (an " +
+ "archilyzer-citations set) and citations.csv are its citations as " +
+ "data. A video or audio citation's moment page is /m/<channel>/<id>/" +
+ "<start>-<end>/ (moment.json beside it), a post's /m/<channel>/<id>/.",
+ };
+ }
if (opts.hasArchives) {
corpus.bulkArchives = {
manifest: join(base, "/archives/manifest.json"),
@@ -355,9 +392,27 @@ export function buildHubCorpus(
// index. (Bounded either way — this is per-channel, never per-video.)
const LLMS_INLINE_LIMIT = 100;
+// A published report as llms.txt and the sitemap list it: its title and page.
+export type LlmsReport = { title: string; href: string; subtitle?: string };
+
+function pushReports(out: string[], base: string | undefined, reports: readonly LlmsReport[]): void {
+ for (const r of reports.slice(0, LLMS_INLINE_LIMIT)) {
+ out.push(`- [${r.title}](${join(base, r.href)})${r.subtitle ? `: ${r.subtitle}` : ""}`);
+ }
+ if (reports.length > LLMS_INLINE_LIMIT) {
+ out.push(`- …and ${reports.length - LLMS_INLINE_LIMIT} more — see the report index.`);
+ }
+}
+
// Render the per-site llms.txt (llmstxt.org convention: H1 + blockquote summary
-// + linked sections).
-export function renderSiteLlmsTxt(corpus: SiteCorpus): string {
+// + linked sections). A CITED site gets its own variant: its reports, and no
+// corpus layer, since it publishes none. `reports` lists the site's published
+// reports, in order (compose's reports stage).
+export function renderSiteLlmsTxt(
+ corpus: SiteCorpus,
+ opts: { reports?: readonly LlmsReport[] } = {},
+): string {
+ if (corpus.site.scope === "cited") return renderCitedLlmsTxt(corpus, opts.reports ?? []);
const base = corpus.site.url;
const out: string[] = [];
out.push(`# ${corpus.site.title}`);
@@ -408,6 +463,15 @@ export function renderSiteLlmsTxt(corpus: SiteCorpus): string {
`and live-chat zips for offline ingestion.`,
);
}
+ if (corpus.reports && opts.reports?.length) {
+ out.push("");
+ out.push("## Reports");
+ out.push(
+ `- [Report index](${corpus.reports.index}): cited reports — each citation ` +
+ `opens a moment page with the quote, its evidence and a link into the corpus.`,
+ );
+ pushReports(out, base, opts.reports);
+ }
out.push("");
out.push("## Channels");
const shown = corpus.channels.slice(0, LLMS_INLINE_LIMIT);
@@ -424,6 +488,48 @@ export function renderSiteLlmsTxt(corpus: SiteCorpus): string {
return out.join("\n") + "\n";
}
+// The cited variant: a site that publishes reports and the moments they cite,
+// and nothing else — no search, no transcripts, no shards to describe.
+function renderCitedLlmsTxt(corpus: SiteCorpus, reports: readonly LlmsReport[]): string {
+ const base = corpus.site.url;
+ const out: string[] = [];
+ out.push(`# ${corpus.site.title}`);
+ out.push("");
+ out.push(
+ `> ${corpus.site.description ? corpus.site.description.trim() + " " : ""}` +
+ `${reports.length} cited report(s). This site publishes only its reports and ` +
+ `the moments they cite — there is no searchable corpus here.`,
+ );
+ out.push("");
+ out.push("## Reports");
+ if (corpus.reports) {
+ out.push(
+ `- [Report index](${corpus.reports.index}): every report, as JSON; ` +
+ `/reports/<id>/page.json is one report with its citations resolved.`,
+ );
+ }
+ pushReports(out, base, reports);
+ out.push("");
+ out.push("## Citations");
+ out.push(
+ `- Each report's citations as data: /reports/<id>/citations.json (an ` +
+ `archilyzer-citations set) and /reports/<id>/citations.csv.`,
+ );
+ out.push(
+ `- A cited video or audio span has a moment page, /m/<channel>/<id>/<start>-<end>/ ` +
+ `(moment.json beside it): the quote, the evidence clip, the transcript lines ` +
+ `around it, the record, a link to the original and every report citing it. ` +
+ `A cited post's is /m/<channel>/<id>/.`,
+ );
+ out.push(
+ `- [corpus.json](${corpusUrl(base)}): this site's contract — \`site.scope\` is ` +
+ `"cited", and there are no channels or shards.`,
+ );
+ out.push("");
+ out.push(`Generated by ${corpus.generator}`);
+ return out.join("\n") + "\n";
+}
+
// Render the aggregate hub llms.txt.
export function renderHubLlmsTxt(corpus: HubCorpus): string {
const base = corpus.hub.url;