Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 066f7e7cea04c3bd50969796a5477bbc25d449e9
parent 3189cc259ad00aca07f2f3974318d0f88b573813
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 03:30:39 -0400

common: corpus spec 5 — corpus.json announces reports, a cited site says site.scope cited with its audience; llms.txt lists reports, and a cited site gets its own variant

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/controller/curatedTagsBuild.test.ts | 2+-
Mcommon/lib/archive/contract.test.ts | 2+-
Mcommon/lib/archive/contract.ts | 6++++--
Mcommon/lib/corpus.test.ts | 9+++++----
Mcommon/lib/corpus.ts | 114++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
5 files changed, 121 insertions(+), 12 deletions(-)

diff --git a/common/controller/curatedTagsBuild.test.ts b/common/controller/curatedTagsBuild.test.ts @@ -217,7 +217,7 @@ test("compose 1: the site publishes /tags.json and corpus.json points at it", as const corpus = JSON.parse( readFileSync(path.join(paths.exportPublicDir, "corpus.json"), "utf8"), ); - assert.equal(corpus.spec, 4); + assert.equal(corpus.spec, 5); assert.equal(corpus.tags.videoField, "curatedTags"); assert.match(corpus.tags.url, /\/tags\.json$/); assert.match( diff --git a/common/lib/archive/contract.test.ts b/common/lib/archive/contract.test.ts @@ -164,7 +164,7 @@ test("shipsPwa: the site flag, or hub mode", () => { test("CONTRACT is frozen where it is published", () => { // These are on the wire. A change here is a change to every deployed archive's // machine contract, so it belongs in a slice that says so, not in a refactor. - assert.equal(CONTRACT.corpusSpec, 4); + assert.equal(CONTRACT.corpusSpec, 5); assert.equal(CONTRACT.siteDescriptor, 1); assert.equal(CONTRACT.manifest, 3); assert.equal(CONTRACT.transcriptsManifest, 1); diff --git a/common/lib/archive/contract.ts b/common/lib/archive/contract.ts @@ -31,8 +31,10 @@ import { TAGS_FILENAME } from "../curatedTags"; // per-field notes in lib/corpus.ts for what a bump means to a client. export const CONTRACT = { // /corpus.json's own `spec`. `generator` is deliberately unversioned; see the - // note in lib/corpus.ts for why a credit line did not bump the spec. - corpusSpec: 4, + // note in lib/corpus.ts for why a credit line did not bump the spec. 5: a + // site may publish reports (`reports`), and a CITED site (`site.scope: + // "cited"`) publishes only them — no channels, no shards. + corpusSpec: 5, // /site.json's `contract` (siteDescriptor.ts). siteDescriptor: 1, // The four manifest versions (manifest.ts). Each is the version field of one diff --git a/common/lib/corpus.test.ts b/common/lib/corpus.test.ts @@ -190,14 +190,15 @@ test("Use with AI is the homepage's AI and MCP doc; the chat is the instance's / } }); -test("the spec is 4, and `generator` is still not why", () => { +test("the spec is 5, and `generator` is still not why", () => { // Guard on the reasoning, not just the number: a bump announces a new - // FETCHABLE document. spec 4 is /tags.json. The informational credit string + // FETCHABLE document. spec 4 is /tags.json, spec 5 the reports index (and a + // cited site that publishes only reports). The informational credit string // added at spec 3's time breaks no reader and did not bump anything — if // either assertion is ever updated, the version-history comment in corpus.ts // must justify why. - assert.equal(CORPUS_SPEC_VERSION, 4); - assert.equal(buildSiteCorpus(descriptor(), { hasArchives: false }).spec, 4); + assert.equal(CORPUS_SPEC_VERSION, 5); + assert.equal(buildSiteCorpus(descriptor(), { hasArchives: false }).spec, 5); }); test("the tags pointer is present only when the site published one", () => { diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts @@ -1,6 +1,7 @@ import type { PublicSiteDescriptor } from "./siteDescriptor"; import { AI_DOC_URL, PROJECT_GENERATOR } from "./project"; import { TAGS_FILENAME } from "./curatedTags"; +import { REPORTS_INDEX_PATH } from "./report/views"; import { CONTRACT, archiveUrl, @@ -34,6 +35,13 @@ import { // where X collabs". The per-record key itself is additive and bumps no manifest // version — an older reader ignores an unknown field, as it always has. // +// v5: reports — a site may publish cited reports, announced as `reports` +// ({ index: "/reports/index.json" }, the report index the export's Reports +// pages read); and a CITED site (site.json `publish: "cited"`) publishes ONLY +// them, saying so as `site.scope: "cited"` with no channels and zero totals. +// An older reader sees an empty corpus there, which is the safe reading: there +// is no shard to fetch. +// // NOT a v4: the `generator` field added below is deliberately unversioned. Every // prior bump announced a new FETCHABLE LAYER — a reader that ignored it would // miss data it could otherwise have retrieved. `generator` is an informational @@ -177,7 +185,12 @@ export type SiteCorpus = { hubUrl?: string; // Present only on a PRIVATE site's build (site.json `audience`, release 17 // slice XP): the operator's own reading copy, which no deploy path ships. - audience?: "private"; + // A CITED site's build always says who it is for, "public" included. + audience?: "private" | "public"; + // Present only on a CITED site's build: it publishes its reports and the + // moments they cite, and nothing else — `channels` is empty, there are no + // shards (spec 5). + scope?: "cited"; }; totals: { channels: number; videos: number }; channels: CorpusChannel[]; @@ -193,6 +206,10 @@ export type SiteCorpus = { tags?: { url: string; videoField: "curatedTags"; description: string }; // Present when this build ships bulk-download archives (whole-channel zips). bulkArchives?: { manifest: string; note: string }; + // Present when this site publishes at least one report (spec 5): the report + // index, each entry linking its page; every report's page view and + // citations sit beside it under /reports/<id>/. + reports?: { index: string; count: number; description: string }; // Pointer to the human page on using the archive with AI: the homepage's AI // and MCP doc (AI_DOC_URL; the site's own /use-with-ai page until release // 16). The BYO-key chat is the site's /ask/. @@ -235,6 +252,12 @@ export function buildSiteCorpus( // A private site's build (site.json `audience: "private"`): corpus.json's // `site.audience` says so. Absent/false leaves corpus.json as before. private?: boolean; + // How many reports this build published (compose's reports stage). Absent + // or 0 leaves corpus.json without a `reports` pointer. + reportCount?: number; + // A CITED site's build: `site.scope: "cited"` and `site.audience` always. + // Its descriptor carries no channels, so the corpus has none. + cited?: boolean; }, ): SiteCorpus { const base = descriptor.siteUrl; @@ -274,7 +297,8 @@ export function buildSiteCorpus( description: descriptor.siteDescription, ...(descriptor.siteUrl ? { url: descriptor.siteUrl } : {}), ...(descriptor.hubUrl ? { hubUrl: descriptor.hubUrl } : {}), - ...(opts.private ? { audience: "private" as const } : {}), + ...(opts.private ? { audience: "private" as const } : opts.cited ? { audience: "public" as const } : {}), + ...(opts.cited ? { scope: "cited" as const } : {}), }, totals: { channels: channels.length, videos }, channels, @@ -304,6 +328,19 @@ export function buildSiteCorpus( "the platform's own keywords. Absent on archives built before spec 4.", }; } + if (opts.reportCount) { + corpus.reports = { + index: join(base, REPORTS_INDEX_PATH), + count: opts.reportCount, + description: + "Cited reports. The index lists each report (title, kind, counts, " + + "`href` of its page); /reports/<id>/page.json is one report with every " + + "citation resolved, and /reports/<id>/citations.json (an " + + "archilyzer-citations set) and citations.csv are its citations as " + + "data. A video or audio citation's moment page is /m/<channel>/<id>/" + + "<start>-<end>/ (moment.json beside it), a post's /m/<channel>/<id>/.", + }; + } if (opts.hasArchives) { corpus.bulkArchives = { manifest: join(base, "/archives/manifest.json"), @@ -355,9 +392,27 @@ export function buildHubCorpus( // index. (Bounded either way — this is per-channel, never per-video.) const LLMS_INLINE_LIMIT = 100; +// A published report as llms.txt and the sitemap list it: its title and page. +export type LlmsReport = { title: string; href: string; subtitle?: string }; + +function pushReports(out: string[], base: string | undefined, reports: readonly LlmsReport[]): void { + for (const r of reports.slice(0, LLMS_INLINE_LIMIT)) { + out.push(`- [${r.title}](${join(base, r.href)})${r.subtitle ? `: ${r.subtitle}` : ""}`); + } + if (reports.length > LLMS_INLINE_LIMIT) { + out.push(`- …and ${reports.length - LLMS_INLINE_LIMIT} more — see the report index.`); + } +} + // Render the per-site llms.txt (llmstxt.org convention: H1 + blockquote summary -// + linked sections). -export function renderSiteLlmsTxt(corpus: SiteCorpus): string { +// + linked sections). A CITED site gets its own variant: its reports, and no +// corpus layer, since it publishes none. `reports` lists the site's published +// reports, in order (compose's reports stage). +export function renderSiteLlmsTxt( + corpus: SiteCorpus, + opts: { reports?: readonly LlmsReport[] } = {}, +): string { + if (corpus.site.scope === "cited") return renderCitedLlmsTxt(corpus, opts.reports ?? []); const base = corpus.site.url; const out: string[] = []; out.push(`# ${corpus.site.title}`); @@ -408,6 +463,15 @@ export function renderSiteLlmsTxt(corpus: SiteCorpus): string { `and live-chat zips for offline ingestion.`, ); } + if (corpus.reports && opts.reports?.length) { + out.push(""); + out.push("## Reports"); + out.push( + `- [Report index](${corpus.reports.index}): cited reports — each citation ` + + `opens a moment page with the quote, its evidence and a link into the corpus.`, + ); + pushReports(out, base, opts.reports); + } out.push(""); out.push("## Channels"); const shown = corpus.channels.slice(0, LLMS_INLINE_LIMIT); @@ -424,6 +488,48 @@ export function renderSiteLlmsTxt(corpus: SiteCorpus): string { return out.join("\n") + "\n"; } +// The cited variant: a site that publishes reports and the moments they cite, +// and nothing else — no search, no transcripts, no shards to describe. +function renderCitedLlmsTxt(corpus: SiteCorpus, reports: readonly LlmsReport[]): string { + const base = corpus.site.url; + const out: string[] = []; + out.push(`# ${corpus.site.title}`); + out.push(""); + out.push( + `> ${corpus.site.description ? corpus.site.description.trim() + " " : ""}` + + `${reports.length} cited report(s). This site publishes only its reports and ` + + `the moments they cite — there is no searchable corpus here.`, + ); + out.push(""); + out.push("## Reports"); + if (corpus.reports) { + out.push( + `- [Report index](${corpus.reports.index}): every report, as JSON; ` + + `/reports/<id>/page.json is one report with its citations resolved.`, + ); + } + pushReports(out, base, reports); + out.push(""); + out.push("## Citations"); + out.push( + `- Each report's citations as data: /reports/<id>/citations.json (an ` + + `archilyzer-citations set) and /reports/<id>/citations.csv.`, + ); + out.push( + `- A cited video or audio span has a moment page, /m/<channel>/<id>/<start>-<end>/ ` + + `(moment.json beside it): the quote, the evidence clip, the transcript lines ` + + `around it, the record, a link to the original and every report citing it. ` + + `A cited post's is /m/<channel>/<id>/.`, + ); + out.push( + `- [corpus.json](${corpusUrl(base)}): this site's contract — \`site.scope\` is ` + + `"cited", and there are no channels or shards.`, + ); + out.push(""); + out.push(`Generated by ${corpus.generator}`); + return out.join("\n") + "\n"; +} + // Render the aggregate hub llms.txt. export function renderHubLlmsTxt(corpus: HubCorpus): string { const base = corpus.hub.url;