commit b3532d20c2705f3520a6bdad141b775f616bc3cc
parent 1362716c5c8b15293d2542cf549602f5c2c151ec
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 05:24:38 -0400
Merge report-r5-consumers (readers read a cited corpus as empty and name its reports; MCP list_reports/get_report; a cited site is never listed by the hub or homepage)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
25 files changed, 1124 insertions(+), 15 deletions(-)
diff --git a/PUBLISH.md b/PUBLISH.md
@@ -100,6 +100,14 @@ shell's own pages and files, or a file over 25 MiB, or more than 20,000 files, f
the build, and every deploy path refuses it. A site switched to cited is refused at
deploy until it is built again.
+A cited site is not a searchable archive, so the family never lists it, whatever
+`listed` says (`isListedSite`): it is no hub member, has no homepage card, counts in
+no family total and no other site's footer links to it. A full site with reports is
+listed as before. Readers of the contract see a cited site as an empty corpus with
+reports: the MCP server's `list_reports` / `get_report` read them (and its
+`list_channels` says "cited-only site: N report(s)"), and umtool's cue walk refuses
+one by name — it has no transcripts to cut a report video from.
+
Deploy-only ships whatever is in `export/out`, which the basic build composes one
site at a time into a single shared directory — so it **refuses, before starting a
job, if `export/out` holds a build of another site** (or no build at all), naming the
diff --git a/SITE.md b/SITE.md
@@ -163,7 +163,7 @@ Default: absent
## `listed`
-Whether the family lists this site. Opt-OUT: absent/true = listed, only an explicit `false` is written. An unlisted site still builds and deploys as before, and its own pages are unchanged; it is left out of the homepage (cards, chart, `/stats`), the hub (members, federated search, `/corpus.json`, `/llms.txt`), every other site's footer, and the published `channel-sites.json` and pooled `stats/`. A channel only unlisted sites expose is in none of the family's public totals; a channel a listed site also exposes is credited to the listed one.
+Whether the family lists this site. Opt-OUT: absent/true = listed, only an explicit `false` is written. An unlisted site still builds and deploys as before, and its own pages are unchanged; it is left out of the homepage (cards, chart, `/stats`), the hub (members, federated search, `/corpus.json`, `/llms.txt`), every other site's footer, and the published `channel-sites.json` and pooled `stats/`. A channel only unlisted sites expose is in none of the family's public totals; a channel a listed site also exposes is credited to the listed one. A private site and a cited site (`publish: "cited"`) are never listed, whatever `listed` says.
Default: `true`
@@ -175,7 +175,7 @@ Default: absent
## `publish`
-What this site publishes. `"full"` (the default; absent): the searchable corpus of its `channels`. `"cited"`: only the site's `reports` and the moments they cite — no search, no browse, no full transcripts, no archives; `channels` is then the pool its citations may resolve against. The cited scope is applied by the reports pipeline at build time. Only `"cited"` is written; any other value reads as `"full"`.
+What this site publishes. `"full"` (the default; absent): the searchable corpus of its `channels`. `"cited"`: only the site's `reports` and the moments they cite — no search, no browse, no full transcripts, no archives; `channels` is then the pool its citations may resolve against. The cited scope is applied by the reports pipeline at build time. A cited site is not a searchable archive, so the family does not list it (as `listed: false`, whatever `listed` says): no hub membership, no homepage card or totals, no footer link from other sites. Only `"cited"` is written; any other value reads as `"full"`.
Default: `"full"`
diff --git a/common/bin/compose-hub.test.ts b/common/bin/compose-hub.test.ts
@@ -172,6 +172,47 @@ test("an unlisted site is in none of the hub's files; a listed one is in each",
}
});
+// Report sites: a cited site (`publish: "cited"`) publishes reports, not a
+// searchable archive — no hub member, whatever `listed` says. A full site with
+// reports stays one.
+test("a cited site is no hub member; a full site with reports is", async () => {
+ const root = mkdtempSync(path.join(tmpdir(), "compose-hub-"));
+ const log = console.log;
+ try {
+ const paths = fixturePaths(root);
+ const write = (id: string, site: Record<string, unknown>) => {
+ mkdirSync(path.join(paths.sitesDir, id), { recursive: true });
+ writeFileSync(path.join(paths.sitesDir, id, "site.json"), JSON.stringify(site));
+ };
+ write("fixture-full", {
+ siteTitle: "Full Fixture",
+ siteUrl: "https://fixture-full.example",
+ reports: ["demo-report"],
+ });
+ write("fixture-cited", {
+ siteTitle: "Cited Fixture",
+ siteUrl: "https://fixture-cited.example",
+ publish: "cited",
+ listed: true,
+ });
+ console.log = () => {};
+ await main({ paths });
+ console.log = log;
+ const pool = JSON.parse(
+ readFileSync(path.join(paths.exportPublicDir, "hub-sites.json"), "utf8"),
+ ) as Array<{ siteId: string }>;
+ assert.deepEqual(pool.map((s) => s.siteId), ["fixture-full"]);
+ for (const f of ["corpus.json", "llms.txt"]) {
+ const text = readFileSync(path.join(paths.exportPublicDir, f), "utf8");
+ assert.ok(text.includes("fixture-full.example"), `${f} lists the full site`);
+ assert.ok(!text.includes("fixture-cited"), `${f} names the cited site`);
+ }
+ } finally {
+ console.log = log;
+ rmSync(root, { recursive: true, force: true });
+ }
+});
+
// Release 17 slice XP (the review's HIGH 1): a site's compose leaves its data
// in public/ — a private site's X posts included — and the hub builds from
// public/ next. compose-hub removes every per-site entry, through a link only
diff --git a/common/bin/compose-hub.ts b/common/bin/compose-hub.ts
@@ -5,7 +5,8 @@
// reads every archive cross-origin at runtime. It emits:
//
// public/hub-sites.json <- the built-in trusted pool (listSites with a siteUrl,
-// listed — site.json `listed`, isListedSite)
+// listed — site.json `listed`, isListedSite; never
+// a private or a cited report site)
// public/hub-summary.json <- the official instances' numbers, the homepage's
// own (lib/hubSummary.ts) — OPTIONAL: skipped when
// there is no index to walk
diff --git a/common/controller/poolSummary.test.ts b/common/controller/poolSummary.test.ts
@@ -27,6 +27,16 @@ test("channel-sites.json names listed sites only; a channel only an unlisted sit
assert.ok(!JSON.stringify(channelSitesOf(sites)).includes("fixture-unlisted"));
});
+// Report sites: a CITED site is never listed either — its channels are its
+// citations' pool, not a published archive.
+test("channel-sites.json leaves out a cited site", () => {
+ const sites = [
+ parseSite("fixture-a", { channels: [{ slug: "shared" }] }),
+ parseSite("fixture-cited", { publish: "cited", channels: [{ slug: "shared" }, { slug: "own" }] }),
+ ];
+ assert.deepEqual(channelSitesOf(sites), { shared: ["fixture-a"] });
+});
+
// Release 17 slice XP: a PRIVATE site is never listed, so it is in neither; and
// with X posts private an X channel a public site leaves out of its build is
// not mapped to that site (buildPoolSummary passes the narrowing).
diff --git a/common/lib/archive/reader-fs.ts b/common/lib/archive/reader-fs.ts
@@ -29,6 +29,13 @@ import { DUPLICATES_FILENAME, type DuplicateReport } from "../duplicates";
import type { ChannelDigestsManifest, VideoDigest } from "../digests";
import type { StatsManifest, VideoStat } from "../stats";
import { pageFileName } from "./contract";
+import { isReportId } from "../report/schema";
+import {
+ REPORTS_INDEX_PATH,
+ reportViewPath,
+ type ReportIndexEntry,
+ type ReportPageView,
+} from "../report/views";
import { recordRead } from "./io-stats";
import {
DEFAULT_PAGE_CONCURRENCY,
@@ -38,11 +45,15 @@ import {
buildDuplicateIndex,
buildStatsIndex,
buildVideoIndex,
+ coerceReportIndex,
+ coerceReportPage,
+ corpusScopeOf,
pageCacheBudgetBytes,
parseGroupsManifest,
type ArchiveReader,
type ChannelGroups,
type ChannelRef,
+ type CorpusScope,
type DuplicateIndex,
type IndexedVideo,
type SiteCorpusJson,
@@ -84,6 +95,9 @@ export class LocalSource implements ArchiveReader {
// The composed site's own declared origin, learned from corpus.json the first
// time the channel list is read. Undefined = not looked at yet.
private siteOrigin: string | null | undefined;
+ // What corpus.json said the site publishes, learned with the channel list.
+ private siteScope?: CorpusScope;
+ private reportIndex?: Promise<ReportIndexEntry[]>;
constructor(private dir: string) {
this.label = `local:${dir}`;
@@ -131,6 +145,42 @@ export class LocalSource implements ArchiveReader {
this.duplicates = undefined;
this.stats = undefined;
this.siteOrigin = undefined;
+ this.siteScope = undefined;
+ this.reportIndex = undefined;
+ }
+
+ async scope(): Promise<CorpusScope> {
+ await this.listChannels();
+ return this.siteScope ?? "full";
+ }
+
+ // A composed dir with no reports/index.json publishes none (ENOENT is data,
+ // as for tags.json).
+ reports(): Promise<ReportIndexEntry[]> {
+ this.reportIndex ??= readLocalJson<unknown>(
+ path.join(this.dir, REPORTS_INDEX_PATH),
+ "reportIndex",
+ )
+ .then(coerceReportIndex)
+ .catch(() => []);
+ return this.reportIndex;
+ }
+
+ // The id is validated before it touches a path: a report id is a slug, and
+ // anything else (a `..`, a slash) names no report rather than another file.
+ async reportPage(id: string): Promise<ReportPageView | null> {
+ if (!isReportId(id)) return null;
+ let raw: unknown;
+ try {
+ raw = await readLocalJson<unknown>(
+ path.join(this.dir, reportViewPath(id)),
+ "reportPage",
+ );
+ } catch (e) {
+ if ((e as NodeJS.ErrnoException).code === "ENOENT") return null;
+ throw e;
+ }
+ return coerceReportPage(raw, id);
}
subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
@@ -234,6 +284,10 @@ export class LocalSource implements ArchiveReader {
const declared = corpus.site?.url?.trim();
this.siteOrigin =
declared && !PREFER_PLATFORM_LINKS ? declared.replace(/\/+$/, "") : null;
+ // A CITED site (spec 5) has no channels by design: never fall back to
+ // listing a transcripts/ directory a stale public dir might still hold.
+ this.siteScope = corpusScopeOf(corpus);
+ if (this.siteScope === "cited") return [];
if (Array.isArray(corpus.channels) && corpus.channels.length > 0) {
return corpus.channels.map((c) => ({
key: c.slug,
diff --git a/common/lib/archive/reader-reports.test.ts b/common/lib/archive/reader-reports.test.ts
@@ -0,0 +1,235 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, mkdir, rm, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { CONTRACT } from "./contract";
+import { RemoteSource, coerceReportIndex, corpusScopeOf } from "./reader";
+import { LocalSource } from "./reader-fs";
+import { HubSource } from "./reader-hub";
+import {
+ REPORT_INDEX_FORMAT,
+ REPORT_PAGE_FORMAT,
+ REPORT_VIEWS_VERSION,
+ type ReportIndexView,
+ type ReportPageView,
+} from "../report/views";
+
+// Spec 5 for the three readers: a CITED site (corpus.json `site.scope:
+// "cited"`) reads as an empty corpus without an error and names its reports; a
+// full site with reports reads its channels exactly as before.
+
+const ORIGIN = "https://reports.example";
+
+const CITED_CORPUS = {
+ spec: CONTRACT.corpusSpec,
+ kind: "site",
+ site: { id: "demo-reports", title: "Demo reports", url: ORIGIN, scope: "cited", audience: "public" },
+ totals: { channels: 0, videos: 0 },
+ channels: [],
+ reports: { index: `${ORIGIN}/reports/index.json`, count: 1 },
+};
+
+const FULL_CORPUS = {
+ spec: CONTRACT.corpusSpec,
+ kind: "site",
+ site: { id: "demo-full", title: "Demo full", url: ORIGIN },
+ channels: [{ slug: "demo-channel", name: "Demo Channel", videoCount: 1 }],
+ reports: { index: `${ORIGIN}/reports/index.json`, count: 1 },
+};
+
+const INDEX: ReportIndexView = {
+ format: REPORT_INDEX_FORMAT,
+ version: REPORT_VIEWS_VERSION,
+ reports: [
+ {
+ id: "demo-report",
+ kind: "factcheck",
+ title: "A demo report",
+ href: "/reports/demo-report/",
+ citationCount: 1,
+ claimCount: 1,
+ tally: [{ verdict: "CORROBORATED", count: 1 }],
+ },
+ ],
+};
+
+const PAGE = {
+ format: REPORT_PAGE_FORMAT,
+ version: REPORT_VIEWS_VERSION,
+ id: "demo-report",
+ kind: "factcheck",
+ title: "A demo report",
+ sources: {},
+ verdicts: {},
+ citations: {},
+ sections: [{ id: "s1", title: "One", claims: [] }],
+} as unknown as ReportPageView;
+
+function installFetch(body: Record<string, unknown>, fetches: string[] = []): () => void {
+ const real = globalThis.fetch;
+ globalThis.fetch = (async (input: RequestInfo | URL) => {
+ const url = String(input);
+ fetches.push(url);
+ if (!(url in body)) return new Response("not found", { status: 404, statusText: "Not Found" });
+ return new Response(JSON.stringify(body[url]), { status: 200 });
+ }) as typeof fetch;
+ return () => {
+ globalThis.fetch = real;
+ };
+}
+
+test("corpusScopeOf: only an explicit cited scope is cited; a pre-spec-5 corpus is full", () => {
+ assert.equal(corpusScopeOf(CITED_CORPUS), "cited");
+ assert.equal(corpusScopeOf(FULL_CORPUS), "full");
+ assert.equal(corpusScopeOf({ site: { scope: "other" } }), "full");
+ assert.equal(corpusScopeOf(null), "full");
+});
+
+test("coerceReportIndex keeps the entries of an index and nothing else", () => {
+ assert.deepEqual(coerceReportIndex(INDEX).map((e) => e.id), ["demo-report"]);
+ assert.deepEqual(coerceReportIndex({ reports: INDEX.reports }), [], "no format: not an index");
+ assert.deepEqual(coerceReportIndex(null), []);
+ assert.deepEqual(
+ coerceReportIndex({ ...INDEX, reports: [{ id: 1 }, null, ...INDEX.reports] }).map((e) => e.id),
+ ["demo-report"],
+ );
+});
+
+test("remote: a cited site is an empty corpus with its reports, read without an error", async () => {
+ const fetches: string[] = [];
+ const restore = installFetch(
+ {
+ [`${ORIGIN}/corpus.json`]: CITED_CORPUS,
+ [`${ORIGIN}/reports/index.json`]: INDEX,
+ [`${ORIGIN}/reports/demo-report/page.json`]: PAGE,
+ },
+ fetches,
+ );
+ try {
+ const r = new RemoteSource(ORIGIN);
+ assert.deepEqual(await r.listChannels(), []);
+ assert.equal(await r.scope(), "cited");
+ assert.deepEqual((await r.reports()).map((e) => e.title), ["A demo report"]);
+ assert.equal((await r.reportPage("demo-report"))?.title, "A demo report");
+ assert.equal(await r.reportPage("nope"), null, "an unknown report is null, not a throw");
+ // The empty reads a tool makes on any source degrade, they do not throw.
+ assert.deepEqual((await r.loadGroups()).groups, []);
+ assert.equal((await r.videoIndex()).size, 0);
+ await r.reports();
+ assert.equal(fetches.filter((u) => u.endsWith("/reports/index.json")).length, 1, "the index is cached");
+ } finally {
+ restore();
+ }
+});
+
+test("remote: a report id is one path segment — it cannot leave /reports/", async () => {
+ const fetches: string[] = [];
+ const restore = installFetch({ [`${ORIGIN}/corpus.json`]: CITED_CORPUS }, fetches);
+ try {
+ assert.equal(await new RemoteSource(ORIGIN).reportPage("../corpus"), null);
+ assert.deepEqual(fetches, [`${ORIGIN}/reports/..%2Fcorpus/page.json`]);
+ } finally {
+ restore();
+ }
+});
+
+test("remote: a full site with reports reads its channels as before; one without has no reports", async () => {
+ const restore = installFetch({
+ [`${ORIGIN}/corpus.json`]: FULL_CORPUS,
+ [`${ORIGIN}/reports/index.json`]: INDEX,
+ });
+ try {
+ const r = new RemoteSource(ORIGIN);
+ assert.deepEqual((await r.listChannels()).map((c) => c.slug), ["demo-channel"]);
+ assert.equal(await r.scope(), "full");
+ assert.equal((await r.reports()).length, 1);
+ } finally {
+ restore();
+ }
+ const restore2 = installFetch({ [`${ORIGIN}/corpus.json`]: { ...FULL_CORPUS, spec: 4, reports: undefined } });
+ try {
+ const r = new RemoteSource(ORIGIN);
+ assert.equal(await r.scope(), "full");
+ assert.deepEqual(await r.reports(), [], "a 404 index is no reports");
+ } finally {
+ restore2();
+ }
+});
+
+test("hub: a cited member federates no channels and fails nothing", async () => {
+ const HUB = "https://hub.example";
+ const FULL = "https://full.example";
+ const restore = installFetch({
+ [`${HUB}/corpus.json`]: {
+ kind: "hub",
+ sites: [
+ { siteId: "demo-reports", title: "Demo reports", url: ORIGIN },
+ { siteId: "demo-full", title: "Demo full", url: FULL },
+ ],
+ },
+ [`${ORIGIN}/corpus.json`]: CITED_CORPUS,
+ [`${FULL}/corpus.json`]: { ...FULL_CORPUS, site: { ...FULL_CORPUS.site, url: FULL } },
+ });
+ try {
+ const hub = new HubSource(HUB);
+ const channels = await hub.listChannels();
+ assert.deepEqual(channels.map((c) => c.key), ["demo-full/demo-channel"]);
+ assert.equal((await hub.videoIndex()).size, 0);
+ } finally {
+ restore();
+ }
+});
+
+async function writeDir(files: Record<string, unknown>): Promise<string> {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "reader-reports-"));
+ for (const [rel, body] of Object.entries(files)) {
+ const file = path.join(dir, rel);
+ await mkdir(path.dirname(file), { recursive: true });
+ await writeFile(file, JSON.stringify(body));
+ }
+ return dir;
+}
+
+test("local: a cited dir never falls back to a stale transcripts/ listing", async () => {
+ const dir = await writeDir({
+ "corpus.json": CITED_CORPUS,
+ "reports/index.json": INDEX,
+ "reports/demo-report/page.json": PAGE,
+ // A stale channel tree a cited build must not be read as.
+ "transcripts/demo-channel/manifest.json": { version: 1, pageCount: 0, slugToPage: {} },
+ });
+ try {
+ const l = new LocalSource(dir);
+ assert.deepEqual(await l.listChannels(), []);
+ assert.equal(await l.scope(), "cited");
+ assert.equal(l.publicOrigin(), ORIGIN);
+ assert.deepEqual((await l.reports()).map((e) => e.id), ["demo-report"]);
+ assert.equal((await l.reportPage("demo-report"))?.id, "demo-report");
+ assert.equal(await l.reportPage("missing"), null);
+ assert.equal(await l.reportPage("../corpus.json"), null, "not a report id");
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("local: a full dir with no reports reads as before", async () => {
+ const dir = await writeDir({ "corpus.json": { ...FULL_CORPUS, reports: undefined } });
+ try {
+ const l = new LocalSource(dir);
+ assert.deepEqual((await l.listChannels()).map((c) => c.slug), ["demo-channel"]);
+ assert.equal(await l.scope(), "full");
+ assert.deepEqual(await l.reports(), []);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("local: a page.json that is not a report page is an error, not an absence", async () => {
+ const dir = await writeDir({ "corpus.json": CITED_CORPUS, "reports/demo-report/page.json": { hello: 1 } });
+ try {
+ await assert.rejects(() => new LocalSource(dir).reportPage("demo-report"), /not a report page view/);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
diff --git a/common/lib/archive/reader.ts b/common/lib/archive/reader.ts
@@ -50,6 +50,13 @@ import {
type HubSite,
} from "./contract";
import { recordRead } from "./io-stats";
+import {
+ REPORT_INDEX_FORMAT,
+ REPORTS_INDEX_PATH,
+ reportViewPath,
+ type ReportIndexEntry,
+ type ReportPageView,
+} from "../report/views";
// A channel the source can serve. `siteId`/`siteUrl` are only populated in hub
// mode (so results can be attributed to the owning member site); `key` is the
@@ -400,6 +407,61 @@ export interface ArchiveReader {
// channel per call. OPTIONAL on the interface for the usual reason.
digestsManifest?(ch: ChannelRef): Promise<ChannelDigestsManifest | null>;
digestPage?(ch: ChannelRef, page: number): Promise<VideoDigest[]>;
+ // What the site publishes (corpus spec 5): "cited" for a report site, whose
+ // channel list is EMPTY BY DESIGN — no channels, no shards, only its reports
+ // and the moments they cite. A caller says so rather than reporting an empty
+ // archive. Learned from corpus.json with the channel list. OPTIONAL: a stub
+ // has no corpus.json, and a hub's members each have their own.
+ scope?(): Promise<CorpusScope>;
+ // The site's report index (/reports/index.json, spec 5): one entry per
+ // published report. [] when it publishes none — absent, malformed, or a site
+ // built before spec 5. Cached per source.
+ reports?(): Promise<ReportIndexEntry[]>;
+ // One report's page view (/reports/<id>/page.json), every citation
+ // resolved; null when the site publishes no report of that id.
+ reportPage?(id: string): Promise<ReportPageView | null>;
+}
+
+// What a site publishes, as its corpus.json says (spec 5). "full" is the
+// searchable corpus every site has always been, and what a corpus.json from
+// before spec 5 reads as; "cited" is a report site.
+export type CorpusScope = "full" | "cited";
+
+export function corpusScopeOf(corpus: SiteCorpusJson | null | undefined): CorpusScope {
+ return corpus?.site?.scope === "cited" ? "cited" : "full";
+}
+
+// A report index as served, folded to its entries. Tolerant like the tags
+// read: a document that is not an index, or an entry without an id and a
+// title, is dropped rather than thrown — an index the reader cannot use is a
+// site with no reports it can name.
+export function coerceReportIndex(raw: unknown): ReportIndexEntry[] {
+ const doc = (raw ?? {}) as { format?: unknown; reports?: unknown };
+ if (doc.format !== REPORT_INDEX_FORMAT || !Array.isArray(doc.reports)) return [];
+ return doc.reports.filter(
+ (e): e is ReportIndexEntry =>
+ !!e &&
+ typeof e === "object" &&
+ typeof (e as ReportIndexEntry).id === "string" &&
+ typeof (e as ReportIndexEntry).title === "string",
+ );
+}
+
+// A report page view as served, or a throw: unlike the index, a page asked for
+// by id that does not parse as one is an error worth naming, not an absence.
+export function coerceReportPage(raw: unknown, id: string): ReportPageView {
+ const v = (raw ?? {}) as Partial<ReportPageView>;
+ if (typeof v.id !== "string" || !Array.isArray(v.sections) || typeof v.citations !== "object") {
+ throw new Error(`report ${id}: page.json is not a report page view`);
+ }
+ return v as ReportPageView;
+}
+
+// A report id as ONE path segment of a URL. An id is a lowercase slug
+// (lib/report/schema.ts REPORT_ID_RE); anything else is encoded so it cannot
+// leave /reports/ — it simply names no report.
+function reportPagePathFor(id: string): string {
+ return reportViewPath(encodeURIComponent(id));
}
// An HTTP STATUS the archive itself returned, as opposed to a transport
@@ -435,7 +497,10 @@ export type SiteCorpusJson = {
channels?: CorpusJsonChannel[];
// The composed site's own declared public origin — the deployed archilyzer
// viewer these shards were built for. Present in every spec-3 corpus.json.
- site?: { id?: string; title?: string; url?: string };
+ // `scope: "cited"` (spec 5): a report site, with no channels (corpusScopeOf).
+ site?: { id?: string; title?: string; url?: string; scope?: string };
+ // Spec 5: the site's reports, when it publishes any.
+ reports?: { index?: string; count?: number };
};
export type HubCorpusJson = {
@@ -600,6 +665,10 @@ export class RemoteSource implements ArchiveReader {
private postsManifests = new PromiseMap<ChannelPostsManifest | null>();
private transcriptManifests = new PromiseMap<ChannelTranscriptsManifest>();
private transcriptPages: PageCache<TranscriptDetail[]>;
+ // What corpus.json said the site publishes, learned by readChannels().
+ private siteScope?: CorpusScope;
+ private reportIndex?: Promise<ReportIndexEntry[]>;
+ private reportPages = new PromiseMap<ReportPageView | null>();
// Over HTTP the cost is latency, not parse, so a wider window is the dominant
// win — this is where bounded concurrency actually pays.
@@ -632,6 +701,41 @@ export class RemoteSource implements ArchiveReader {
this.index = undefined;
this.duplicates = undefined;
this.stats = undefined;
+ this.siteScope = undefined;
+ this.reportIndex = undefined;
+ this.reportPages.clear();
+ }
+
+ async scope(): Promise<CorpusScope> {
+ await this.listChannels();
+ return this.siteScope ?? "full";
+ }
+
+ // A 404 is the answer "no reports" (a full site without any, or one built
+ // before spec 5) and is cached; any other failure is not memoised.
+ reports(): Promise<ReportIndexEntry[]> {
+ this.reportIndex ??= this.readReportIndex()
+ .then(coerceReportIndex)
+ .catch((e: unknown) => {
+ if (e instanceof ArchiveHttpError && e.status === 404) return [];
+ this.reportIndex = undefined;
+ throw e;
+ });
+ return this.reportIndex;
+ }
+
+ reportPage(id: string): Promise<ReportPageView | null> {
+ return this.reportPages.take(id, async () => {
+ try {
+ return coerceReportPage(
+ await this.getJson<unknown>(reportPagePathFor(id), "reportPage"),
+ id,
+ );
+ } catch (e) {
+ if (e instanceof ArchiveHttpError && e.status === 404) return null;
+ throw e;
+ }
+ });
}
subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
@@ -795,6 +899,10 @@ export class RemoteSource implements ArchiveReader {
return this.getJson<unknown>(rootFileUrl("search-aliases.json"));
}
+ readReportIndex(): Promise<unknown> {
+ return this.getJson<unknown>(REPORTS_INDEX_PATH, "reportIndex");
+ }
+
readTags(): Promise<unknown> {
return this.getJson<unknown>(rootFileUrl(TAGS_FILENAME));
}
@@ -867,8 +975,13 @@ export class RemoteSource implements ArchiveReader {
return this.channelList;
}
+ // A CITED site's corpus.json lists no channels (spec 5); its scope is
+ // recorded so a caller can say "a report site" rather than "empty", and a
+ // stray channel entry on one is not read as a corpus.
private async readChannels(): Promise<ChannelRef[]> {
const corpus = await this.readCorpus();
+ this.siteScope = corpusScopeOf(corpus);
+ if (this.siteScope === "cited") return [];
return (corpus.channels ?? []).map((c) => ({
key: c.slug,
slug: c.slug,
diff --git a/common/lib/homepageSummary.test.ts b/common/lib/homepageSummary.test.ts
@@ -305,6 +305,17 @@ test("an unlisted site is in no array and no total; a channel it shares is the l
assert.equal(s.version, 6);
});
+// Report sites: a cited site (`publish: "cited"`) is not a searchable archive,
+// so the homepage lists it no more than an unlisted one.
+test("a cited site is in no array and no total, exactly as an unlisted one", () => {
+ const cited = { ...site("zeta", ["q1", "a1"], "https://zeta.example"), publish: "cited" } as Site;
+ const channelSites = { ...CHANNEL_SITES, a1: ["alpha", "zeta"], q1: ["zeta"] };
+ const own = [stat({ channelSlug: "q1", id: "q-1", uploadDate: "20251101", duration: 7200 })];
+ const s = buildHomepageSummary([...STATS, ...own], channelSites, [...SITES, cited], NOW);
+ assert.deepEqual(s, buildHomepageSummary(STATS, CHANNEL_SITES, SITES, NOW));
+ assert.ok(!JSON.stringify(s).includes("zeta"));
+});
+
test("an unlisted site with no siteUrl, or with every channel shared, changes nothing either", () => {
const base = buildHomepageSummary(STATS, CHANNEL_SITES, SITES, NOW);
// No siteUrl: never public, and its own channel is still in no total (unlike
diff --git a/common/lib/homepageSummary.ts b/common/lib/homepageSummary.ts
@@ -12,7 +12,8 @@ import { VIDEO_STATES, type VideoState } from "./availability";
// multi-MB whole-pool stats dataset.
//
// Scope: the chart "universe" is PUBLIC sites only (those with a siteUrl that
-// are listed — site.json `listed`, lib/siteSchema.ts isListedSite), and every
+// are listed — site.json `listed`, lib/siteSchema.ts isListedSite; never a
+// private or a cited report site), and every
// video is attributed to a single PRIMARY public site (the first, by sorted id,
// exposing its channel) so the Site and Channel breakdowns partition the same
// set and combined totals stay honest. The KPI `totals` (and `availability`)
@@ -43,6 +44,10 @@ import { VIDEO_STATES, type VideoState } from "./availability";
// and a channel only unlisted sites expose is in no total — `totals` and
// `availability` included. No field was added or removed; the number says the
// totals' scope moved.
+//
+// Still v6: a cited report site (site.json `publish: "cited"`) is left out
+// exactly as an unlisted one (lib/siteSchema.ts isListedSite) — the same scope
+// rule, applied to a kind of site no summary had yet counted.
export const HOMEPAGE_SUMMARY_VERSION = 6;
// Day buckets are capped to this many trailing days so the embedded summary stays
diff --git a/common/lib/siteSchema.test.ts b/common/lib/siteSchema.test.ts
@@ -376,6 +376,19 @@ test("isListedSite is the key's default; channelsOnlyOnUnlistedSites keeps a sha
assert.deepEqual([...channelsOnlyOnUnlistedSites([sites[0]])], []);
});
+test("a cited site is never listed, whatever `listed` says; a full site with reports is listed as before", () => {
+ assert.equal(isListedSite({ publish: "cited" }), false);
+ assert.equal(isListedSite({ publish: "cited", listed: true }), false);
+ assert.equal(isListedSite({ publish: "full" }), true);
+ assert.equal(isListedSite(parseSite("full-with-reports", { reports: ["demo-report"] })), true);
+ const sites = [
+ parseSite("shown", { channels: [{ slug: "shared" }] }),
+ parseSite("reports", { publish: "cited", channels: [{ slug: "shared" }, { slug: "pool-only" }] }),
+ ];
+ // A cited site's channels are its citations' pool, not a published archive.
+ assert.deepEqual([...channelsOnlyOnUnlistedSites(sites)], ["pool-only"]);
+});
+
test("the footer never links an unlisted sibling, and an unlisted site's own footer still lists the rest", () => {
const current = parseSite("cur", {
siteUrl: "https://cur.example",
diff --git a/common/lib/siteSchema.ts b/common/lib/siteSchema.ts
@@ -184,11 +184,11 @@ export const SITE_FIELD_DOCS: FieldDocs<Site> = {
siteUrl:
"Absolute public URL of this site's deployment, e.g. `https://jeralyzer.pages.dev` (trimmed, trailing slashes removed; anything not absolute http(s) is dropped). Drives the cross-site footer: a site with no siteUrl is omitted from every other site's list.",
listed:
- "Whether the family lists this site. Opt-OUT: absent/true = listed, only an explicit `false` is written. An unlisted site still builds and deploys as before, and its own pages are unchanged; it is left out of the homepage (cards, chart, `/stats`), the hub (members, federated search, `/corpus.json`, `/llms.txt`), every other site's footer, and the published `channel-sites.json` and pooled `stats/`. A channel only unlisted sites expose is in none of the family's public totals; a channel a listed site also exposes is credited to the listed one.",
+ "Whether the family lists this site. Opt-OUT: absent/true = listed, only an explicit `false` is written. An unlisted site still builds and deploys as before, and its own pages are unchanged; it is left out of the homepage (cards, chart, `/stats`), the hub (members, federated search, `/corpus.json`, `/llms.txt`), every other site's footer, and the published `channel-sites.json` and pooled `stats/`. A channel only unlisted sites expose is in none of the family's public totals; a channel a listed site also exposes is credited to the listed one. A private site and a cited site (`publish: \"cited\"`) are never listed, whatever `listed` says.",
audience:
'Who this site is built for. `"public"` (the default; absent) or `"private"`: the operator\'s own reading copy, built on this machine and never deployed — every deploy path (Build & deploy, Deploy, `archilyzer deploy site`, Build & deploy all, docker/publish-site.sh) refuses it before any upload, while a build without a deploy still works. A private site is never listed (as `listed: false`, whatever `listed` says), publishes no `hubUrl`, and its `/corpus.json` says `"audience": "private"`. Content kept from the public — X posts while `social.x.visibility` is `"private"` — is built only into private sites. Only `"private"` is written.',
publish:
- 'What this site publishes. `"full"` (the default; absent): the searchable corpus of its `channels`. `"cited"`: only the site\'s `reports` and the moments they cite — no search, no browse, no full transcripts, no archives; `channels` is then the pool its citations may resolve against. The cited scope is applied by the reports pipeline at build time. Only `"cited"` is written; any other value reads as `"full"`.',
+ 'What this site publishes. `"full"` (the default; absent): the searchable corpus of its `channels`. `"cited"`: only the site\'s `reports` and the moments they cite — no search, no browse, no full transcripts, no archives; `channels` is then the pool its citations may resolve against. The cited scope is applied by the reports pipeline at build time. A cited site is not a searchable archive, so the family does not list it (as `listed: false`, whatever `listed` says): no hub membership, no homepage card or totals, no footer link from other sites. Only `"cited"` is written; any other value reads as `"full"`.',
reports:
"The site's published reports, in display order: report ids (lowercase slugs, `[a-z0-9][a-z0-9-]*`), each a directory under `sites/<siteId>/reports/`. A report directory not named here is a draft and is not published. Invalid and repeated ids are dropped. Absent/empty = no reports.",
relatedSites:
@@ -224,9 +224,13 @@ export function isValidSiteId(id: unknown): id is string {
// pure summary builder can use it without importing file I/O.
//
// A PRIVATE site (`audience: "private"`) is never listed, whatever `listed`
-// says: it is never deployed, so there is nothing at its URL to list.
-export function isListedSite(site: Pick<Site, "listed" | "audience">): boolean {
- return site.listed !== false && !isPrivateSite(site);
+// says: it is never deployed, so there is nothing at its URL to list. Nor is a
+// CITED site (`publish: "cited"`): it publishes reports and the moments they
+// cite, not a searchable archive, so it is no hub member (federated search
+// would find no channel there), no homepage card and in no family total. A
+// full site with reports is listed as before.
+export function isListedSite(site: Pick<Site, "listed" | "audience" | "publish">): boolean {
+ return site.listed !== false && !isPrivateSite(site) && !isCitedSite(site);
}
// The channels whose content belongs to unlisted sites alone: exposed by at
@@ -235,7 +239,7 @@ export function isListedSite(site: Pick<Site, "listed" | "audience">): boolean {
// and a channel no site exposes (pool-only) is not here either — the family's
// instance-wide totals have always counted it.
export function channelsOnlyOnUnlistedSites(
- sites: readonly Pick<Site, "listed" | "audience" | "channels">[],
+ sites: readonly Pick<Site, "listed" | "audience" | "publish" | "channels">[],
): Set<string> {
const onListed = new Set<string>();
const onUnlisted = new Set<string>();
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **A report video's cue lookup names a site that publishes only its reports.** Pointing a report-to-video manifest at such a site (`corpus.json` `site.scope: "cited"`) used to fail with "channel … is not in corpus.json"; it now says the site publishes no transcripts and to use a full archive or a local corpus. The site form's **Publish** hint says a cited-only site is never listed on the homepage or the hub.
- **A site's build composes its reports, and a site that publishes only its reports ships nothing else.** Every site's compose now writes the reports its `site.json` publishes: each report's page and its citations as `citations.json` and `citations.csv` under `/reports/<id>/`, its cited stills, a page per cited moment with the record, the transcript lines around the span and every report that cites it, and the clips and post captures `archilyzer reports prepare` made for it, only the cited ones. Each quote is checked against the record as it is composed (a span's against its cues within 5 s either side, read from `en-orig` when the `en` track has no cues; a post's against its text) and the score, time and method are written into the citation, replacing any typed by hand. The build stops with the list of every problem before anything is written: an invalid report, a citation of a channel outside the site or of a post the site may not carry, a missing record, still or post, a quote that matches less than 60 % of what the record says, and a citation without prepared media or with media cut for another span (`--allow-missing-media` on `archilyzer compose site` and `build site` lets those two through, without a clip). A site with `publish: "cited"` removes everything corpus-shaped from `export/public` before it writes its reports, and its built `out/` is checked against what a cited site may hold: anything else, a file over 25 MiB or more than 20,000 files fails the build, and every deploy path (the Publish tab, `deploy site`, Build & deploy, Build & deploy all, the container build) refuses it, as it refuses a site set to cited whose last build was a full one. The hub's compose removes a report site's files too.
- **A site has a Reports tab.** `/sites/<site>/reports` lists every report under the site's `reports/` directory — the published ones in their order, then the drafts — with its kind, dates, sections, claims, citations by kind and, for a fact-check, how many claims carry each verdict. Each report's problems, from the same checker the prepare step and the build use, open under it. A draft with no problems can be published, and a published report moved up or down or unpublished; each writes only the site's `reports` list, applied to the list as it is on disk at that moment, so it never overwrites another change to the site. "Prepare evidence media" queues the `reports-prepare` job, and beside it the tab shows the last prepared media (moments by kind, total size, problems by kind) and links the last prepare job. What the site publishes (full or cited) is shown with a link to Settings, where it is changed.
- **A site can say what it publishes, and which reports.** `site.json` takes `publish` — `"full"`, the searchable corpus every site has been (the default, never written), or `"cited"`, only the site's reports and the moments they cite — and `reports`, the ordered ids of its published reports (each a slug; invalid and repeated ids are dropped). The site form has a Publish control and lists the site's reports read-only; saving the form keeps the stored list. A cited site still builds as a full one until the reports pipeline applies the scope. SITE.md documents both keys.
diff --git a/editor/app/sites/components/SiteForm.tsx b/editor/app/sites/components/SiteForm.tsx
@@ -377,6 +377,8 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) {
A cited-only site publishes no search, browse, full transcripts or
archives; its channels are the pool its reports' citations resolve
against. The reports pipeline applies this scope when the site is built.
+ It is never listed on the homepage or the hub: it is not a searchable
+ archive.
</span>
</label>
<div className="flex flex-col gap-1 text-sm">
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,6 +1,8 @@
# Changelog
## [Unreleased]
+- **MCP: a site's reports can be read, and a site that publishes only reports says so instead of looking empty.** Two new tools: `list_reports` lists the reports a site publishes (id, title, kind, claim and citation counts, a fact-check's verdict tally, its page), and `get_report` reads one — its tally, then each section's claims with their verdicts and findings and, for every citation, the verbatim quote, the original (the platform at the cited second, the post, the document) and the site's moment page; `section` reads one section. On a site that publishes only its reports, `list_channels`, `list_sources` and `resolve_source` say "cited-only site: N report(s)" where they said "No channels found"; on a site with reports as well, `list_sources` and `resolve_source` say how many. The archive readers (local, remote) read `corpus.json` spec 5: a cited site is an empty corpus without an error, and a local copy of one is never read from a stale `transcripts/` folder. A hub has no reports of its own; `list_reports` says to name a member site.
+- **The hub leaves out a site that publishes only its reports.** A site with `publish: "cited"` is not a searchable archive, so it is no hub member (not in federated search, the hub's `corpus.json` or `llms.txt`) and no other site's footer links to it, whatever its **List on the Archilyzer homepage and hub** setting says. A site with reports that publishes its full corpus is a member as before. Needs a hub rebuild and deploy once such a site exists.
- **`corpus.json` is spec 5: it names a site's reports, and a site that publishes only reports says so.** A site with reports adds `reports` to its `corpus.json` (`index`: `/reports/index.json`, the count, and how to read a report's page, its citations and its moment pages) and a Reports section to `llms.txt`; its sitemap lists the report and moment pages. A site that publishes only its reports has `"scope": "cited"` and its audience under `site`, no channels and zero totals, an `llms.txt` that lists its reports and how their citations and moment pages are read, and a `site.json` with no channels. A reader that does not know spec 5 sees an empty corpus there. Needs a rebuild and deploy of each site.
- **A site can show cited reports, and every citation opens on a page of its own.** A site built with reports has a **Reports** link in its header and a page at `/reports/` listing them. A report's page has its title, subtitle, dates and the document under review, its archive links listed once under it and folded away ("N archive links in context"); a fact-check's tally of verdicts; the summary; the sections and their claims, each with its verdict, the document's own sentence as an image with a link back to the document and only the archive links that sit in that sentence (at most five), the findings and the evidence cards; a numbered reference list; and links to download its citations as JSON and CSV. A citation in the text shows as its words plus a number: hovering it, focusing the number or tapping it once shows a card of the citation (the quote, who said it and when, a picture or the post's screenshot, and how closely the quote matched the transcript when it was checked); the words open what it cites and the number jumps to its reference. A cited span of a video or audio record opens at `/m/<channel>/<id>/<start>-<end>/`: a short clip of the span with a little context either side, the quote, the transcript lines around it, the record's title, channel and date, a link to the original at that time, and every report on the site that cites it. A cited post opens at `/m/<channel>/<id>/` with its screenshot and text. A site that publishes only its reports (`site.json` `publish: "cited"`) opens on the report index and has no search, Ask AI, downloads or duplicates. A site with no reports is unchanged. Needs a rebuild and deploy of each site.
diff --git a/homepage/CHANGELOG.md b/homepage/CHANGELOG.md
@@ -1,6 +1,7 @@
# Homepage Changelog
## [Unreleased]
+- **A site that publishes only its reports is not on the homepage.** A site with `publish: "cited"` has no Official Instances card, chart series, `/stats` entry or recent item, is not in `channel-sites.json`, and the channels only it carries count in no total — as an unlisted site, whatever its listing setting says. A site with reports that publishes its full corpus is listed as before.
- **The AI and MCP doc has a Ten-minute setup.** Right after the MCP server's introduction, one block runs Claude Code against a published archive, the Jeralyzer as the example: clone the source (or unpack the tarball on Downloads), `pnpm install`, `claude mcp add archilyzer`, start `claude` and try `/ask`; then what it needs, why the server must be registered as `archilyzer` (the shipped `/ask` and `/sweep` call `mcp__archilyzer__…`), the two optional editor lines for `fetch_clip`, `TRANSCRIPT_HUB_URL`, where the `mcp.json` form for other clients is, and WSL2 on Windows. "What it can do" is a heading of its own after it. Every archive's **Use with AI** link now lands on this page.
- **The Ten-minute setup starts the server with the source's own command, `pnpm --silent -C "$PWD" archilyzer mcp`**, as the README and the MCP server's README do. `--silent` keeps pnpm's own lines off the output the client reads the server's replies on, and a note says so. The note on other clients gives the `mcp.json` entry's arguments in the same form.
- **`/source/` links the source's history.** A History block — how many commits (past 10,000, "the latest 10,000 of N"), the newest one (linking to its page), and links to the Log, the Refs and the Atom feed — shows when the build published the history pages (`/source/git/`, rendered by stagit); without them there is no History block. The history pages open on the homepage's ground (the reader's stored choice, else Dark; without JavaScript, the system's), start with one line back to `/source/`, and their Files page is an index into the raw tree. The e2e shows the page with and without a history from a fixture publish (`E2E_SOURCE_PUBLIC_DIR`, never read by a production build), and walks the real pages when the checkout has published them.
diff --git a/mcp/README.md b/mcp/README.md
@@ -15,6 +15,7 @@ clip window; the MCP itself still writes nothing.
| Tool | What it does |
|------|--------------|
| `list_channels` | List channels **organized under their channel groups** (name, slug, video count; site in hub mode), with a compact group cheat-sheet (`id · name · N channels`) for scoping. |
+| `list_reports` / `get_report` | The cited reports a site publishes (corpus spec 5): the index (title, kind, claim and citation counts, a fact-check's verdict tally), then one report's sections and claims with every citation's **verbatim quote, original URL and the site's moment page**. See [Cited reports](#cited-reports). |
| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets — **each `[mm:ss]` is a clickable link to that exact moment** (or a compact `[mm:ss\|sec]` with `link_style:"base"`). Alias-aware, pageable, and **filterable** (`states`, `date_from`/`date_to`, `media_type`, `age`, `exclude`, `scopes`). A page that isn't the whole match set is flagged **above** the hits. |
| `enumerate_matches` | A query's **complete** match set as a worklist (id/title/channel/date + batch count) in **one scan**. Takes the **same filters** as `search_transcripts`, so the two can never disagree about coverage. The tool to use whenever you need to count or cover everything. |
| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around one query or up to 8 (`queries`), with per-query counts; or full transcripts without a query. Reads posts too. |
@@ -23,7 +24,7 @@ clip window; the MCP itself still writes nothing.
| `get_video_metadata` | Everything known about one video without the transcript body: metadata, plus **view/like counts, cue count and transcript coverage** (`stats/`), **other archived copies of the same recording** with an explicit timings-aligned verdict (`duplicates.json`), and **AI chapters/tags** where they exist (`digests/`). |
| `fetch_clip` | The media behind a cited moment, **fetched by the local editor** (`POST /api/media/fetch-window`) through its paced, cookie-aware, provenanced job — never a yt-dlp run by hand. Needs `ARCHILYZER_EDITOR_URL` (default `http://localhost:3001`) and `WORKER_TOKEN` (the editor's own) in this server's env; without them it says so and fetches nothing. The editor must already archive the cited channel (a channel dir under its `transcripts/`), else it answers 404 `Channel "<slug>" not found`: an MCP pointed at a public site with a fresh editor gets that on every clip. A window is the cited span ± `pad` (default 3 s), at most 15 min, and lands at `channels/<slug>/data/<id>/clips/`; `full: true` fetches the whole recording into the saved-video store (needs a video the editor already knows). `maxHeight` (144–2160) caps the source height: a window is fetched at or under it (default 720); a whole recording at 720 or less is saved as the editor's 720p H.264 preset and above 720 at the original quality (omitted, the channel's source-video quality applies). A file already on disk is returned as it is, never re-fetched for a different cap, and the answer gives its height and says when it is taller than asked. Waits up to `wait_seconds` (default 90, max 300), then returns the job id to resume with `job`; a client with a 60 s default request timeout must raise it or pass `wait_seconds` ≤ 50 — the fetch continues on the editor either way; resume it with `job`, and once it has finished the same request finds it cached. While it waits it sends one progress notification per poll to a client that asked for progress (a `progressToken`), which keeps a reset-on-progress timeout alive. A Rumble embed id is mapped to the editor's slug id through the record's `webpageUrl`, so pass the citing corpus as `source`; a video not in `source` is passed through as cited (known limitation). The file is a read-only corpus artifact. |
| `open_link` | Paste an archilyzer viewer **share link** to re-run that exact search here (query tree + every filter, at full fidelity) — plan, results and corpus handle in **one** call. `dry_run:true` for the plan alone. |
-| `list_sources` | Show the **default** corpus and, with a hub, its member sites as ready-to-paste handles. |
+| `list_sources` | Show the **default** corpus and, with a hub, its member sites as ready-to-paste handles. A site with reports says how many; a cited-only site says it is one. |
| `resolve_source` | Turn a URL or site name into the canonical `source` handle and check it can be read. Changes nothing. |
| `sweep_plan` / `ask_plan` | Turn a plain-English request (plus an optional pasted link) into a resolved, step-by-step plan. What `/sweep` and `/ask` call. |
@@ -320,6 +321,22 @@ so a translated citation would look perfectly plausible and point at the wrong
moment of a different upload. Absent means *not measured*, and not measured means
*no*.
+## Cited reports
+
+A site may publish cited reports (corpus spec 5): `/reports/index.json` lists them
+and `/reports/<id>/page.json` is one report with every citation resolved — the
+same files the site's Reports pages read. `list_reports` reads the index;
+`get_report` (`report`, optionally one `section`) renders a report as text: its
+verdict tally, then each claim with its verdict, its findings and, per citation,
+the quote, the original (the platform at the cited second, the post, the
+document) and the site's moment page (`/m/<channel>/<id>/<start>-<end>/`).
+
+A **cited-only** site (`corpus.json` `site.scope: "cited"`) publishes its reports
+and the moments they cite and nothing else: no channels, no transcripts, nothing
+to search. `list_channels`, `list_sources` and `resolve_source` say "cited-only
+site: N report(s)" there rather than reporting an empty corpus. Reports are per
+site: a hub has none of its own, so name a member with `source:"remote:<url>"`.
+
## How it reads the corpus
Three changes, in increasing order of how much they buy:
diff --git a/mcp/src/protocol.test.ts b/mcp/src/protocol.test.ts
@@ -35,6 +35,8 @@ const TSX = path.join(HERE, "..", "node_modules", ".bin", "tsx");
const EXPECTED_TOOLS = [
"list_channels",
"list_tags",
+ "list_reports",
+ "get_report",
"search_transcripts",
"enumerate_matches",
"get_transcript",
diff --git a/mcp/src/reports.test.ts b/mcp/src/reports.test.ts
@@ -0,0 +1,224 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, mkdir, rm, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { Client, InMemoryTransport } from "@modelcontextprotocol/client";
+import type { Report } from "yt-dlp-transcript-common/lib/report/schema";
+import {
+ REPORT_INDEX_FORMAT,
+ REPORT_VIEWS_VERSION,
+ buildReportPageView,
+ reportIndexEntry,
+} from "yt-dlp-transcript-common/lib/report/views";
+import { CONTRACT } from "yt-dlp-transcript-common/lib/archive/contract";
+import { LocalSource, type ShardSource } from "./source";
+import { createServer } from "./server";
+import { renderReportIndex, reportsLine } from "./reports";
+
+// list_reports / get_report and the "cited-only site" line, over a composed
+// public dir on disk read by the real LocalSource.
+
+const ORIGIN = "https://reports.example";
+
+const REPORT: Report = {
+ format: "archilyzer-report",
+ version: 1,
+ id: "demo-report",
+ kind: "factcheck",
+ title: "A demo fact-check",
+ subtitle: "Of an article",
+ summary: "It starts [here](cite:v1).",
+ published: "2026-10-01",
+ subject: { source: "s0" },
+ sources: {
+ s0: { kind: "article", title: "An article", url: "https://example.org/a", publisher: "Example" },
+ },
+ citations: {
+ v1: { kind: "video", channel: "demo-channel", id: "abc123", start: 61, end: 75.5, quote: "the cited words" },
+ p1: { kind: "post", channel: "demo-social", id: "123", quote: "a posted line" },
+ s1: { kind: "source", source: "s0", quote: "the article's sentence" },
+ },
+ sections: [
+ {
+ id: "one",
+ title: "Section one",
+ claims: [
+ {
+ id: "c1",
+ text: "The article claims a thing.",
+ verdict: "CONTRADICTED",
+ sourceQuote: { citation: "s1" },
+ findings: "The recording says otherwise [here](cite:v1).",
+ citations: ["p1"],
+ },
+ ],
+ },
+ { id: "two", title: "Section two", body: "Nothing to check.", claims: [] },
+ ],
+};
+
+const VIEW = buildReportPageView(REPORT, {
+ record: (c) => ({
+ channel: c.channel,
+ channelTitle: "Demo Channel",
+ id: c.id,
+ title: `Recording ${c.id}`,
+ date: "2026-01-02",
+ originalUrl:
+ c.kind === "post" ? `https://social.example/${c.id}` : `https://video.example/watch?v=${c.id}&t=61`,
+ }),
+ post: () => ({ author: "@demo", text: "a posted line, and more" }),
+});
+
+const INDEX = {
+ format: REPORT_INDEX_FORMAT,
+ version: REPORT_VIEWS_VERSION,
+ reports: [reportIndexEntry(VIEW)],
+};
+
+async function writeSite(opts: { cited: boolean; reports: boolean }): Promise<string> {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "mcp-reports-"));
+ const files: Record<string, unknown> = {
+ "corpus.json": {
+ spec: CONTRACT.corpusSpec,
+ kind: "site",
+ site: {
+ id: "demo-site",
+ title: "Demo",
+ url: ORIGIN,
+ ...(opts.cited ? { scope: "cited", audience: "public" } : {}),
+ },
+ channels: opts.cited ? [] : [{ slug: "demo-channel", name: "Demo Channel", videoCount: 1 }],
+ ...(opts.reports ? { reports: { index: `${ORIGIN}/reports/index.json`, count: 1 } } : {}),
+ },
+ };
+ if (opts.reports) {
+ files["reports/index.json"] = INDEX;
+ files["reports/demo-report/page.json"] = VIEW;
+ }
+ for (const [rel, body] of Object.entries(files)) {
+ const file = path.join(dir, rel);
+ await mkdir(path.dirname(file), { recursive: true });
+ await writeFile(file, JSON.stringify(body));
+ }
+ return dir;
+}
+
+async function connect(source: ShardSource): Promise<Client> {
+ const server = createServer(source);
+ const [ct, st] = InMemoryTransport.createLinkedPair();
+ const client = new Client({ name: "test", version: "0" }, { capabilities: {} });
+ await Promise.all([server.connect(st), client.connect(ct)]);
+ return client;
+}
+
+async function call(client: Client, name: string, args: Record<string, unknown> = {}) {
+ const res = (await client.callTool({ name, arguments: args })) as {
+ content: { text: string }[];
+ isError?: boolean;
+ };
+ return { text: res.content.map((c) => c.text).join("\n"), isError: res.isError === true };
+}
+
+test("a cited site: discovery tools say cited-only with its report count, not an empty corpus", async () => {
+ const dir = await writeSite({ cited: true, reports: true });
+ const client = await connect(new LocalSource(dir));
+ try {
+ const channels = await call(client, "list_channels");
+ assert.equal(channels.isError, false);
+ assert.match(channels.text, /cited-only site: 1 report\(s\)/);
+ assert.doesNotMatch(channels.text, /No channels found/);
+
+ assert.match((await call(client, "list_sources")).text, /reports: cited-only site: 1 report\(s\)/);
+ assert.match((await call(client, "resolve_source", { source: "default" })).text, /reports: cited-only site/);
+ } finally {
+ await client.close();
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("list_reports names each report with its counts, tally and page", async () => {
+ const dir = await writeSite({ cited: true, reports: true });
+ const client = await connect(new LocalSource(dir));
+ try {
+ const out = (await call(client, "list_reports")).text;
+ assert.match(out, /1 report\(s\) in local:.*cited-only site/);
+ assert.match(out, /- demo-report · A demo fact-check — Of an article/);
+ assert.match(out, /factcheck · 1 claim\(s\) · 3 citation\(s\) · 2026-10-01/);
+ assert.match(out, /verdicts: Contradicted 1/);
+ assert.match(out, new RegExp(`${ORIGIN}/reports/demo-report/`));
+ } finally {
+ await client.close();
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("get_report: claims with their citations' quote, original and moment URLs", async () => {
+ const dir = await writeSite({ cited: true, reports: true });
+ const client = await connect(new LocalSource(dir));
+ try {
+ const { text, isError } = await call(client, "get_report", { report: "demo-report" });
+ assert.equal(isError, false);
+ assert.match(text, /# A demo fact-check/);
+ assert.match(text, /verdicts: Contradicted 1/);
+ assert.match(text, /under review: An article \(Example\) https:\/\/example\.org\/a/);
+ assert.match(text, /\[Contradicted\] The article claims a thing\. \(#c1\)/);
+ assert.match(text, /findings: The recording says otherwise/);
+ // The source sentence, the span the findings cite, then the claim's post.
+ const at = (s: string) => text.indexOf(s);
+ assert.ok(at('"the article\'s sentence"') < at('"the cited words"'));
+ assert.ok(at('"the cited words"') < at('"a posted line"'));
+ assert.match(text, /video Demo Channel · Recording abc123 · 2026-01-02 @ 61–75 s/);
+ assert.match(text, /original: https:\/\/video\.example\/watch\?v=abc123&t=61/);
+ assert.match(text, new RegExp(`moment: ${ORIGIN}/m/demo-channel/abc123/61\\.00-75\\.50/`));
+ assert.match(text, /original: https:\/\/social\.example\/123/);
+ assert.match(text, new RegExp(`moment: ${ORIGIN}/m/demo-social/123/`));
+ assert.match(text, new RegExp(`page: ${ORIGIN}/reports/demo-report/`));
+ assert.match(text, /## Section two \(#two\)/);
+
+ const one = (await call(client, "get_report", { report: "demo-report", section: "two" })).text;
+ assert.match(one, /## Section two/);
+ assert.doesNotMatch(one, /Section one/);
+
+ const badSection = await call(client, "get_report", { report: "demo-report", section: "nope" });
+ assert.equal(badSection.isError, true);
+ assert.match(badSection.text, /no section nope — its sections: one, two/);
+
+ const missing = await call(client, "get_report", { report: "nope" });
+ assert.equal(missing.isError, true);
+ assert.match(missing.text, /report not found: nope — this source publishes: demo-report/);
+ } finally {
+ await client.close();
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("a full site: channels as before, and its reports are counted where it has any", async () => {
+ const withReports = await writeSite({ cited: false, reports: true });
+ const without = await writeSite({ cited: false, reports: false });
+ const a = await connect(new LocalSource(withReports));
+ const b = await connect(new LocalSource(without));
+ try {
+ const channels = (await call(a, "list_channels")).text;
+ assert.match(channels, /1 channel\(s\)/);
+ assert.match((await call(a, "list_sources")).text, /reports: 1 report\(s\) published/);
+
+ assert.doesNotMatch((await call(b, "list_sources")).text, /reports:/);
+ assert.match((await call(b, "list_reports")).text, /publishes no reports/);
+ const missing = await call(b, "get_report", { report: "demo-report" });
+ assert.equal(missing.isError, true);
+ assert.match(missing.text, /this source publishes no reports/);
+ } finally {
+ await a.close();
+ await b.close();
+ await rm(withReports, { recursive: true, force: true });
+ await rm(without, { recursive: true, force: true });
+ }
+});
+
+test("a source with no reports of its own (a hub) points at its members", () => {
+ const out = renderReportIndex("hub:https://hub.example", { supported: false, scope: "full", reports: [] }, null);
+ assert.match(out, /reports are per site/);
+ assert.equal(reportsLine({ supported: false, scope: "full", reports: [] }), null);
+});
diff --git a/mcp/src/reports.ts b/mcp/src/reports.ts
@@ -0,0 +1,246 @@
+// Cited reports over MCP: list_reports / get_report, and the one sentence every
+// discovery tool says about a source's reports.
+//
+// A site may publish reports (corpus spec 5): /reports/index.json lists them
+// and /reports/<id>/page.json is one report with every citation resolved
+// (common/lib/report/views.ts — the export site's pages read the same files).
+// A CITED site publishes nothing else: no channels, no shards. Its empty
+// channel list is the site's shape, not a failure, and a tool that reported "no
+// channels" there would read as a broken corpus — so every discovery tool says
+// "cited-only site: N report(s)" instead.
+//
+// Rendering is pure (views in, text out); the reads are the reader's
+// (ArchiveReader.scope / reports / reportPage), optional because a hub and the
+// in-memory stubs have no reports of their own.
+
+import { extractCiteRefs } from "yt-dlp-transcript-common/lib/citations/inline";
+import {
+ orderedCitations,
+ reportPagePath,
+ verdictTally,
+ type CitationView,
+ type ClaimView,
+ type ReportIndexEntry,
+ type ReportPageView,
+ type SectionView,
+} from "yt-dlp-transcript-common/lib/report/views";
+import type { CorpusScope, ShardSource } from "./source";
+
+// What a source says about its reports. `supported` is false for a source with
+// no reports of its own (a hub, a stub); a read that fails reads as none.
+export type SourceReports = {
+ supported: boolean;
+ scope: CorpusScope;
+ reports: ReportIndexEntry[];
+};
+
+export async function sourceReports(source: ShardSource): Promise<SourceReports> {
+ if (typeof source.reports !== "function") {
+ return { supported: false, scope: "full", reports: [] };
+ }
+ let scope: CorpusScope = "full";
+ try {
+ scope = typeof source.scope === "function" ? await source.scope() : "full";
+ } catch {
+ // an unreadable corpus.json is reported by the tool that needs it
+ }
+ let reports: ReportIndexEntry[] = [];
+ try {
+ reports = await source.reports();
+ } catch {
+ // a transport failure on the index: no reports this call can name
+ }
+ return { supported: true, scope, reports };
+}
+
+// The one line a discovery tool adds, or null when there is nothing to say (a
+// full site with no reports, or a source with no reports of its own).
+export function reportsLine(r: SourceReports): string | null {
+ if (r.scope === "cited") {
+ return (
+ `cited-only site: ${r.reports.length} report(s) — it publishes its ` +
+ `reports and the moments they cite, no channels or transcripts to ` +
+ `search. list_reports / get_report read them.`
+ );
+ }
+ if (r.reports.length > 0) {
+ return `${r.reports.length} report(s) published — list_reports / get_report read them.`;
+ }
+ return null;
+}
+
+// A site-root path made absolute against the site's origin, when one is known.
+function absolute(origin: string | null, href: string): string {
+ if (!origin || /^https?:\/\//i.test(href)) return href;
+ return `${origin.replace(/\/+$/, "")}${href.startsWith("/") ? href : `/${href}`}`;
+}
+
+function tallyText(
+ tally: readonly { verdict: string; count: number }[] | undefined,
+ styles?: Record<string, { label: string }>,
+): string {
+ if (!tally || tally.length === 0) return "";
+ return tally.map((t) => `${styles?.[t.verdict]?.label ?? t.verdict} ${t.count}`).join(", ");
+}
+
+export function renderReportIndex(
+ label: string,
+ r: SourceReports,
+ origin: string | null,
+): string {
+ if (!r.supported) {
+ return (
+ `${label} has no reports of its own — reports are per site. Pass a ` +
+ `site's handle (remote:<site url>) as \`source\`; list_sources lists a ` +
+ `hub's members.`
+ );
+ }
+ if (r.reports.length === 0) {
+ return r.scope === "cited"
+ ? `${label} is a cited-only site, but its report index lists no reports.`
+ : `${label} publishes no reports (no /reports/index.json — none published, or a site built before corpus spec 5).`;
+ }
+ const head =
+ r.scope === "cited"
+ ? `${r.reports.length} report(s) in ${label} (cited-only site: no channels or transcripts to search):`
+ : `${r.reports.length} report(s) in ${label}:`;
+ const lines = r.reports.map((e) => {
+ const parts = [
+ e.kind,
+ `${e.claimCount} claim(s)`,
+ `${e.citationCount} citation(s)`,
+ ];
+ const dated = e.updated ?? e.published;
+ if (dated) parts.push(dated);
+ const tally = tallyText(e.tally, e.verdicts);
+ return (
+ `- ${e.id} · ${e.title}` +
+ (e.subtitle ? ` — ${e.subtitle}` : "") +
+ `\n ${parts.join(" · ")}` +
+ (tally ? `\n verdicts: ${tally}` : "") +
+ `\n ${absolute(origin, e.href)}`
+ );
+ });
+ return `${head}\n\n${lines.join("\n")}\n\n(get_report with report:"${r.reports[0].id}" for its sections, claims and citations.)`;
+}
+
+// The ids a claim cites, in reading order: its source sentence, the `cite:`
+// links in its findings, then its own list.
+function claimCitationIds(claim: ClaimView): string[] {
+ const ids: string[] = [];
+ const add = (id: string) => {
+ if (!ids.includes(id)) ids.push(id);
+ };
+ if (claim.sourceQuote) add(claim.sourceQuote);
+ for (const ref of extractCiteRefs(claim.findings)) add(ref.id);
+ for (const id of claim.citations) add(id);
+ return ids;
+}
+
+// One citation as an agent cites it: the verbatim quote, where it was said,
+// the original (the platform at the cited second, the post, the document) and
+// the site's own moment page.
+function citationLines(c: CitationView, origin: string | null): string[] {
+ const n = c.number !== undefined ? `[${c.number}]` : `[${c.id}]`;
+ const lines: string[] = [];
+ const quote = `"${c.quote}"`;
+ switch (c.kind) {
+ case "video":
+ case "audio": {
+ const where = [c.record.channelTitle ?? c.record.channel, c.record.title, c.record.date]
+ .filter(Boolean)
+ .join(" · ");
+ lines.push(`${n} ${c.kind} ${where} @ ${Math.floor(c.start)}–${Math.floor(c.end)} s`);
+ lines.push(` ${quote}` + (c.speaker ? ` — ${c.speaker}` : ""));
+ if (c.record.originalUrl) lines.push(` original: ${c.record.originalUrl}`);
+ lines.push(` moment: ${absolute(origin, c.href)}`);
+ break;
+ }
+ case "post": {
+ const where = [c.author ?? c.record.channelTitle ?? c.record.channel, c.record.date]
+ .filter(Boolean)
+ .join(" · ");
+ lines.push(`${n} post ${where}`);
+ lines.push(` ${quote}`);
+ if (c.record.originalUrl) lines.push(` original: ${c.record.originalUrl}`);
+ lines.push(` moment: ${absolute(origin, c.href)}`);
+ break;
+ }
+ case "source":
+ lines.push(`${n} source ${c.sourceTitle}`);
+ lines.push(` ${quote}`);
+ if (c.href) lines.push(` original: ${c.href}`);
+ break;
+ case "page":
+ lines.push(`${n} page ${c.title ?? c.href}`);
+ lines.push(` ${quote}`);
+ lines.push(` original: ${c.href}`);
+ if (c.archiveUrl) lines.push(` archived: ${c.archiveUrl}`);
+ break;
+ }
+ const score = c.verification?.quoteScore;
+ if (typeof score === "number") lines.push(` quote check: ${Math.round(score * 100)} %`);
+ return lines;
+}
+
+function renderClaim(view: ReportPageView, claim: ClaimView, origin: string | null): string {
+ const verdict = claim.verdict
+ ? `[${view.verdicts[claim.verdict]?.label ?? claim.verdict}] `
+ : "";
+ const out = [`- ${verdict}${claim.title ? `${claim.title}: ` : ""}${claim.text} (#${claim.id})`];
+ if (claim.findings) out.push(` findings: ${claim.findings.replace(/\s*\n\s*/g, " ")}`);
+ for (const id of claimCitationIds(claim)) {
+ const c = view.citations[id];
+ if (!c) continue;
+ for (const line of citationLines(c, origin)) out.push(` ${line}`);
+ }
+ return out.join("\n");
+}
+
+function renderSection(view: ReportPageView, s: SectionView, origin: string | null): string {
+ const out = [`## ${s.title} (#${s.id})`];
+ if (s.body) out.push(s.body.trim());
+ // Citations the section's prose cites, outside any claim.
+ const bodyIds = [...new Set(extractCiteRefs(s.body).map((r) => r.id))];
+ for (const id of bodyIds) {
+ const c = view.citations[id];
+ if (c) out.push(...citationLines(c, origin));
+ }
+ for (const claim of s.claims) out.push(renderClaim(view, claim, origin));
+ return out.join("\n");
+}
+
+export function renderReportPage(
+ view: ReportPageView,
+ origin: string | null,
+ sectionId?: string,
+): string {
+ const head = [`# ${view.title}`];
+ if (view.subtitle) head.push(view.subtitle);
+ const meta: string[] = [view.kind];
+ if (view.published) meta.push(`published ${view.published}`);
+ if (view.updated) meta.push(`updated ${view.updated}`);
+ meta.push(`${orderedCitations(view).length} citation(s)`);
+ head.push(meta.join(" · "));
+ head.push(`page: ${absolute(origin, reportPagePath(view.id))}`);
+ if (view.kind === "factcheck") {
+ const tally = tallyText(verdictTally(view), view.verdicts);
+ if (tally) head.push(`verdicts: ${tally}`);
+ }
+ const subject = view.subject ? view.sources[view.subject] : undefined;
+ if (subject) {
+ head.push(
+ `under review: ${subject.title}` +
+ (subject.publisher ? ` (${subject.publisher})` : "") +
+ (subject.url ? ` ${subject.url}` : ""),
+ );
+ }
+ if (view.summary && !sectionId) head.push("", view.summary.trim());
+
+ const sections = sectionId ? view.sections.filter((s) => s.id === sectionId) : view.sections;
+ const body = sections.map((s) => renderSection(view, s, origin));
+ const outline = sectionId
+ ? ""
+ : `\n\n(sections: ${view.sections.map((s) => s.id).join(", ")} — pass section:"<id>" for one)`;
+ return `${head.join("\n")}\n\n${body.join("\n\n")}${outline}`;
+}
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -63,6 +63,12 @@ import {
type PlanContext,
} from "./instructions";
import {
+ renderReportIndex,
+ renderReportPage,
+ reportsLine,
+ sourceReports,
+} from "./reports";
+import {
fetchClip,
renderFetchClip,
validateFetchClipArgs,
@@ -356,6 +362,41 @@ export const TOOLS: Tool[] = [
},
},
{
+ name: "list_reports",
+ description:
+ "List the cited reports a SITE publishes (corpus spec 5): each report's " +
+ "id, title, kind, claim and citation counts, a fact-check's verdict " +
+ "tally, and its page. A cited-only site publishes nothing else — no " +
+ "channels or transcripts to search. A hub has none of its own; pass a " +
+ "member's remote: handle.",
+ inputSchema: {
+ type: "object",
+ properties: { ...SOURCE_ARG },
+ additionalProperties: false,
+ },
+ },
+ {
+ name: "get_report",
+ description:
+ "Read one published report: title, kind, verdict tally, then each " +
+ "section's claims (verdict, findings) with every citation's verbatim " +
+ "quote, original URL (the platform at the cited second, the post, the " +
+ "document) and the site's moment page. Ids from list_reports.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ ...SOURCE_ARG,
+ report: { type: "string", description: "The report id." },
+ section: {
+ type: "string",
+ description: "Optional: one section's id, to read just that section.",
+ },
+ },
+ required: ["report"],
+ additionalProperties: false,
+ },
+ },
+ {
name: "search_transcripts",
description:
"Search the archive for a term or phrase. The corpus holds video " +
@@ -1187,6 +1228,10 @@ export function createServer(
return handleListChannels(source, args);
case "list_tags":
return handleListTags(source);
+ case "list_reports":
+ return handleListReports(source);
+ case "get_report":
+ return handleGetReport(source, args);
case "search_transcripts":
return handleSearch(source, args);
case "enumerate_matches":
@@ -1293,7 +1338,13 @@ async function handleListChannels(
args: Record<string, unknown>,
): Promise<ToolResult> {
const channels = await source.listChannels({ refresh: args.refresh === true });
- if (channels.length === 0) return text(`No channels found in ${source.label}.`);
+ if (channels.length === 0) {
+ // A cited-only site has no channels by design; say what it does publish.
+ const line = reportsLine(await sourceReports(source));
+ return text(
+ line ? `${source.label} — ${line}` : `No channels found in ${source.label}.`,
+ );
+ }
const { groups, defaultGroupId } = await source.loadGroups();
// Bucket channels by their resolved group id (browser semantics).
@@ -1393,6 +1444,46 @@ async function handleListTags(source: ShardSource): Promise<ToolResult> {
);
}
+// The reports a site publishes (spec 5). Read through sourceReports, so a hub
+// or a stub (no reports of its own) and a site with none each get a sentence.
+async function handleListReports(source: ShardSource): Promise<ToolResult> {
+ const reports = await sourceReports(source);
+ return text(renderReportIndex(source.label, reports, source.publicOrigin()));
+}
+
+// A local source learns its site's origin from corpus.json with the channel
+// list, so read that first: a moment link is then absolute, as it is remotely.
+async function reportOrigin(source: ShardSource): Promise<string | null> {
+ await source.listChannels().catch(() => []);
+ return source.publicOrigin();
+}
+
+async function handleGetReport(
+ source: ShardSource,
+ args: Record<string, unknown>,
+): Promise<ToolResult> {
+ const id = typeof args.report === "string" ? args.report.trim() : "";
+ if (!id) return errorText("get_report: `report` (a report id) is required — list_reports names them.");
+ if (typeof source.reportPage !== "function") {
+ return errorText(renderReportIndex(source.label, await sourceReports(source), null));
+ }
+ const view = await source.reportPage(id);
+ if (!view) {
+ const known = (await sourceReports(source)).reports.map((r) => r.id);
+ return errorText(
+ `report not found: ${id}` +
+ (known.length > 0 ? ` — this source publishes: ${known.join(", ")}` : " — this source publishes no reports"),
+ );
+ }
+ const section = typeof args.section === "string" && args.section.trim() ? args.section.trim() : undefined;
+ if (section && !view.sections.some((s) => s.id === section)) {
+ return errorText(
+ `report ${id} has no section ${section} — its sections: ${view.sections.map((s) => s.id).join(", ")}`,
+ );
+ }
+ return text(renderReportPage(view, await reportOrigin(source), section));
+}
+
// Every tag read goes through here so "this source cannot report tags at all"
// (an in-memory stub, which has no such concept) and "this source publishes
// none" land in the same place, as the same empty list.
@@ -2518,6 +2609,9 @@ async function handleListSources(
` default (used when a call omits source): ${registry.defaultHandle}`,
);
}
+ // A cited-only site would otherwise look like an empty corpus here.
+ const reports = reportsLine(await sourceReports(resolved.source));
+ if (reports) lines.push(` reports: ${reports}`);
const hubUrl =
resolved.spec.kind === "hub" ? resolved.spec.url : registry.hubUrl();
@@ -2588,6 +2682,8 @@ async function handleResolveSource(
` reachable: yes — ${channels.length} channel(s)` +
(groups.length > 0 ? `, ${groups.length} group(s)` : ""),
);
+ const reports = reportsLine(await sourceReports(resolved.source));
+ if (reports) lines.push(` reports: ${reports}`);
// Named here so a caller learns whether a `tags` filter is even available
// BEFORE it writes one and reads the empty result as an answer. The two
// states are different and both are said: some tags, or none at all.
diff --git a/mcp/src/source.ts b/mcp/src/source.ts
@@ -17,6 +17,7 @@ export type {
ChannelGroups,
ChannelRef,
ClusterMembership,
+ CorpusScope,
DuplicateIndex,
IndexedVideo,
VideoAvailability,
diff --git a/plans/report-sites.md b/plans/report-sites.md
@@ -131,8 +131,9 @@ Citations are the core; reports are one consumer.
`audience`, and `reports: { index: "/reports/index.json" }` — the private-deploy guard and compose's
cross-site check read them. `CONTRACT.corpusSpec` → 5; readers tolerate a cited scope (an old reader sees an
empty corpus, which is safe). `llms.txt` and the sitemap get a cited variant listing report routes.
-- Hub and homepage exclude `cited` sites for now; MCP may later gain `list_reports` / `get_report`.
-- umtool's cue walk (`cues.mjs`) cannot resolve cues off a cited site — documented.
+- Hub and homepage exclude `cited` sites (`isListedSite` folds them out, as a private site); MCP serves
+ `list_reports` / `get_report` and says "cited-only site: N report(s)" (R5).
+- umtool's cue walk (`cues.mjs`) cannot resolve cues off a cited site — it refuses one by name (R5).
## Converters (to make reports from what exists)
diff --git a/umtool/report-to-video/cues.mjs b/umtool/report-to-video/cues.mjs
@@ -208,6 +208,16 @@ export function createCueSource({
async function channelEntry(origin, channelSlug) {
const corpus = await getJson(`${origin}/corpus.json`);
+ // A cited report site (corpus spec 5, `site.scope: "cited"`) publishes its
+ // reports and the moments they cite — no channels, no transcript shards —
+ // so there are no cues to walk to. Say that, not "channel not found".
+ if (corpus.site?.scope === "cited") {
+ throw new CueLookupError(
+ `${origin} is a cited report site (corpus.json site.scope "cited"): it publishes no transcripts, ` +
+ `so cues cannot be read from it — point the manifest at a full archive or use a local corpus`,
+ { channelSlug, videoId: null, tried: [`${origin}/corpus.json`] },
+ );
+ }
const found = (corpus.channels ?? []).find((c) => c.slug === channelSlug);
if (!found) {
throw new CueLookupError(
diff --git a/umtool/report-to-video/cues.test.mjs b/umtool/report-to-video/cues.test.mjs
@@ -474,6 +474,17 @@ test("a channel absent from corpus.json is reported as such", async () => {
});
});
+test("a cited report site is named as such: it has no cues to read", async () => {
+ const cited = {
+ [`${ORIGIN}/corpus.json`]: { spec: 5, site: { scope: "cited" }, channels: [] },
+ };
+ await assert.rejects(() => source({ fetchImpl: stubFetch(cited) }).load("chan", "vid1"), (err) => {
+ assert.match(err.message, /cited report site/);
+ assert.match(err.message, /publishes no transcripts/);
+ return true;
+ });
+});
+
// --- reality check ----------------------------------------------------------
// Opt-in: `LIVE=1 pnpm test:scripts`. The stubs above encode assumptions about a