commit 06209852b27ec8008b311e4cf8a0e3fc34721d47
parent d2aef78ce52ee9157972f5b4a2e8cd6e49c8c9bb
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 05:21:13 -0400
mcp: list_reports and get_report read a site's cited reports; list_channels, list_sources and resolve_source say "cited-only site: N report(s)" rather than an empty corpus
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
6 files changed, 588 insertions(+), 2 deletions(-)
diff --git a/mcp/README.md b/mcp/README.md
@@ -15,6 +15,7 @@ clip window; the MCP itself still writes nothing.
| Tool | What it does |
|------|--------------|
| `list_channels` | List channels **organized under their channel groups** (name, slug, video count; site in hub mode), with a compact group cheat-sheet (`id · name · N channels`) for scoping. |
+| `list_reports` / `get_report` | The cited reports a site publishes (corpus spec 5): the index (title, kind, claim and citation counts, a fact-check's verdict tally), then one report's sections and claims with every citation's **verbatim quote, original URL and the site's moment page**. See [Cited reports](#cited-reports). |
| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets — **each `[mm:ss]` is a clickable link to that exact moment** (or a compact `[mm:ss\|sec]` with `link_style:"base"`). Alias-aware, pageable, and **filterable** (`states`, `date_from`/`date_to`, `media_type`, `age`, `exclude`, `scopes`). A page that isn't the whole match set is flagged **above** the hits. |
| `enumerate_matches` | A query's **complete** match set as a worklist (id/title/channel/date + batch count) in **one scan**. Takes the **same filters** as `search_transcripts`, so the two can never disagree about coverage. The tool to use whenever you need to count or cover everything. |
| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around one query or up to 8 (`queries`), with per-query counts; or full transcripts without a query. Reads posts too. |
@@ -23,7 +24,7 @@ clip window; the MCP itself still writes nothing.
| `get_video_metadata` | Everything known about one video without the transcript body: metadata, plus **view/like counts, cue count and transcript coverage** (`stats/`), **other archived copies of the same recording** with an explicit timings-aligned verdict (`duplicates.json`), and **AI chapters/tags** where they exist (`digests/`). |
| `fetch_clip` | The media behind a cited moment, **fetched by the local editor** (`POST /api/media/fetch-window`) through its paced, cookie-aware, provenanced job — never a yt-dlp run by hand. Needs `ARCHILYZER_EDITOR_URL` (default `http://localhost:3001`) and `WORKER_TOKEN` (the editor's own) in this server's env; without them it says so and fetches nothing. The editor must already archive the cited channel (a channel dir under its `transcripts/`), else it answers 404 `Channel "<slug>" not found`: an MCP pointed at a public site with a fresh editor gets that on every clip. A window is the cited span ± `pad` (default 3 s), at most 15 min, and lands at `channels/<slug>/data/<id>/clips/`; `full: true` fetches the whole recording into the saved-video store (needs a video the editor already knows). `maxHeight` (144–2160) caps the source height: a window is fetched at or under it (default 720); a whole recording at 720 or less is saved as the editor's 720p H.264 preset and above 720 at the original quality (omitted, the channel's source-video quality applies). A file already on disk is returned as it is, never re-fetched for a different cap, and the answer gives its height and says when it is taller than asked. Waits up to `wait_seconds` (default 90, max 300), then returns the job id to resume with `job`; a client with a 60 s default request timeout must raise it or pass `wait_seconds` ≤ 50 — the fetch continues on the editor either way; resume it with `job`, and once it has finished the same request finds it cached. While it waits it sends one progress notification per poll to a client that asked for progress (a `progressToken`), which keeps a reset-on-progress timeout alive. A Rumble embed id is mapped to the editor's slug id through the record's `webpageUrl`, so pass the citing corpus as `source`; a video not in `source` is passed through as cited (known limitation). The file is a read-only corpus artifact. |
| `open_link` | Paste an archilyzer viewer **share link** to re-run that exact search here (query tree + every filter, at full fidelity) — plan, results and corpus handle in **one** call. `dry_run:true` for the plan alone. |
-| `list_sources` | Show the **default** corpus and, with a hub, its member sites as ready-to-paste handles. |
+| `list_sources` | Show the **default** corpus and, with a hub, its member sites as ready-to-paste handles. A site with reports says how many; a cited-only site says it is one. |
| `resolve_source` | Turn a URL or site name into the canonical `source` handle and check it can be read. Changes nothing. |
| `sweep_plan` / `ask_plan` | Turn a plain-English request (plus an optional pasted link) into a resolved, step-by-step plan. What `/sweep` and `/ask` call. |
@@ -320,6 +321,22 @@ so a translated citation would look perfectly plausible and point at the wrong
moment of a different upload. Absent means *not measured*, and not measured means
*no*.
+## Cited reports
+
+A site may publish cited reports (corpus spec 5): `/reports/index.json` lists them
+and `/reports/<id>/page.json` is one report with every citation resolved — the
+same files the site's Reports pages read. `list_reports` reads the index;
+`get_report` (`report`, optionally one `section`) renders a report as text: its
+verdict tally, then each claim with its verdict, its findings and, per citation,
+the quote, the original (the platform at the cited second, the post, the
+document) and the site's moment page (`/m/<channel>/<id>/<start>-<end>/`).
+
+A **cited-only** site (`corpus.json` `site.scope: "cited"`) publishes its reports
+and the moments they cite and nothing else: no channels, no transcripts, nothing
+to search. `list_channels`, `list_sources` and `resolve_source` say "cited-only
+site: N report(s)" there rather than reporting an empty corpus. Reports are per
+site: a hub has none of its own, so name a member with `source:"remote:<url>"`.
+
## How it reads the corpus
Three changes, in increasing order of how much they buy:
diff --git a/mcp/src/protocol.test.ts b/mcp/src/protocol.test.ts
@@ -35,6 +35,8 @@ const TSX = path.join(HERE, "..", "node_modules", ".bin", "tsx");
const EXPECTED_TOOLS = [
"list_channels",
"list_tags",
+ "list_reports",
+ "get_report",
"search_transcripts",
"enumerate_matches",
"get_transcript",
diff --git a/mcp/src/reports.test.ts b/mcp/src/reports.test.ts
@@ -0,0 +1,224 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, mkdir, rm, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { Client, InMemoryTransport } from "@modelcontextprotocol/client";
+import type { Report } from "yt-dlp-transcript-common/lib/report/schema";
+import {
+ REPORT_INDEX_FORMAT,
+ REPORT_VIEWS_VERSION,
+ buildReportPageView,
+ reportIndexEntry,
+} from "yt-dlp-transcript-common/lib/report/views";
+import { CONTRACT } from "yt-dlp-transcript-common/lib/archive/contract";
+import { LocalSource, type ShardSource } from "./source";
+import { createServer } from "./server";
+import { renderReportIndex, reportsLine } from "./reports";
+
+// list_reports / get_report and the "cited-only site" line, over a composed
+// public dir on disk read by the real LocalSource.
+
+const ORIGIN = "https://reports.example";
+
+const REPORT: Report = {
+ format: "archilyzer-report",
+ version: 1,
+ id: "demo-report",
+ kind: "factcheck",
+ title: "A demo fact-check",
+ subtitle: "Of an article",
+ summary: "It starts [here](cite:v1).",
+ published: "2026-10-01",
+ subject: { source: "s0" },
+ sources: {
+ s0: { kind: "article", title: "An article", url: "https://example.org/a", publisher: "Example" },
+ },
+ citations: {
+ v1: { kind: "video", channel: "demo-channel", id: "abc123", start: 61, end: 75.5, quote: "the cited words" },
+ p1: { kind: "post", channel: "demo-social", id: "123", quote: "a posted line" },
+ s1: { kind: "source", source: "s0", quote: "the article's sentence" },
+ },
+ sections: [
+ {
+ id: "one",
+ title: "Section one",
+ claims: [
+ {
+ id: "c1",
+ text: "The article claims a thing.",
+ verdict: "CONTRADICTED",
+ sourceQuote: { citation: "s1" },
+ findings: "The recording says otherwise [here](cite:v1).",
+ citations: ["p1"],
+ },
+ ],
+ },
+ { id: "two", title: "Section two", body: "Nothing to check.", claims: [] },
+ ],
+};
+
+const VIEW = buildReportPageView(REPORT, {
+ record: (c) => ({
+ channel: c.channel,
+ channelTitle: "Demo Channel",
+ id: c.id,
+ title: `Recording ${c.id}`,
+ date: "2026-01-02",
+ originalUrl:
+ c.kind === "post" ? `https://social.example/${c.id}` : `https://video.example/watch?v=${c.id}&t=61`,
+ }),
+ post: () => ({ author: "@demo", text: "a posted line, and more" }),
+});
+
+const INDEX = {
+ format: REPORT_INDEX_FORMAT,
+ version: REPORT_VIEWS_VERSION,
+ reports: [reportIndexEntry(VIEW)],
+};
+
+async function writeSite(opts: { cited: boolean; reports: boolean }): Promise<string> {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "mcp-reports-"));
+ const files: Record<string, unknown> = {
+ "corpus.json": {
+ spec: CONTRACT.corpusSpec,
+ kind: "site",
+ site: {
+ id: "demo-site",
+ title: "Demo",
+ url: ORIGIN,
+ ...(opts.cited ? { scope: "cited", audience: "public" } : {}),
+ },
+ channels: opts.cited ? [] : [{ slug: "demo-channel", name: "Demo Channel", videoCount: 1 }],
+ ...(opts.reports ? { reports: { index: `${ORIGIN}/reports/index.json`, count: 1 } } : {}),
+ },
+ };
+ if (opts.reports) {
+ files["reports/index.json"] = INDEX;
+ files["reports/demo-report/page.json"] = VIEW;
+ }
+ for (const [rel, body] of Object.entries(files)) {
+ const file = path.join(dir, rel);
+ await mkdir(path.dirname(file), { recursive: true });
+ await writeFile(file, JSON.stringify(body));
+ }
+ return dir;
+}
+
+async function connect(source: ShardSource): Promise<Client> {
+ const server = createServer(source);
+ const [ct, st] = InMemoryTransport.createLinkedPair();
+ const client = new Client({ name: "test", version: "0" }, { capabilities: {} });
+ await Promise.all([server.connect(st), client.connect(ct)]);
+ return client;
+}
+
+async function call(client: Client, name: string, args: Record<string, unknown> = {}) {
+ const res = (await client.callTool({ name, arguments: args })) as {
+ content: { text: string }[];
+ isError?: boolean;
+ };
+ return { text: res.content.map((c) => c.text).join("\n"), isError: res.isError === true };
+}
+
+test("a cited site: discovery tools say cited-only with its report count, not an empty corpus", async () => {
+ const dir = await writeSite({ cited: true, reports: true });
+ const client = await connect(new LocalSource(dir));
+ try {
+ const channels = await call(client, "list_channels");
+ assert.equal(channels.isError, false);
+ assert.match(channels.text, /cited-only site: 1 report\(s\)/);
+ assert.doesNotMatch(channels.text, /No channels found/);
+
+ assert.match((await call(client, "list_sources")).text, /reports: cited-only site: 1 report\(s\)/);
+ assert.match((await call(client, "resolve_source", { source: "default" })).text, /reports: cited-only site/);
+ } finally {
+ await client.close();
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("list_reports names each report with its counts, tally and page", async () => {
+ const dir = await writeSite({ cited: true, reports: true });
+ const client = await connect(new LocalSource(dir));
+ try {
+ const out = (await call(client, "list_reports")).text;
+ assert.match(out, /1 report\(s\) in local:.*cited-only site/);
+ assert.match(out, /- demo-report · A demo fact-check — Of an article/);
+ assert.match(out, /factcheck · 1 claim\(s\) · 3 citation\(s\) · 2026-10-01/);
+ assert.match(out, /verdicts: Contradicted 1/);
+ assert.match(out, new RegExp(`${ORIGIN}/reports/demo-report/`));
+ } finally {
+ await client.close();
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("get_report: claims with their citations' quote, original and moment URLs", async () => {
+ const dir = await writeSite({ cited: true, reports: true });
+ const client = await connect(new LocalSource(dir));
+ try {
+ const { text, isError } = await call(client, "get_report", { report: "demo-report" });
+ assert.equal(isError, false);
+ assert.match(text, /# A demo fact-check/);
+ assert.match(text, /verdicts: Contradicted 1/);
+ assert.match(text, /under review: An article \(Example\) https:\/\/example\.org\/a/);
+ assert.match(text, /\[Contradicted\] The article claims a thing\. \(#c1\)/);
+ assert.match(text, /findings: The recording says otherwise/);
+ // The source sentence, the span the findings cite, then the claim's post.
+ const at = (s: string) => text.indexOf(s);
+ assert.ok(at('"the article\'s sentence"') < at('"the cited words"'));
+ assert.ok(at('"the cited words"') < at('"a posted line"'));
+ assert.match(text, /video Demo Channel · Recording abc123 · 2026-01-02 @ 61–75 s/);
+ assert.match(text, /original: https:\/\/video\.example\/watch\?v=abc123&t=61/);
+ assert.match(text, new RegExp(`moment: ${ORIGIN}/m/demo-channel/abc123/61\\.00-75\\.50/`));
+ assert.match(text, /original: https:\/\/social\.example\/123/);
+ assert.match(text, new RegExp(`moment: ${ORIGIN}/m/demo-social/123/`));
+ assert.match(text, new RegExp(`page: ${ORIGIN}/reports/demo-report/`));
+ assert.match(text, /## Section two \(#two\)/);
+
+ const one = (await call(client, "get_report", { report: "demo-report", section: "two" })).text;
+ assert.match(one, /## Section two/);
+ assert.doesNotMatch(one, /Section one/);
+
+ const badSection = await call(client, "get_report", { report: "demo-report", section: "nope" });
+ assert.equal(badSection.isError, true);
+ assert.match(badSection.text, /no section nope — its sections: one, two/);
+
+ const missing = await call(client, "get_report", { report: "nope" });
+ assert.equal(missing.isError, true);
+ assert.match(missing.text, /report not found: nope — this source publishes: demo-report/);
+ } finally {
+ await client.close();
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("a full site: channels as before, and its reports are counted where it has any", async () => {
+ const withReports = await writeSite({ cited: false, reports: true });
+ const without = await writeSite({ cited: false, reports: false });
+ const a = await connect(new LocalSource(withReports));
+ const b = await connect(new LocalSource(without));
+ try {
+ const channels = (await call(a, "list_channels")).text;
+ assert.match(channels, /1 channel\(s\)/);
+ assert.match((await call(a, "list_sources")).text, /reports: 1 report\(s\) published/);
+
+ assert.doesNotMatch((await call(b, "list_sources")).text, /reports:/);
+ assert.match((await call(b, "list_reports")).text, /publishes no reports/);
+ const missing = await call(b, "get_report", { report: "demo-report" });
+ assert.equal(missing.isError, true);
+ assert.match(missing.text, /this source publishes no reports/);
+ } finally {
+ await a.close();
+ await b.close();
+ await rm(withReports, { recursive: true, force: true });
+ await rm(without, { recursive: true, force: true });
+ }
+});
+
+test("a source with no reports of its own (a hub) points at its members", () => {
+ const out = renderReportIndex("hub:https://hub.example", { supported: false, scope: "full", reports: [] }, null);
+ assert.match(out, /reports are per site/);
+ assert.equal(reportsLine({ supported: false, scope: "full", reports: [] }), null);
+});
diff --git a/mcp/src/reports.ts b/mcp/src/reports.ts
@@ -0,0 +1,246 @@
+// Cited reports over MCP: list_reports / get_report, and the one sentence every
+// discovery tool says about a source's reports.
+//
+// A site may publish reports (corpus spec 5): /reports/index.json lists them
+// and /reports/<id>/page.json is one report with every citation resolved
+// (common/lib/report/views.ts — the export site's pages read the same files).
+// A CITED site publishes nothing else: no channels, no shards. Its empty
+// channel list is the site's shape, not a failure, and a tool that reported "no
+// channels" there would read as a broken corpus — so every discovery tool says
+// "cited-only site: N report(s)" instead.
+//
+// Rendering is pure (views in, text out); the reads are the reader's
+// (ArchiveReader.scope / reports / reportPage), optional because a hub and the
+// in-memory stubs have no reports of their own.
+
+import { extractCiteRefs } from "yt-dlp-transcript-common/lib/citations/inline";
+import {
+ orderedCitations,
+ reportPagePath,
+ verdictTally,
+ type CitationView,
+ type ClaimView,
+ type ReportIndexEntry,
+ type ReportPageView,
+ type SectionView,
+} from "yt-dlp-transcript-common/lib/report/views";
+import type { CorpusScope, ShardSource } from "./source";
+
+// What a source says about its reports. `supported` is false for a source with
+// no reports of its own (a hub, a stub); a read that fails reads as none.
+export type SourceReports = {
+ supported: boolean;
+ scope: CorpusScope;
+ reports: ReportIndexEntry[];
+};
+
+export async function sourceReports(source: ShardSource): Promise<SourceReports> {
+ if (typeof source.reports !== "function") {
+ return { supported: false, scope: "full", reports: [] };
+ }
+ let scope: CorpusScope = "full";
+ try {
+ scope = typeof source.scope === "function" ? await source.scope() : "full";
+ } catch {
+ // an unreadable corpus.json is reported by the tool that needs it
+ }
+ let reports: ReportIndexEntry[] = [];
+ try {
+ reports = await source.reports();
+ } catch {
+ // a transport failure on the index: no reports this call can name
+ }
+ return { supported: true, scope, reports };
+}
+
+// The one line a discovery tool adds, or null when there is nothing to say (a
+// full site with no reports, or a source with no reports of its own).
+export function reportsLine(r: SourceReports): string | null {
+ if (r.scope === "cited") {
+ return (
+ `cited-only site: ${r.reports.length} report(s) — it publishes its ` +
+ `reports and the moments they cite, no channels or transcripts to ` +
+ `search. list_reports / get_report read them.`
+ );
+ }
+ if (r.reports.length > 0) {
+ return `${r.reports.length} report(s) published — list_reports / get_report read them.`;
+ }
+ return null;
+}
+
+// A site-root path made absolute against the site's origin, when one is known.
+function absolute(origin: string | null, href: string): string {
+ if (!origin || /^https?:\/\//i.test(href)) return href;
+ return `${origin.replace(/\/+$/, "")}${href.startsWith("/") ? href : `/${href}`}`;
+}
+
+function tallyText(
+ tally: readonly { verdict: string; count: number }[] | undefined,
+ styles?: Record<string, { label: string }>,
+): string {
+ if (!tally || tally.length === 0) return "";
+ return tally.map((t) => `${styles?.[t.verdict]?.label ?? t.verdict} ${t.count}`).join(", ");
+}
+
+export function renderReportIndex(
+ label: string,
+ r: SourceReports,
+ origin: string | null,
+): string {
+ if (!r.supported) {
+ return (
+ `${label} has no reports of its own — reports are per site. Pass a ` +
+ `site's handle (remote:<site url>) as \`source\`; list_sources lists a ` +
+ `hub's members.`
+ );
+ }
+ if (r.reports.length === 0) {
+ return r.scope === "cited"
+ ? `${label} is a cited-only site, but its report index lists no reports.`
+ : `${label} publishes no reports (no /reports/index.json — none published, or a site built before corpus spec 5).`;
+ }
+ const head =
+ r.scope === "cited"
+ ? `${r.reports.length} report(s) in ${label} (cited-only site: no channels or transcripts to search):`
+ : `${r.reports.length} report(s) in ${label}:`;
+ const lines = r.reports.map((e) => {
+ const parts = [
+ e.kind,
+ `${e.claimCount} claim(s)`,
+ `${e.citationCount} citation(s)`,
+ ];
+ const dated = e.updated ?? e.published;
+ if (dated) parts.push(dated);
+ const tally = tallyText(e.tally, e.verdicts);
+ return (
+ `- ${e.id} · ${e.title}` +
+ (e.subtitle ? ` — ${e.subtitle}` : "") +
+ `\n ${parts.join(" · ")}` +
+ (tally ? `\n verdicts: ${tally}` : "") +
+ `\n ${absolute(origin, e.href)}`
+ );
+ });
+ return `${head}\n\n${lines.join("\n")}\n\n(get_report with report:"${r.reports[0].id}" for its sections, claims and citations.)`;
+}
+
+// The ids a claim cites, in reading order: its source sentence, the `cite:`
+// links in its findings, then its own list.
+function claimCitationIds(claim: ClaimView): string[] {
+ const ids: string[] = [];
+ const add = (id: string) => {
+ if (!ids.includes(id)) ids.push(id);
+ };
+ if (claim.sourceQuote) add(claim.sourceQuote);
+ for (const ref of extractCiteRefs(claim.findings)) add(ref.id);
+ for (const id of claim.citations) add(id);
+ return ids;
+}
+
+// One citation as an agent cites it: the verbatim quote, where it was said,
+// the original (the platform at the cited second, the post, the document) and
+// the site's own moment page.
+function citationLines(c: CitationView, origin: string | null): string[] {
+ const n = c.number !== undefined ? `[${c.number}]` : `[${c.id}]`;
+ const lines: string[] = [];
+ const quote = `"${c.quote}"`;
+ switch (c.kind) {
+ case "video":
+ case "audio": {
+ const where = [c.record.channelTitle ?? c.record.channel, c.record.title, c.record.date]
+ .filter(Boolean)
+ .join(" · ");
+ lines.push(`${n} ${c.kind} ${where} @ ${Math.floor(c.start)}–${Math.floor(c.end)} s`);
+ lines.push(` ${quote}` + (c.speaker ? ` — ${c.speaker}` : ""));
+ if (c.record.originalUrl) lines.push(` original: ${c.record.originalUrl}`);
+ lines.push(` moment: ${absolute(origin, c.href)}`);
+ break;
+ }
+ case "post": {
+ const where = [c.author ?? c.record.channelTitle ?? c.record.channel, c.record.date]
+ .filter(Boolean)
+ .join(" · ");
+ lines.push(`${n} post ${where}`);
+ lines.push(` ${quote}`);
+ if (c.record.originalUrl) lines.push(` original: ${c.record.originalUrl}`);
+ lines.push(` moment: ${absolute(origin, c.href)}`);
+ break;
+ }
+ case "source":
+ lines.push(`${n} source ${c.sourceTitle}`);
+ lines.push(` ${quote}`);
+ if (c.href) lines.push(` original: ${c.href}`);
+ break;
+ case "page":
+ lines.push(`${n} page ${c.title ?? c.href}`);
+ lines.push(` ${quote}`);
+ lines.push(` original: ${c.href}`);
+ if (c.archiveUrl) lines.push(` archived: ${c.archiveUrl}`);
+ break;
+ }
+ const score = c.verification?.quoteScore;
+ if (typeof score === "number") lines.push(` quote check: ${Math.round(score * 100)} %`);
+ return lines;
+}
+
+function renderClaim(view: ReportPageView, claim: ClaimView, origin: string | null): string {
+ const verdict = claim.verdict
+ ? `[${view.verdicts[claim.verdict]?.label ?? claim.verdict}] `
+ : "";
+ const out = [`- ${verdict}${claim.title ? `${claim.title}: ` : ""}${claim.text} (#${claim.id})`];
+ if (claim.findings) out.push(` findings: ${claim.findings.replace(/\s*\n\s*/g, " ")}`);
+ for (const id of claimCitationIds(claim)) {
+ const c = view.citations[id];
+ if (!c) continue;
+ for (const line of citationLines(c, origin)) out.push(` ${line}`);
+ }
+ return out.join("\n");
+}
+
+function renderSection(view: ReportPageView, s: SectionView, origin: string | null): string {
+ const out = [`## ${s.title} (#${s.id})`];
+ if (s.body) out.push(s.body.trim());
+ // Citations the section's prose cites, outside any claim.
+ const bodyIds = [...new Set(extractCiteRefs(s.body).map((r) => r.id))];
+ for (const id of bodyIds) {
+ const c = view.citations[id];
+ if (c) out.push(...citationLines(c, origin));
+ }
+ for (const claim of s.claims) out.push(renderClaim(view, claim, origin));
+ return out.join("\n");
+}
+
+export function renderReportPage(
+ view: ReportPageView,
+ origin: string | null,
+ sectionId?: string,
+): string {
+ const head = [`# ${view.title}`];
+ if (view.subtitle) head.push(view.subtitle);
+ const meta: string[] = [view.kind];
+ if (view.published) meta.push(`published ${view.published}`);
+ if (view.updated) meta.push(`updated ${view.updated}`);
+ meta.push(`${orderedCitations(view).length} citation(s)`);
+ head.push(meta.join(" · "));
+ head.push(`page: ${absolute(origin, reportPagePath(view.id))}`);
+ if (view.kind === "factcheck") {
+ const tally = tallyText(verdictTally(view), view.verdicts);
+ if (tally) head.push(`verdicts: ${tally}`);
+ }
+ const subject = view.subject ? view.sources[view.subject] : undefined;
+ if (subject) {
+ head.push(
+ `under review: ${subject.title}` +
+ (subject.publisher ? ` (${subject.publisher})` : "") +
+ (subject.url ? ` ${subject.url}` : ""),
+ );
+ }
+ if (view.summary && !sectionId) head.push("", view.summary.trim());
+
+ const sections = sectionId ? view.sections.filter((s) => s.id === sectionId) : view.sections;
+ const body = sections.map((s) => renderSection(view, s, origin));
+ const outline = sectionId
+ ? ""
+ : `\n\n(sections: ${view.sections.map((s) => s.id).join(", ")} — pass section:"<id>" for one)`;
+ return `${head.join("\n")}\n\n${body.join("\n\n")}${outline}`;
+}
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -63,6 +63,12 @@ import {
type PlanContext,
} from "./instructions";
import {
+ renderReportIndex,
+ renderReportPage,
+ reportsLine,
+ sourceReports,
+} from "./reports";
+import {
fetchClip,
renderFetchClip,
validateFetchClipArgs,
@@ -356,6 +362,41 @@ export const TOOLS: Tool[] = [
},
},
{
+ name: "list_reports",
+ description:
+ "List the cited reports a SITE publishes (corpus spec 5): each report's " +
+ "id, title, kind, claim and citation counts, a fact-check's verdict " +
+ "tally, and its page. A cited-only site publishes nothing else — no " +
+ "channels or transcripts to search. A hub has none of its own; pass a " +
+ "member's remote: handle.",
+ inputSchema: {
+ type: "object",
+ properties: { ...SOURCE_ARG },
+ additionalProperties: false,
+ },
+ },
+ {
+ name: "get_report",
+ description:
+ "Read one published report: title, kind, verdict tally, then each " +
+ "section's claims (verdict, findings) with every citation's verbatim " +
+ "quote, original URL (the platform at the cited second, the post, the " +
+ "document) and the site's moment page. Ids from list_reports.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ ...SOURCE_ARG,
+ report: { type: "string", description: "The report id." },
+ section: {
+ type: "string",
+ description: "Optional: one section's id, to read just that section.",
+ },
+ },
+ required: ["report"],
+ additionalProperties: false,
+ },
+ },
+ {
name: "search_transcripts",
description:
"Search the archive for a term or phrase. The corpus holds video " +
@@ -1187,6 +1228,10 @@ export function createServer(
return handleListChannels(source, args);
case "list_tags":
return handleListTags(source);
+ case "list_reports":
+ return handleListReports(source);
+ case "get_report":
+ return handleGetReport(source, args);
case "search_transcripts":
return handleSearch(source, args);
case "enumerate_matches":
@@ -1293,7 +1338,13 @@ async function handleListChannels(
args: Record<string, unknown>,
): Promise<ToolResult> {
const channels = await source.listChannels({ refresh: args.refresh === true });
- if (channels.length === 0) return text(`No channels found in ${source.label}.`);
+ if (channels.length === 0) {
+ // A cited-only site has no channels by design; say what it does publish.
+ const line = reportsLine(await sourceReports(source));
+ return text(
+ line ? `${source.label} — ${line}` : `No channels found in ${source.label}.`,
+ );
+ }
const { groups, defaultGroupId } = await source.loadGroups();
// Bucket channels by their resolved group id (browser semantics).
@@ -1393,6 +1444,46 @@ async function handleListTags(source: ShardSource): Promise<ToolResult> {
);
}
+// The reports a site publishes (spec 5). Read through sourceReports, so a hub
+// or a stub (no reports of its own) and a site with none each get a sentence.
+async function handleListReports(source: ShardSource): Promise<ToolResult> {
+ const reports = await sourceReports(source);
+ return text(renderReportIndex(source.label, reports, source.publicOrigin()));
+}
+
+// A local source learns its site's origin from corpus.json with the channel
+// list, so read that first: a moment link is then absolute, as it is remotely.
+async function reportOrigin(source: ShardSource): Promise<string | null> {
+ await source.listChannels().catch(() => []);
+ return source.publicOrigin();
+}
+
+async function handleGetReport(
+ source: ShardSource,
+ args: Record<string, unknown>,
+): Promise<ToolResult> {
+ const id = typeof args.report === "string" ? args.report.trim() : "";
+ if (!id) return errorText("get_report: `report` (a report id) is required — list_reports names them.");
+ if (typeof source.reportPage !== "function") {
+ return errorText(renderReportIndex(source.label, await sourceReports(source), null));
+ }
+ const view = await source.reportPage(id);
+ if (!view) {
+ const known = (await sourceReports(source)).reports.map((r) => r.id);
+ return errorText(
+ `report not found: ${id}` +
+ (known.length > 0 ? ` — this source publishes: ${known.join(", ")}` : " — this source publishes no reports"),
+ );
+ }
+ const section = typeof args.section === "string" && args.section.trim() ? args.section.trim() : undefined;
+ if (section && !view.sections.some((s) => s.id === section)) {
+ return errorText(
+ `report ${id} has no section ${section} — its sections: ${view.sections.map((s) => s.id).join(", ")}`,
+ );
+ }
+ return text(renderReportPage(view, await reportOrigin(source), section));
+}
+
// Every tag read goes through here so "this source cannot report tags at all"
// (an in-memory stub, which has no such concept) and "this source publishes
// none" land in the same place, as the same empty list.
@@ -2518,6 +2609,9 @@ async function handleListSources(
` default (used when a call omits source): ${registry.defaultHandle}`,
);
}
+ // A cited-only site would otherwise look like an empty corpus here.
+ const reports = reportsLine(await sourceReports(resolved.source));
+ if (reports) lines.push(` reports: ${reports}`);
const hubUrl =
resolved.spec.kind === "hub" ? resolved.spec.url : registry.hubUrl();
@@ -2588,6 +2682,8 @@ async function handleResolveSource(
` reachable: yes — ${channels.length} channel(s)` +
(groups.length > 0 ? `, ${groups.length} group(s)` : ""),
);
+ const reports = reportsLine(await sourceReports(resolved.source));
+ if (reports) lines.push(` reports: ${reports}`);
// Named here so a caller learns whether a `tags` filter is even available
// BEFORE it writes one and reads the empty result as an answer. The two
// states are different and both are said: some tags, or none at all.
diff --git a/mcp/src/source.ts b/mcp/src/source.ts
@@ -17,6 +17,7 @@ export type {
ChannelGroups,
ChannelRef,
ClusterMembership,
+ CorpusScope,
DuplicateIndex,
IndexedVideo,
VideoAvailability,