// Cited reports over MCP: list_reports / get_report, and the one sentence every // discovery tool says about a source's reports. // // A site may publish reports (corpus spec 5): /reports/index.json lists them // and /reports//page.json is one report with every citation resolved // (common/lib/report/views.ts — the export site's pages read the same files). // A CITED site publishes nothing else: no channels, no shards. Its empty // channel list is the site's shape, not a failure, and a tool that reported "no // channels" there would read as a broken corpus — so every discovery tool says // "cited-only site: N report(s)" instead. // // Rendering is pure (views in, text out); the reads are the reader's // (ArchiveReader.scope / reports / reportPage), optional because a hub and the // in-memory stubs have no reports of their own. import { extractCiteRefs } from "yt-dlp-transcript-common/lib/citations/inline"; import { orderedCitations, reportFullTitle, reportPagePath, verdictTally, type CitationView, type ClaimView, type EntryView, type ReportIndexEntry, type ReportPageView, type SectionView, } from "yt-dlp-transcript-common/lib/report/views"; import type { CorpusScope, ShardSource } from "./source"; // What a source says about its reports. `supported` is false for a source with // no reports of its own (a hub, a stub); a read that fails reads as none. export type SourceReports = { supported: boolean; scope: CorpusScope; reports: ReportIndexEntry[]; }; export async function sourceReports(source: ShardSource): Promise { if (typeof source.reports !== "function") { return { supported: false, scope: "full", reports: [] }; } let scope: CorpusScope = "full"; try { scope = typeof source.scope === "function" ? await source.scope() : "full"; } catch { // an unreadable corpus.json is reported by the tool that needs it } let reports: ReportIndexEntry[] = []; try { reports = await source.reports(); } catch { // a transport failure on the index: no reports this call can name } return { supported: true, scope, reports }; } // The one line a discovery tool adds, or null when there is nothing to say (a // full site with no reports, or a source with no reports of its own). export function reportsLine(r: SourceReports): string | null { if (r.scope === "cited") { return ( `cited-only site: ${r.reports.length} report(s) — it publishes its ` + `reports and the moments they cite, no channels or transcripts to ` + `search. list_reports / get_report read them.` ); } if (r.reports.length > 0) { return `${r.reports.length} report(s) published — list_reports / get_report read them.`; } return null; } // A site-root path made absolute against the site's origin, when one is known. function absolute(origin: string | null, href: string): string { if (!origin || /^https?:\/\//i.test(href)) return href; return `${origin.replace(/\/+$/, "")}${href.startsWith("/") ? href : `/${href}`}`; } function tallyText( tally: readonly { verdict: string; count: number }[] | undefined, styles?: Record, ): string { if (!tally || tally.length === 0) return ""; return tally.map((t) => `${styles?.[t.verdict]?.label ?? t.verdict} ${t.count}`).join(", "); } export function renderReportIndex( label: string, r: SourceReports, origin: string | null, ): string { if (!r.supported) { return ( `${label} has no reports of its own — reports are per site. Pass a ` + `site's handle (remote:) as \`source\`; list_sources lists a ` + `hub's members.` ); } if (r.reports.length === 0) { return r.scope === "cited" ? `${label} is a cited-only site, but its report index lists no reports.` : `${label} publishes no reports (no /reports/index.json — none published, or a site built before corpus spec 5).`; } const head = r.scope === "cited" ? `${r.reports.length} report(s) in ${label} (cited-only site: no channels or transcripts to search):` : `${r.reports.length} report(s) in ${label}:`; const lines = r.reports.map((e) => { const parts = [ e.kind, `${e.claimCount} claim(s)`, `${e.citationCount} citation(s)`, ]; const dated = e.updated ?? e.published; if (dated) parts.push(dated); const tally = tallyText(e.tally, e.verdicts); return ( `- ${e.id} · ${e.title}` + (e.subtitle ? ` — ${e.subtitle}` : "") + `\n ${parts.join(" · ")}` + (tally ? `\n verdicts: ${tally}` : "") + `\n ${absolute(origin, e.href)}` ); }); return `${head}\n\n${lines.join("\n")}\n\n(get_report with report:"${r.reports[0].id}" for its sections, claims and citations.)`; } // The ids a claim cites, in reading order: its source sentence, the `cite:` // links in its findings, then its own list. function claimCitationIds(claim: ClaimView): string[] { const ids: string[] = []; const add = (id: string) => { if (!ids.includes(id)) ids.push(id); }; if (claim.sourceQuote) add(claim.sourceQuote); for (const ref of extractCiteRefs(claim.findings)) add(ref.id); for (const id of claim.citations) add(id); return ids; } // One citation as an agent cites it: the verbatim quote, where it was said, // the original (the platform at the cited second, the post, the document) and // the site's own moment page. function citationLines(c: CitationView, origin: string | null): string[] { const n = c.number !== undefined ? `[${c.number}]` : `[${c.id}]`; const lines: string[] = []; const quote = `"${c.quote}"`; switch (c.kind) { case "video": case "audio": { const where = [c.record.channelTitle ?? c.record.channel, c.record.title, c.record.date] .filter(Boolean) .join(" · "); lines.push(`${n} ${c.kind} ${where} @ ${Math.floor(c.start)}–${Math.floor(c.end)} s`); lines.push(` ${quote}` + (c.speaker ? ` — ${c.speaker}` : "")); if (c.record.originalUrl) { lines.push(` original${c.record.originalLabel ? ` (${c.record.originalLabel})` : ""}: ${c.record.originalUrl}`); } for (const d of c.record.downloads ?? []) lines.push(` ${d.label}: ${d.url}`); lines.push(` moment: ${absolute(origin, c.href)}`); break; } case "post": { const where = [c.author ?? c.record.channelTitle ?? c.record.channel, c.record.date] .filter(Boolean) .join(" · "); lines.push(`${n} post ${where}`); lines.push(` ${quote}`); if (c.record.originalUrl) lines.push(` original: ${c.record.originalUrl}`); lines.push(` moment: ${absolute(origin, c.href)}`); break; } case "source": lines.push(`${n} source ${c.sourceTitle}`); lines.push(` ${quote}`); if (c.href) lines.push(` original: ${c.href}`); break; case "page": lines.push(`${n} page ${c.title ?? c.href}`); lines.push(` ${quote}`); lines.push(` original: ${c.href}`); if (c.archiveUrl) lines.push(` archived: ${c.archiveUrl}`); break; } const score = c.verification?.quoteScore; if (typeof score === "number") lines.push(` quote check: ${Math.round(score * 100)} %`); return lines; } function renderClaim(view: ReportPageView, claim: ClaimView, origin: string | null): string { const verdict = claim.verdict ? `[${view.verdicts[claim.verdict]?.label ?? claim.verdict}] ` : ""; const out = [`- ${verdict}${claim.title ? `${claim.title}: ` : ""}${claim.text} (#${claim.id})`]; if (claim.findings) out.push(` findings: ${claim.findings.replace(/\s*\n\s*/g, " ")}`); for (const id of claimCitationIds(claim)) { const c = view.citations[id]; if (!c) continue; for (const line of citationLines(c, origin)) out.push(` ${line}`); } return out.join("\n"); } function renderSection(view: ReportPageView, s: SectionView, origin: string | null): string { const out = [`## ${s.title} (#${s.id})`]; if (s.body) out.push(s.body.trim()); // Citations the section's prose cites, outside any claim. const bodyIds = [...new Set(extractCiteRefs(s.body).map((r) => r.id))]; for (const id of bodyIds) { const c = view.citations[id]; if (c) out.push(...citationLines(c, origin)); } for (const claim of s.claims) out.push(renderClaim(view, claim, origin)); return out.join("\n"); } // A timeline entry: its date, title and anchor, its body, and the citations // it cites. function renderEntry(view: ReportPageView, e: EntryView, origin: string | null): string { const out = [`### ${e.date}${e.updated ? ` (updated ${e.updated})` : ""} — ${e.title} (#${e.id})`]; out.push(e.body.trim()); for (const id of [...new Set(extractCiteRefs(e.body).map((r) => r.id))]) { const c = view.citations[id]; if (c) out.push(...citationLines(c, origin)); } return out.join("\n"); } // One report as text. `sectionId` narrows it to one section — or one // timeline entry, by its id. export function renderReportPage( view: ReportPageView, origin: string | null, sectionId?: string, ): string { const head = [`# ${reportFullTitle(view)}`]; if (view.subtitle) head.push(view.subtitle); const meta: string[] = [view.kind]; if (view.published) meta.push(`published ${view.published}`); if (view.updated) meta.push(`updated ${view.updated}`); meta.push(`${orderedCitations(view).length} citation(s)`); head.push(meta.join(" · ")); head.push(`page: ${absolute(origin, reportPagePath(view.id))}`); if (view.kind === "factcheck") { const tally = tallyText(verdictTally(view), view.verdicts); if (tally) head.push(`verdicts: ${tally}`); } const subject = view.subject ? view.sources[view.subject] : undefined; if (subject) { head.push( `under review: ${subject.title}` + (subject.publisher ? ` (${subject.publisher})` : "") + (subject.url ? ` ${subject.url}` : ""), ); } if (view.summary && !sectionId) head.push("", view.summary.trim()); // The timeline, newest first, before the sections (as the page has it). const allEntries = view.entries ?? []; const entries = sectionId ? allEntries.filter((e) => e.id === sectionId) : allEntries; const body: string[] = []; if (entries.length > 0) { body.push([`## Timeline (newest first)`, ...entries.map((e) => renderEntry(view, e, origin))].join("\n\n")); } const sections = sectionId ? view.sections.filter((s) => s.id === sectionId) : view.sections; body.push(...sections.map((s) => renderSection(view, s, origin))); const outline = sectionId ? "" : `\n\n(sections: ${view.sections.map((s) => s.id).join(", ")}` + (allEntries.length > 0 ? `; timeline entries: ${allEntries.map((e) => e.id).join(", ")}` : "") + ` — pass section:"" for one)`; return `${head.join("\n")}\n\n${body.join("\n\n")}${outline}`; }