commit d2aef78ce52ee9157972f5b4a2e8cd6e49c8c9bb
parent 1362716c5c8b15293d2542cf549602f5c2c151ec
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 05:17:53 -0400
archive readers: a cited corpus (spec 5) reads as an empty corpus and names its reports — scope(), reports(), reportPage() on the local and remote readers; a cited local dir never falls back to a transcripts/ listing
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
3 files changed, 403 insertions(+), 1 deletion(-)
diff --git a/common/lib/archive/reader-fs.ts b/common/lib/archive/reader-fs.ts
@@ -29,6 +29,13 @@ import { DUPLICATES_FILENAME, type DuplicateReport } from "../duplicates";
import type { ChannelDigestsManifest, VideoDigest } from "../digests";
import type { StatsManifest, VideoStat } from "../stats";
import { pageFileName } from "./contract";
+import { isReportId } from "../report/schema";
+import {
+ REPORTS_INDEX_PATH,
+ reportViewPath,
+ type ReportIndexEntry,
+ type ReportPageView,
+} from "../report/views";
import { recordRead } from "./io-stats";
import {
DEFAULT_PAGE_CONCURRENCY,
@@ -38,11 +45,15 @@ import {
buildDuplicateIndex,
buildStatsIndex,
buildVideoIndex,
+ coerceReportIndex,
+ coerceReportPage,
+ corpusScopeOf,
pageCacheBudgetBytes,
parseGroupsManifest,
type ArchiveReader,
type ChannelGroups,
type ChannelRef,
+ type CorpusScope,
type DuplicateIndex,
type IndexedVideo,
type SiteCorpusJson,
@@ -84,6 +95,9 @@ export class LocalSource implements ArchiveReader {
// The composed site's own declared origin, learned from corpus.json the first
// time the channel list is read. Undefined = not looked at yet.
private siteOrigin: string | null | undefined;
+ // What corpus.json said the site publishes, learned with the channel list.
+ private siteScope?: CorpusScope;
+ private reportIndex?: Promise<ReportIndexEntry[]>;
constructor(private dir: string) {
this.label = `local:${dir}`;
@@ -131,6 +145,42 @@ export class LocalSource implements ArchiveReader {
this.duplicates = undefined;
this.stats = undefined;
this.siteOrigin = undefined;
+ this.siteScope = undefined;
+ this.reportIndex = undefined;
+ }
+
+ async scope(): Promise<CorpusScope> {
+ await this.listChannels();
+ return this.siteScope ?? "full";
+ }
+
+ // A composed dir with no reports/index.json publishes none (ENOENT is data,
+ // as for tags.json).
+ reports(): Promise<ReportIndexEntry[]> {
+ this.reportIndex ??= readLocalJson<unknown>(
+ path.join(this.dir, REPORTS_INDEX_PATH),
+ "reportIndex",
+ )
+ .then(coerceReportIndex)
+ .catch(() => []);
+ return this.reportIndex;
+ }
+
+ // The id is validated before it touches a path: a report id is a slug, and
+ // anything else (a `..`, a slash) names no report rather than another file.
+ async reportPage(id: string): Promise<ReportPageView | null> {
+ if (!isReportId(id)) return null;
+ let raw: unknown;
+ try {
+ raw = await readLocalJson<unknown>(
+ path.join(this.dir, reportViewPath(id)),
+ "reportPage",
+ );
+ } catch (e) {
+ if ((e as NodeJS.ErrnoException).code === "ENOENT") return null;
+ throw e;
+ }
+ return coerceReportPage(raw, id);
}
subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
@@ -234,6 +284,10 @@ export class LocalSource implements ArchiveReader {
const declared = corpus.site?.url?.trim();
this.siteOrigin =
declared && !PREFER_PLATFORM_LINKS ? declared.replace(/\/+$/, "") : null;
+ // A CITED site (spec 5) has no channels by design: never fall back to
+ // listing a transcripts/ directory a stale public dir might still hold.
+ this.siteScope = corpusScopeOf(corpus);
+ if (this.siteScope === "cited") return [];
if (Array.isArray(corpus.channels) && corpus.channels.length > 0) {
return corpus.channels.map((c) => ({
key: c.slug,
diff --git a/common/lib/archive/reader-reports.test.ts b/common/lib/archive/reader-reports.test.ts
@@ -0,0 +1,235 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, mkdir, rm, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { CONTRACT } from "./contract";
+import { RemoteSource, coerceReportIndex, corpusScopeOf } from "./reader";
+import { LocalSource } from "./reader-fs";
+import { HubSource } from "./reader-hub";
+import {
+ REPORT_INDEX_FORMAT,
+ REPORT_PAGE_FORMAT,
+ REPORT_VIEWS_VERSION,
+ type ReportIndexView,
+ type ReportPageView,
+} from "../report/views";
+
+// Spec 5 for the three readers: a CITED site (corpus.json `site.scope:
+// "cited"`) reads as an empty corpus without an error and names its reports; a
+// full site with reports reads its channels exactly as before.
+
+const ORIGIN = "https://reports.example";
+
+const CITED_CORPUS = {
+ spec: CONTRACT.corpusSpec,
+ kind: "site",
+ site: { id: "demo-reports", title: "Demo reports", url: ORIGIN, scope: "cited", audience: "public" },
+ totals: { channels: 0, videos: 0 },
+ channels: [],
+ reports: { index: `${ORIGIN}/reports/index.json`, count: 1 },
+};
+
+const FULL_CORPUS = {
+ spec: CONTRACT.corpusSpec,
+ kind: "site",
+ site: { id: "demo-full", title: "Demo full", url: ORIGIN },
+ channels: [{ slug: "demo-channel", name: "Demo Channel", videoCount: 1 }],
+ reports: { index: `${ORIGIN}/reports/index.json`, count: 1 },
+};
+
+const INDEX: ReportIndexView = {
+ format: REPORT_INDEX_FORMAT,
+ version: REPORT_VIEWS_VERSION,
+ reports: [
+ {
+ id: "demo-report",
+ kind: "factcheck",
+ title: "A demo report",
+ href: "/reports/demo-report/",
+ citationCount: 1,
+ claimCount: 1,
+ tally: [{ verdict: "CORROBORATED", count: 1 }],
+ },
+ ],
+};
+
+const PAGE = {
+ format: REPORT_PAGE_FORMAT,
+ version: REPORT_VIEWS_VERSION,
+ id: "demo-report",
+ kind: "factcheck",
+ title: "A demo report",
+ sources: {},
+ verdicts: {},
+ citations: {},
+ sections: [{ id: "s1", title: "One", claims: [] }],
+} as unknown as ReportPageView;
+
+function installFetch(body: Record<string, unknown>, fetches: string[] = []): () => void {
+ const real = globalThis.fetch;
+ globalThis.fetch = (async (input: RequestInfo | URL) => {
+ const url = String(input);
+ fetches.push(url);
+ if (!(url in body)) return new Response("not found", { status: 404, statusText: "Not Found" });
+ return new Response(JSON.stringify(body[url]), { status: 200 });
+ }) as typeof fetch;
+ return () => {
+ globalThis.fetch = real;
+ };
+}
+
+test("corpusScopeOf: only an explicit cited scope is cited; a pre-spec-5 corpus is full", () => {
+ assert.equal(corpusScopeOf(CITED_CORPUS), "cited");
+ assert.equal(corpusScopeOf(FULL_CORPUS), "full");
+ assert.equal(corpusScopeOf({ site: { scope: "other" } }), "full");
+ assert.equal(corpusScopeOf(null), "full");
+});
+
+test("coerceReportIndex keeps the entries of an index and nothing else", () => {
+ assert.deepEqual(coerceReportIndex(INDEX).map((e) => e.id), ["demo-report"]);
+ assert.deepEqual(coerceReportIndex({ reports: INDEX.reports }), [], "no format: not an index");
+ assert.deepEqual(coerceReportIndex(null), []);
+ assert.deepEqual(
+ coerceReportIndex({ ...INDEX, reports: [{ id: 1 }, null, ...INDEX.reports] }).map((e) => e.id),
+ ["demo-report"],
+ );
+});
+
+test("remote: a cited site is an empty corpus with its reports, read without an error", async () => {
+ const fetches: string[] = [];
+ const restore = installFetch(
+ {
+ [`${ORIGIN}/corpus.json`]: CITED_CORPUS,
+ [`${ORIGIN}/reports/index.json`]: INDEX,
+ [`${ORIGIN}/reports/demo-report/page.json`]: PAGE,
+ },
+ fetches,
+ );
+ try {
+ const r = new RemoteSource(ORIGIN);
+ assert.deepEqual(await r.listChannels(), []);
+ assert.equal(await r.scope(), "cited");
+ assert.deepEqual((await r.reports()).map((e) => e.title), ["A demo report"]);
+ assert.equal((await r.reportPage("demo-report"))?.title, "A demo report");
+ assert.equal(await r.reportPage("nope"), null, "an unknown report is null, not a throw");
+ // The empty reads a tool makes on any source degrade, they do not throw.
+ assert.deepEqual((await r.loadGroups()).groups, []);
+ assert.equal((await r.videoIndex()).size, 0);
+ await r.reports();
+ assert.equal(fetches.filter((u) => u.endsWith("/reports/index.json")).length, 1, "the index is cached");
+ } finally {
+ restore();
+ }
+});
+
+test("remote: a report id is one path segment — it cannot leave /reports/", async () => {
+ const fetches: string[] = [];
+ const restore = installFetch({ [`${ORIGIN}/corpus.json`]: CITED_CORPUS }, fetches);
+ try {
+ assert.equal(await new RemoteSource(ORIGIN).reportPage("../corpus"), null);
+ assert.deepEqual(fetches, [`${ORIGIN}/reports/..%2Fcorpus/page.json`]);
+ } finally {
+ restore();
+ }
+});
+
+test("remote: a full site with reports reads its channels as before; one without has no reports", async () => {
+ const restore = installFetch({
+ [`${ORIGIN}/corpus.json`]: FULL_CORPUS,
+ [`${ORIGIN}/reports/index.json`]: INDEX,
+ });
+ try {
+ const r = new RemoteSource(ORIGIN);
+ assert.deepEqual((await r.listChannels()).map((c) => c.slug), ["demo-channel"]);
+ assert.equal(await r.scope(), "full");
+ assert.equal((await r.reports()).length, 1);
+ } finally {
+ restore();
+ }
+ const restore2 = installFetch({ [`${ORIGIN}/corpus.json`]: { ...FULL_CORPUS, spec: 4, reports: undefined } });
+ try {
+ const r = new RemoteSource(ORIGIN);
+ assert.equal(await r.scope(), "full");
+ assert.deepEqual(await r.reports(), [], "a 404 index is no reports");
+ } finally {
+ restore2();
+ }
+});
+
+test("hub: a cited member federates no channels and fails nothing", async () => {
+ const HUB = "https://hub.example";
+ const FULL = "https://full.example";
+ const restore = installFetch({
+ [`${HUB}/corpus.json`]: {
+ kind: "hub",
+ sites: [
+ { siteId: "demo-reports", title: "Demo reports", url: ORIGIN },
+ { siteId: "demo-full", title: "Demo full", url: FULL },
+ ],
+ },
+ [`${ORIGIN}/corpus.json`]: CITED_CORPUS,
+ [`${FULL}/corpus.json`]: { ...FULL_CORPUS, site: { ...FULL_CORPUS.site, url: FULL } },
+ });
+ try {
+ const hub = new HubSource(HUB);
+ const channels = await hub.listChannels();
+ assert.deepEqual(channels.map((c) => c.key), ["demo-full/demo-channel"]);
+ assert.equal((await hub.videoIndex()).size, 0);
+ } finally {
+ restore();
+ }
+});
+
+async function writeDir(files: Record<string, unknown>): Promise<string> {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "reader-reports-"));
+ for (const [rel, body] of Object.entries(files)) {
+ const file = path.join(dir, rel);
+ await mkdir(path.dirname(file), { recursive: true });
+ await writeFile(file, JSON.stringify(body));
+ }
+ return dir;
+}
+
+test("local: a cited dir never falls back to a stale transcripts/ listing", async () => {
+ const dir = await writeDir({
+ "corpus.json": CITED_CORPUS,
+ "reports/index.json": INDEX,
+ "reports/demo-report/page.json": PAGE,
+ // A stale channel tree a cited build must not be read as.
+ "transcripts/demo-channel/manifest.json": { version: 1, pageCount: 0, slugToPage: {} },
+ });
+ try {
+ const l = new LocalSource(dir);
+ assert.deepEqual(await l.listChannels(), []);
+ assert.equal(await l.scope(), "cited");
+ assert.equal(l.publicOrigin(), ORIGIN);
+ assert.deepEqual((await l.reports()).map((e) => e.id), ["demo-report"]);
+ assert.equal((await l.reportPage("demo-report"))?.id, "demo-report");
+ assert.equal(await l.reportPage("missing"), null);
+ assert.equal(await l.reportPage("../corpus.json"), null, "not a report id");
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("local: a full dir with no reports reads as before", async () => {
+ const dir = await writeDir({ "corpus.json": { ...FULL_CORPUS, reports: undefined } });
+ try {
+ const l = new LocalSource(dir);
+ assert.deepEqual((await l.listChannels()).map((c) => c.slug), ["demo-channel"]);
+ assert.equal(await l.scope(), "full");
+ assert.deepEqual(await l.reports(), []);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("local: a page.json that is not a report page is an error, not an absence", async () => {
+ const dir = await writeDir({ "corpus.json": CITED_CORPUS, "reports/demo-report/page.json": { hello: 1 } });
+ try {
+ await assert.rejects(() => new LocalSource(dir).reportPage("demo-report"), /not a report page view/);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
diff --git a/common/lib/archive/reader.ts b/common/lib/archive/reader.ts
@@ -50,6 +50,13 @@ import {
type HubSite,
} from "./contract";
import { recordRead } from "./io-stats";
+import {
+ REPORT_INDEX_FORMAT,
+ REPORTS_INDEX_PATH,
+ reportViewPath,
+ type ReportIndexEntry,
+ type ReportPageView,
+} from "../report/views";
// A channel the source can serve. `siteId`/`siteUrl` are only populated in hub
// mode (so results can be attributed to the owning member site); `key` is the
@@ -400,6 +407,61 @@ export interface ArchiveReader {
// channel per call. OPTIONAL on the interface for the usual reason.
digestsManifest?(ch: ChannelRef): Promise<ChannelDigestsManifest | null>;
digestPage?(ch: ChannelRef, page: number): Promise<VideoDigest[]>;
+ // What the site publishes (corpus spec 5): "cited" for a report site, whose
+ // channel list is EMPTY BY DESIGN — no channels, no shards, only its reports
+ // and the moments they cite. A caller says so rather than reporting an empty
+ // archive. Learned from corpus.json with the channel list. OPTIONAL: a stub
+ // has no corpus.json, and a hub's members each have their own.
+ scope?(): Promise<CorpusScope>;
+ // The site's report index (/reports/index.json, spec 5): one entry per
+ // published report. [] when it publishes none — absent, malformed, or a site
+ // built before spec 5. Cached per source.
+ reports?(): Promise<ReportIndexEntry[]>;
+ // One report's page view (/reports/<id>/page.json), every citation
+ // resolved; null when the site publishes no report of that id.
+ reportPage?(id: string): Promise<ReportPageView | null>;
+}
+
+// What a site publishes, as its corpus.json says (spec 5). "full" is the
+// searchable corpus every site has always been, and what a corpus.json from
+// before spec 5 reads as; "cited" is a report site.
+export type CorpusScope = "full" | "cited";
+
+export function corpusScopeOf(corpus: SiteCorpusJson | null | undefined): CorpusScope {
+ return corpus?.site?.scope === "cited" ? "cited" : "full";
+}
+
+// A report index as served, folded to its entries. Tolerant like the tags
+// read: a document that is not an index, or an entry without an id and a
+// title, is dropped rather than thrown — an index the reader cannot use is a
+// site with no reports it can name.
+export function coerceReportIndex(raw: unknown): ReportIndexEntry[] {
+ const doc = (raw ?? {}) as { format?: unknown; reports?: unknown };
+ if (doc.format !== REPORT_INDEX_FORMAT || !Array.isArray(doc.reports)) return [];
+ return doc.reports.filter(
+ (e): e is ReportIndexEntry =>
+ !!e &&
+ typeof e === "object" &&
+ typeof (e as ReportIndexEntry).id === "string" &&
+ typeof (e as ReportIndexEntry).title === "string",
+ );
+}
+
+// A report page view as served, or a throw: unlike the index, a page asked for
+// by id that does not parse as one is an error worth naming, not an absence.
+export function coerceReportPage(raw: unknown, id: string): ReportPageView {
+ const v = (raw ?? {}) as Partial<ReportPageView>;
+ if (typeof v.id !== "string" || !Array.isArray(v.sections) || typeof v.citations !== "object") {
+ throw new Error(`report ${id}: page.json is not a report page view`);
+ }
+ return v as ReportPageView;
+}
+
+// A report id as ONE path segment of a URL. An id is a lowercase slug
+// (lib/report/schema.ts REPORT_ID_RE); anything else is encoded so it cannot
+// leave /reports/ — it simply names no report.
+function reportPagePathFor(id: string): string {
+ return reportViewPath(encodeURIComponent(id));
}
// An HTTP STATUS the archive itself returned, as opposed to a transport
@@ -435,7 +497,10 @@ export type SiteCorpusJson = {
channels?: CorpusJsonChannel[];
// The composed site's own declared public origin — the deployed archilyzer
// viewer these shards were built for. Present in every spec-3 corpus.json.
- site?: { id?: string; title?: string; url?: string };
+ // `scope: "cited"` (spec 5): a report site, with no channels (corpusScopeOf).
+ site?: { id?: string; title?: string; url?: string; scope?: string };
+ // Spec 5: the site's reports, when it publishes any.
+ reports?: { index?: string; count?: number };
};
export type HubCorpusJson = {
@@ -600,6 +665,10 @@ export class RemoteSource implements ArchiveReader {
private postsManifests = new PromiseMap<ChannelPostsManifest | null>();
private transcriptManifests = new PromiseMap<ChannelTranscriptsManifest>();
private transcriptPages: PageCache<TranscriptDetail[]>;
+ // What corpus.json said the site publishes, learned by readChannels().
+ private siteScope?: CorpusScope;
+ private reportIndex?: Promise<ReportIndexEntry[]>;
+ private reportPages = new PromiseMap<ReportPageView | null>();
// Over HTTP the cost is latency, not parse, so a wider window is the dominant
// win — this is where bounded concurrency actually pays.
@@ -632,6 +701,41 @@ export class RemoteSource implements ArchiveReader {
this.index = undefined;
this.duplicates = undefined;
this.stats = undefined;
+ this.siteScope = undefined;
+ this.reportIndex = undefined;
+ this.reportPages.clear();
+ }
+
+ async scope(): Promise<CorpusScope> {
+ await this.listChannels();
+ return this.siteScope ?? "full";
+ }
+
+ // A 404 is the answer "no reports" (a full site without any, or one built
+ // before spec 5) and is cached; any other failure is not memoised.
+ reports(): Promise<ReportIndexEntry[]> {
+ this.reportIndex ??= this.readReportIndex()
+ .then(coerceReportIndex)
+ .catch((e: unknown) => {
+ if (e instanceof ArchiveHttpError && e.status === 404) return [];
+ this.reportIndex = undefined;
+ throw e;
+ });
+ return this.reportIndex;
+ }
+
+ reportPage(id: string): Promise<ReportPageView | null> {
+ return this.reportPages.take(id, async () => {
+ try {
+ return coerceReportPage(
+ await this.getJson<unknown>(reportPagePathFor(id), "reportPage"),
+ id,
+ );
+ } catch (e) {
+ if (e instanceof ArchiveHttpError && e.status === 404) return null;
+ throw e;
+ }
+ });
}
subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
@@ -795,6 +899,10 @@ export class RemoteSource implements ArchiveReader {
return this.getJson<unknown>(rootFileUrl("search-aliases.json"));
}
+ readReportIndex(): Promise<unknown> {
+ return this.getJson<unknown>(REPORTS_INDEX_PATH, "reportIndex");
+ }
+
readTags(): Promise<unknown> {
return this.getJson<unknown>(rootFileUrl(TAGS_FILENAME));
}
@@ -867,8 +975,13 @@ export class RemoteSource implements ArchiveReader {
return this.channelList;
}
+ // A CITED site's corpus.json lists no channels (spec 5); its scope is
+ // recorded so a caller can say "a report site" rather than "empty", and a
+ // stray channel entry on one is not read as a corpus.
private async readChannels(): Promise<ChannelRef[]> {
const corpus = await this.readCorpus();
+ this.siteScope = corpusScopeOf(corpus);
+ if (this.siteScope === "cited") return [];
return (corpus.channels ?? []).map((c) => ({
key: c.slug,
slug: c.slug,