Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit d2aef78ce52ee9157972f5b4a2e8cd6e49c8c9bb
parent 1362716c5c8b15293d2542cf549602f5c2c151ec
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 05:17:53 -0400

archive readers: a cited corpus (spec 5) reads as an empty corpus and names its reports — scope(), reports(), reportPage() on the local and remote readers; a cited local dir never falls back to a transcripts/ listing

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/lib/archive/reader-fs.ts | 54++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archive/reader-reports.test.ts | 235+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/archive/reader.ts | 115++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
3 files changed, 403 insertions(+), 1 deletion(-)

diff --git a/common/lib/archive/reader-fs.ts b/common/lib/archive/reader-fs.ts @@ -29,6 +29,13 @@ import { DUPLICATES_FILENAME, type DuplicateReport } from "../duplicates"; import type { ChannelDigestsManifest, VideoDigest } from "../digests"; import type { StatsManifest, VideoStat } from "../stats"; import { pageFileName } from "./contract"; +import { isReportId } from "../report/schema"; +import { + REPORTS_INDEX_PATH, + reportViewPath, + type ReportIndexEntry, + type ReportPageView, +} from "../report/views"; import { recordRead } from "./io-stats"; import { DEFAULT_PAGE_CONCURRENCY, @@ -38,11 +45,15 @@ import { buildDuplicateIndex, buildStatsIndex, buildVideoIndex, + coerceReportIndex, + coerceReportPage, + corpusScopeOf, pageCacheBudgetBytes, parseGroupsManifest, type ArchiveReader, type ChannelGroups, type ChannelRef, + type CorpusScope, type DuplicateIndex, type IndexedVideo, type SiteCorpusJson, @@ -84,6 +95,9 @@ export class LocalSource implements ArchiveReader { // The composed site's own declared origin, learned from corpus.json the first // time the channel list is read. Undefined = not looked at yet. private siteOrigin: string | null | undefined; + // What corpus.json said the site publishes, learned with the channel list. + private siteScope?: CorpusScope; + private reportIndex?: Promise<ReportIndexEntry[]>; constructor(private dir: string) { this.label = `local:${dir}`; @@ -131,6 +145,42 @@ export class LocalSource implements ArchiveReader { this.duplicates = undefined; this.stats = undefined; this.siteOrigin = undefined; + this.siteScope = undefined; + this.reportIndex = undefined; + } + + async scope(): Promise<CorpusScope> { + await this.listChannels(); + return this.siteScope ?? "full"; + } + + // A composed dir with no reports/index.json publishes none (ENOENT is data, + // as for tags.json). + reports(): Promise<ReportIndexEntry[]> { + this.reportIndex ??= readLocalJson<unknown>( + path.join(this.dir, REPORTS_INDEX_PATH), + "reportIndex", + ) + .then(coerceReportIndex) + .catch(() => []); + return this.reportIndex; + } + + // The id is validated before it touches a path: a report id is a slug, and + // anything else (a `..`, a slash) names no report rather than another file. + async reportPage(id: string): Promise<ReportPageView | null> { + if (!isReportId(id)) return null; + let raw: unknown; + try { + raw = await readLocalJson<unknown>( + path.join(this.dir, reportViewPath(id)), + "reportPage", + ); + } catch (e) { + if ((e as NodeJS.ErrnoException).code === "ENOENT") return null; + throw e; + } + return coerceReportPage(raw, id); } subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { @@ -234,6 +284,10 @@ export class LocalSource implements ArchiveReader { const declared = corpus.site?.url?.trim(); this.siteOrigin = declared && !PREFER_PLATFORM_LINKS ? declared.replace(/\/+$/, "") : null; + // A CITED site (spec 5) has no channels by design: never fall back to + // listing a transcripts/ directory a stale public dir might still hold. + this.siteScope = corpusScopeOf(corpus); + if (this.siteScope === "cited") return []; if (Array.isArray(corpus.channels) && corpus.channels.length > 0) { return corpus.channels.map((c) => ({ key: c.slug, diff --git a/common/lib/archive/reader-reports.test.ts b/common/lib/archive/reader-reports.test.ts @@ -0,0 +1,235 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, mkdir, rm, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { CONTRACT } from "./contract"; +import { RemoteSource, coerceReportIndex, corpusScopeOf } from "./reader"; +import { LocalSource } from "./reader-fs"; +import { HubSource } from "./reader-hub"; +import { + REPORT_INDEX_FORMAT, + REPORT_PAGE_FORMAT, + REPORT_VIEWS_VERSION, + type ReportIndexView, + type ReportPageView, +} from "../report/views"; + +// Spec 5 for the three readers: a CITED site (corpus.json `site.scope: +// "cited"`) reads as an empty corpus without an error and names its reports; a +// full site with reports reads its channels exactly as before. + +const ORIGIN = "https://reports.example"; + +const CITED_CORPUS = { + spec: CONTRACT.corpusSpec, + kind: "site", + site: { id: "demo-reports", title: "Demo reports", url: ORIGIN, scope: "cited", audience: "public" }, + totals: { channels: 0, videos: 0 }, + channels: [], + reports: { index: `${ORIGIN}/reports/index.json`, count: 1 }, +}; + +const FULL_CORPUS = { + spec: CONTRACT.corpusSpec, + kind: "site", + site: { id: "demo-full", title: "Demo full", url: ORIGIN }, + channels: [{ slug: "demo-channel", name: "Demo Channel", videoCount: 1 }], + reports: { index: `${ORIGIN}/reports/index.json`, count: 1 }, +}; + +const INDEX: ReportIndexView = { + format: REPORT_INDEX_FORMAT, + version: REPORT_VIEWS_VERSION, + reports: [ + { + id: "demo-report", + kind: "factcheck", + title: "A demo report", + href: "/reports/demo-report/", + citationCount: 1, + claimCount: 1, + tally: [{ verdict: "CORROBORATED", count: 1 }], + }, + ], +}; + +const PAGE = { + format: REPORT_PAGE_FORMAT, + version: REPORT_VIEWS_VERSION, + id: "demo-report", + kind: "factcheck", + title: "A demo report", + sources: {}, + verdicts: {}, + citations: {}, + sections: [{ id: "s1", title: "One", claims: [] }], +} as unknown as ReportPageView; + +function installFetch(body: Record<string, unknown>, fetches: string[] = []): () => void { + const real = globalThis.fetch; + globalThis.fetch = (async (input: RequestInfo | URL) => { + const url = String(input); + fetches.push(url); + if (!(url in body)) return new Response("not found", { status: 404, statusText: "Not Found" }); + return new Response(JSON.stringify(body[url]), { status: 200 }); + }) as typeof fetch; + return () => { + globalThis.fetch = real; + }; +} + +test("corpusScopeOf: only an explicit cited scope is cited; a pre-spec-5 corpus is full", () => { + assert.equal(corpusScopeOf(CITED_CORPUS), "cited"); + assert.equal(corpusScopeOf(FULL_CORPUS), "full"); + assert.equal(corpusScopeOf({ site: { scope: "other" } }), "full"); + assert.equal(corpusScopeOf(null), "full"); +}); + +test("coerceReportIndex keeps the entries of an index and nothing else", () => { + assert.deepEqual(coerceReportIndex(INDEX).map((e) => e.id), ["demo-report"]); + assert.deepEqual(coerceReportIndex({ reports: INDEX.reports }), [], "no format: not an index"); + assert.deepEqual(coerceReportIndex(null), []); + assert.deepEqual( + coerceReportIndex({ ...INDEX, reports: [{ id: 1 }, null, ...INDEX.reports] }).map((e) => e.id), + ["demo-report"], + ); +}); + +test("remote: a cited site is an empty corpus with its reports, read without an error", async () => { + const fetches: string[] = []; + const restore = installFetch( + { + [`${ORIGIN}/corpus.json`]: CITED_CORPUS, + [`${ORIGIN}/reports/index.json`]: INDEX, + [`${ORIGIN}/reports/demo-report/page.json`]: PAGE, + }, + fetches, + ); + try { + const r = new RemoteSource(ORIGIN); + assert.deepEqual(await r.listChannels(), []); + assert.equal(await r.scope(), "cited"); + assert.deepEqual((await r.reports()).map((e) => e.title), ["A demo report"]); + assert.equal((await r.reportPage("demo-report"))?.title, "A demo report"); + assert.equal(await r.reportPage("nope"), null, "an unknown report is null, not a throw"); + // The empty reads a tool makes on any source degrade, they do not throw. + assert.deepEqual((await r.loadGroups()).groups, []); + assert.equal((await r.videoIndex()).size, 0); + await r.reports(); + assert.equal(fetches.filter((u) => u.endsWith("/reports/index.json")).length, 1, "the index is cached"); + } finally { + restore(); + } +}); + +test("remote: a report id is one path segment — it cannot leave /reports/", async () => { + const fetches: string[] = []; + const restore = installFetch({ [`${ORIGIN}/corpus.json`]: CITED_CORPUS }, fetches); + try { + assert.equal(await new RemoteSource(ORIGIN).reportPage("../corpus"), null); + assert.deepEqual(fetches, [`${ORIGIN}/reports/..%2Fcorpus/page.json`]); + } finally { + restore(); + } +}); + +test("remote: a full site with reports reads its channels as before; one without has no reports", async () => { + const restore = installFetch({ + [`${ORIGIN}/corpus.json`]: FULL_CORPUS, + [`${ORIGIN}/reports/index.json`]: INDEX, + }); + try { + const r = new RemoteSource(ORIGIN); + assert.deepEqual((await r.listChannels()).map((c) => c.slug), ["demo-channel"]); + assert.equal(await r.scope(), "full"); + assert.equal((await r.reports()).length, 1); + } finally { + restore(); + } + const restore2 = installFetch({ [`${ORIGIN}/corpus.json`]: { ...FULL_CORPUS, spec: 4, reports: undefined } }); + try { + const r = new RemoteSource(ORIGIN); + assert.equal(await r.scope(), "full"); + assert.deepEqual(await r.reports(), [], "a 404 index is no reports"); + } finally { + restore2(); + } +}); + +test("hub: a cited member federates no channels and fails nothing", async () => { + const HUB = "https://hub.example"; + const FULL = "https://full.example"; + const restore = installFetch({ + [`${HUB}/corpus.json`]: { + kind: "hub", + sites: [ + { siteId: "demo-reports", title: "Demo reports", url: ORIGIN }, + { siteId: "demo-full", title: "Demo full", url: FULL }, + ], + }, + [`${ORIGIN}/corpus.json`]: CITED_CORPUS, + [`${FULL}/corpus.json`]: { ...FULL_CORPUS, site: { ...FULL_CORPUS.site, url: FULL } }, + }); + try { + const hub = new HubSource(HUB); + const channels = await hub.listChannels(); + assert.deepEqual(channels.map((c) => c.key), ["demo-full/demo-channel"]); + assert.equal((await hub.videoIndex()).size, 0); + } finally { + restore(); + } +}); + +async function writeDir(files: Record<string, unknown>): Promise<string> { + const dir = await mkdtemp(path.join(os.tmpdir(), "reader-reports-")); + for (const [rel, body] of Object.entries(files)) { + const file = path.join(dir, rel); + await mkdir(path.dirname(file), { recursive: true }); + await writeFile(file, JSON.stringify(body)); + } + return dir; +} + +test("local: a cited dir never falls back to a stale transcripts/ listing", async () => { + const dir = await writeDir({ + "corpus.json": CITED_CORPUS, + "reports/index.json": INDEX, + "reports/demo-report/page.json": PAGE, + // A stale channel tree a cited build must not be read as. + "transcripts/demo-channel/manifest.json": { version: 1, pageCount: 0, slugToPage: {} }, + }); + try { + const l = new LocalSource(dir); + assert.deepEqual(await l.listChannels(), []); + assert.equal(await l.scope(), "cited"); + assert.equal(l.publicOrigin(), ORIGIN); + assert.deepEqual((await l.reports()).map((e) => e.id), ["demo-report"]); + assert.equal((await l.reportPage("demo-report"))?.id, "demo-report"); + assert.equal(await l.reportPage("missing"), null); + assert.equal(await l.reportPage("../corpus.json"), null, "not a report id"); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("local: a full dir with no reports reads as before", async () => { + const dir = await writeDir({ "corpus.json": { ...FULL_CORPUS, reports: undefined } }); + try { + const l = new LocalSource(dir); + assert.deepEqual((await l.listChannels()).map((c) => c.slug), ["demo-channel"]); + assert.equal(await l.scope(), "full"); + assert.deepEqual(await l.reports(), []); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("local: a page.json that is not a report page is an error, not an absence", async () => { + const dir = await writeDir({ "corpus.json": CITED_CORPUS, "reports/demo-report/page.json": { hello: 1 } }); + try { + await assert.rejects(() => new LocalSource(dir).reportPage("demo-report"), /not a report page view/); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); diff --git a/common/lib/archive/reader.ts b/common/lib/archive/reader.ts @@ -50,6 +50,13 @@ import { type HubSite, } from "./contract"; import { recordRead } from "./io-stats"; +import { + REPORT_INDEX_FORMAT, + REPORTS_INDEX_PATH, + reportViewPath, + type ReportIndexEntry, + type ReportPageView, +} from "../report/views"; // A channel the source can serve. `siteId`/`siteUrl` are only populated in hub // mode (so results can be attributed to the owning member site); `key` is the @@ -400,6 +407,61 @@ export interface ArchiveReader { // channel per call. OPTIONAL on the interface for the usual reason. digestsManifest?(ch: ChannelRef): Promise<ChannelDigestsManifest | null>; digestPage?(ch: ChannelRef, page: number): Promise<VideoDigest[]>; + // What the site publishes (corpus spec 5): "cited" for a report site, whose + // channel list is EMPTY BY DESIGN — no channels, no shards, only its reports + // and the moments they cite. A caller says so rather than reporting an empty + // archive. Learned from corpus.json with the channel list. OPTIONAL: a stub + // has no corpus.json, and a hub's members each have their own. + scope?(): Promise<CorpusScope>; + // The site's report index (/reports/index.json, spec 5): one entry per + // published report. [] when it publishes none — absent, malformed, or a site + // built before spec 5. Cached per source. + reports?(): Promise<ReportIndexEntry[]>; + // One report's page view (/reports/<id>/page.json), every citation + // resolved; null when the site publishes no report of that id. + reportPage?(id: string): Promise<ReportPageView | null>; +} + +// What a site publishes, as its corpus.json says (spec 5). "full" is the +// searchable corpus every site has always been, and what a corpus.json from +// before spec 5 reads as; "cited" is a report site. +export type CorpusScope = "full" | "cited"; + +export function corpusScopeOf(corpus: SiteCorpusJson | null | undefined): CorpusScope { + return corpus?.site?.scope === "cited" ? "cited" : "full"; +} + +// A report index as served, folded to its entries. Tolerant like the tags +// read: a document that is not an index, or an entry without an id and a +// title, is dropped rather than thrown — an index the reader cannot use is a +// site with no reports it can name. +export function coerceReportIndex(raw: unknown): ReportIndexEntry[] { + const doc = (raw ?? {}) as { format?: unknown; reports?: unknown }; + if (doc.format !== REPORT_INDEX_FORMAT || !Array.isArray(doc.reports)) return []; + return doc.reports.filter( + (e): e is ReportIndexEntry => + !!e && + typeof e === "object" && + typeof (e as ReportIndexEntry).id === "string" && + typeof (e as ReportIndexEntry).title === "string", + ); +} + +// A report page view as served, or a throw: unlike the index, a page asked for +// by id that does not parse as one is an error worth naming, not an absence. +export function coerceReportPage(raw: unknown, id: string): ReportPageView { + const v = (raw ?? {}) as Partial<ReportPageView>; + if (typeof v.id !== "string" || !Array.isArray(v.sections) || typeof v.citations !== "object") { + throw new Error(`report ${id}: page.json is not a report page view`); + } + return v as ReportPageView; +} + +// A report id as ONE path segment of a URL. An id is a lowercase slug +// (lib/report/schema.ts REPORT_ID_RE); anything else is encoded so it cannot +// leave /reports/ — it simply names no report. +function reportPagePathFor(id: string): string { + return reportViewPath(encodeURIComponent(id)); } // An HTTP STATUS the archive itself returned, as opposed to a transport @@ -435,7 +497,10 @@ export type SiteCorpusJson = { channels?: CorpusJsonChannel[]; // The composed site's own declared public origin — the deployed archilyzer // viewer these shards were built for. Present in every spec-3 corpus.json. - site?: { id?: string; title?: string; url?: string }; + // `scope: "cited"` (spec 5): a report site, with no channels (corpusScopeOf). + site?: { id?: string; title?: string; url?: string; scope?: string }; + // Spec 5: the site's reports, when it publishes any. + reports?: { index?: string; count?: number }; }; export type HubCorpusJson = { @@ -600,6 +665,10 @@ export class RemoteSource implements ArchiveReader { private postsManifests = new PromiseMap<ChannelPostsManifest | null>(); private transcriptManifests = new PromiseMap<ChannelTranscriptsManifest>(); private transcriptPages: PageCache<TranscriptDetail[]>; + // What corpus.json said the site publishes, learned by readChannels(). + private siteScope?: CorpusScope; + private reportIndex?: Promise<ReportIndexEntry[]>; + private reportPages = new PromiseMap<ReportPageView | null>(); // Over HTTP the cost is latency, not parse, so a wider window is the dominant // win — this is where bounded concurrency actually pays. @@ -632,6 +701,41 @@ export class RemoteSource implements ArchiveReader { this.index = undefined; this.duplicates = undefined; this.stats = undefined; + this.siteScope = undefined; + this.reportIndex = undefined; + this.reportPages.clear(); + } + + async scope(): Promise<CorpusScope> { + await this.listChannels(); + return this.siteScope ?? "full"; + } + + // A 404 is the answer "no reports" (a full site without any, or one built + // before spec 5) and is cached; any other failure is not memoised. + reports(): Promise<ReportIndexEntry[]> { + this.reportIndex ??= this.readReportIndex() + .then(coerceReportIndex) + .catch((e: unknown) => { + if (e instanceof ArchiveHttpError && e.status === 404) return []; + this.reportIndex = undefined; + throw e; + }); + return this.reportIndex; + } + + reportPage(id: string): Promise<ReportPageView | null> { + return this.reportPages.take(id, async () => { + try { + return coerceReportPage( + await this.getJson<unknown>(reportPagePathFor(id), "reportPage"), + id, + ); + } catch (e) { + if (e instanceof ArchiveHttpError && e.status === 404) return null; + throw e; + } + }); } subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { @@ -795,6 +899,10 @@ export class RemoteSource implements ArchiveReader { return this.getJson<unknown>(rootFileUrl("search-aliases.json")); } + readReportIndex(): Promise<unknown> { + return this.getJson<unknown>(REPORTS_INDEX_PATH, "reportIndex"); + } + readTags(): Promise<unknown> { return this.getJson<unknown>(rootFileUrl(TAGS_FILENAME)); } @@ -867,8 +975,13 @@ export class RemoteSource implements ArchiveReader { return this.channelList; } + // A CITED site's corpus.json lists no channels (spec 5); its scope is + // recorded so a caller can say "a report site" rather than "empty", and a + // stray channel entry on one is not read as a corpus. private async readChannels(): Promise<ChannelRef[]> { const corpus = await this.readCorpus(); + this.siteScope = corpusScopeOf(corpus); + if (this.siteScope === "cited") return []; return (corpus.channels ?? []).map((c) => ({ key: c.slug, slug: c.slug,