// THE ARCHIVE READER — one walk of the published shard scheme. // // Moved verbatim from mcp/src/source.ts, which was the best of the five // hand-rolled readers in the repo (three transports, promise-coalescing caches, // a byte-budgeted LRU, tolerant of every layer a site legitimately does not // ship). `mcp/src/source.ts` is now a re-export of this file, so nothing in the // MCP server changed shape; what changed is that the viewer, the offline cache // and umtool can reach the same implementation instead of re-deriving it. // // ZERO node imports. This file and reader-hub.ts run in a browser; the only // node-touching transport is reader-fs.ts (LocalSource), which imports FROM // here and is never value-imported BY here — a `next build` of the export app // is the test of that, because tsc cannot tell a `node:fs` import from any // other. // // `process` appears once, behind a `typeof process` guard, to read the page // cache budget knob. Everything else is transport-agnostic. import { type ChannelTranscriptsManifest, type ChannelSubsManifest, type SubsManifest, type Manifest, } from "../manifest"; import type { TranscriptDetail, DisplaySummary } from "../transcripts"; import { summaryState, type VideoState } from "../availability"; import type { SubsDetail } from "../subs"; import type { ChannelPostsManifest, Post, PostsManifest } from "../posts"; import { coerceAliasConfig, type SearchAlias } from "../searchAliases"; import { TAGS_FILENAME, type PublishedTag } from "../curatedTags"; import { coercePublishedTags } from "../publishedTags"; import { resolveCanonicalSlug, DUPLICATES_FILENAME, type DuplicateReport, } from "../duplicates"; import type { ChannelDigestsManifest, VideoDigest } from "../digests"; import type { StatsManifest, VideoStat } from "../stats"; import { parseChannelGroups, resolveDefaultGroupId, DEFAULT_GROUP_FALLBACK_ID, type ChannelGroup, } from "../channelGroups"; import { manifestUrl, pageUrl, rootFileUrl, treeManifestUrl, type HubSite, } from "./contract"; import { recordRead } from "./io-stats"; import { REPORT_INDEX_FORMAT, REPORTS_INDEX_PATH, reportViewPath, type ReportIndexEntry, type ReportPageView, } from "../report/views"; // A channel the source can serve. `siteId`/`siteUrl` are only populated in hub // mode (so results can be attributed to the owning member site); `key` is the // stable, source-unique handle a tool passes back to fetch this channel's data. // `groupId` is the channel's raw group membership as shipped in corpus.json (it // may be unknown/absent — resolve it against loadGroups() with // resolveChannelGroupId before using it). export type ChannelRef = { key: string; slug: string; name: string; videoCount?: number; groupId?: string; siteId?: string; siteTitle?: string; siteUrl?: string; }; // The site's channel-group definitions, as read from summaries/manifest.json — // the same groups the viewer's channel filter renders. `defaultGroupId` is the // bucket unknown/absent channel groupIds fold onto (resolveChannelGroupId). export type ChannelGroups = { groups: ChannelGroup[]; defaultGroupId: string; }; // What a source returns when it has no group definitions (absent/malformed // manifest, or hub mode where per-site groups are a different model). export const EMPTY_GROUPS: ChannelGroups = { groups: [], defaultGroupId: DEFAULT_GROUP_FALLBACK_ID, }; // Per-video availability, joined in from the summaries shards (a // TranscriptDetail record does not carry presence state). Keyed by the // member-local video slug (`/`) — the same slug a transcript // page record carries — so the search engine can apply the `fav` filter. export type VideoAvailability = { state: VideoState }; // One video as the summaries shards describe it. A strict superset of // VideoAvailability, so the same map serves both the availability join and the // filter-first page planner — there is one index, not two parallel reads of the // same files. // // Everything here comes from `summaries/`, which is a GLOBAL index of every // video in the corpus: 1.4 MB and ~0.5 s to parse, against 1.3 GB and ~48 s for // the transcripts. That ratio is the whole basis of filter-first scanning — the // exact page set a filtered query needs is computable from this plus each // channel manifest's `slugToPage`, before a single transcript byte is read. export type IndexedVideo = { state: VideoState; id: string; channelSlug: string; title: string; uploadDate: string; isLivestream: boolean; ageRestricted: boolean; // Curated tag ids (lib/curatedTags.ts), when the record carries any. Present // here for the same reason `state` is: it is what the filter-first page // planner prunes on, and FilterableRecord reads it. OMITTED when empty, and // absent entirely on a site built before corpus spec 4 — which the planner // already handles, since a video the index cannot vouch for gets its page // read anyway. curatedTags?: string[]; }; export type VideoIndex = ReadonlyMap; // One video's membership in a cross-platform duplicate cluster, as shipped in // duplicates.json. Keyed by the member-local slug, like the video index. // // `aligned` is carried per SIBLING, not per cluster, and is the gate on ever // translating a timestamp from one copy to another. Absent means NOT MEASURED, // which must be read as not aligned — a mirror with a longer intro matches on // text at shifted times, so a plausible-looking citation would land in the // wrong place in the wrong upload. That is the failure mode that looks like // success, and the only defence is refusing to guess. export type ClusterMembership = { clusterId: string; // The member that owns derived work for this cluster (resolveCanonicalSlug), // or null when a human marked the cluster not-a-duplicate. canonicalSlug: string | null; isCanonical: boolean; // Every OTHER member of the cluster. siblings: { slug: string; id: string; channelSlug: string; channel: string; platform: string; title: string; duration: number; uploadDate: string; hasTranscript: boolean; aligned: boolean; offsetSeconds: number | null; }[]; // The cluster is a clip-of-a-longer-video relationship: the members overlap // only partially, so nothing may be mapped across wholesale. contained: boolean; // Title+duration only — nothing compared the actual content. An unconfirmed // suspect, not an established duplicate. needsReview: boolean; }; export type DuplicateIndex = ReadonlyMap; // Fold a duplicates.json report into a slug → membership map. Tolerant of // absence throughout: `compose-site.ts` only writes the file when there is at // least one publishable cluster, and corpus.json doesn't even declare it, so a // site legitimately ships none. export function buildDuplicateIndex( report: DuplicateReport | null, ): Map { const map = new Map(); if (!report || !Array.isArray(report.clusters)) return map; for (const cluster of report.clusters) { const refs = cluster.videoRefs ?? []; if (refs.length < 2) continue; const canonicalSlug = resolveCanonicalSlug(cluster); // A human marked it not-a-duplicate — it is not a cluster any more. if (canonicalSlug === null) continue; for (const ref of refs) { map.set(ref.slug, { clusterId: cluster.clusterId, canonicalSlug, isCanonical: ref.slug === canonicalSlug, contained: cluster.contained === true, needsReview: cluster.needsReview === true, siblings: refs .filter((o) => o.slug !== ref.slug) .map((o) => ({ slug: o.slug, id: o.id, channelSlug: o.channelSlug, channel: o.channel, platform: o.platform, title: o.title, duration: o.duration, uploadDate: o.uploadDate, hasTranscript: o.hasTranscript === true, // Alignment is a property of the PAIR, and the report records it // against each member relative to the cluster's canonical. Both // sides must be measured-and-aligned before a timestamp may cross. aligned: ref.aligned === true && o.aligned === true, offsetSeconds: o.offsetSeconds ?? null, })), }); } } return map; } // Fold the shipped stats shards into a slug → stat map. Same tolerant shape as // the summaries read: absent or malformed is an empty map, never an error. // // Note the shape difference that forces a full read rather than a targeted one: // StatsManifest carries `channels` and `pageCount` but NO `slugToPage`, so // there is no way to jump to the page holding one video. Pages are large (up to // STATS_MAX_PAGE_BYTES = 20 MB), which is exactly why this is lazy — nothing // reads it until a tool asks for a stat. export async function buildStatsIndex( readManifest: () => Promise, readPage: (page: number) => Promise, ): Promise> { const map = new Map(); let manifest: StatsManifest | null; try { manifest = await readManifest(); } catch { return map; } if (!manifest || typeof manifest.pageCount !== "number") return map; for (let page = 0; page < manifest.pageCount; page++) { let records: VideoStat[] | null; try { records = await readPage(page); } catch { continue; } if (!records) continue; for (const r of records) { if (typeof r.slug === "string") map.set(r.slug, r); } } return map; } // Read a site's global summaries shards (summaries/manifest.json + // summaries/page-NNNN.json) via `readPage` and fold them into a slug → video // map. Tolerant: an absent/malformed manifest yields an empty map, and a page // that fails to read is skipped. `readManifest`/`readPage` throw or return null // on absence per the source's transport. // // Tolerance is load-bearing for the page planner, not just politeness: a // summaries set that is missing, partial, or older than the transcripts must // degrade to "I don't know about this video", and the planner's rule for // don't-know is to scan the page anyway. export async function buildVideoIndex( readManifest: () => Promise, readPage: (page: number) => Promise, ): Promise> { const map = new Map(); let manifest: Manifest | null; try { manifest = await readManifest(); } catch { return map; } if (!manifest || typeof manifest.pageCount !== "number") return map; for (let page = 0; page < manifest.pageCount; page++) { let records: DisplaySummary[] | null; try { records = await readPage(page); } catch { continue; } if (!records) continue; for (const r of records) { if (typeof r.slug !== "string") continue; // summaryState falls back to the legacy isDeleted/isUnlisted booleans, // which matters here more than anywhere: a hub reads summaries pages // from member origins it does not control, so some of them will have // been built before `state` existed. map.set(r.slug, { state: summaryState(r), id: r.id, channelSlug: r.channelSlug, title: r.title ?? "", uploadDate: r.uploadDate ?? "", isLivestream: r.isLivestream === true, ageRestricted: r.ageRestricted === true, ...(Array.isArray(r.curatedTags) && r.curatedTags.length > 0 ? { curatedTags: r.curatedTags.filter((t) => typeof t === "string") } : {}), }); } } return map; } // Parse a summaries/manifest.json blob into channel-group defs, tolerating any // missing/malformed shape (→ empty fallback). export function parseGroupsManifest(raw: unknown): ChannelGroups { const m = (raw ?? {}) as { groups?: unknown; defaultGroupId?: unknown }; const groups = parseChannelGroups(m.groups); return { groups, defaultGroupId: resolveDefaultGroupId(m.defaultGroupId, groups) }; } // A read-only view over a transcript corpus's paginated JSON shards. Three // implementations (local dir / remote origin / federated hub) all speak the // same three-call contract, which mirrors the documented shard scheme in // corpus.json: list channels, get a channel's manifest (slug -> page map), get // a page of full transcript records. // // NOTE the method that is deliberately absent: there is no `record(layer, slug, // id)`. The contract's walk is manifest -> shard -> record and a caller that // wants one record already holds the shard, so a per-record fetch would be a // structural regression (N reads where the plan needs one) dressed up as // convenience. export interface ArchiveReader { readonly label: string; // The channel list, memoised per source instance. searchTranscripts, // findVideo and findPost all call it, so a 20-id get_transcripts batch used // to cost 20 corpus.json fetches over HTTP. The memo is the promise, so // concurrent callers share one fetch. Pass `refresh` to drop it and re-read — // an explicit staleness escape hatch, deliberately not a TTL. listChannels(opts?: { refresh?: boolean }): Promise; transcriptsManifest(ch: ChannelRef): Promise; transcriptPage(ch: ChannelRef, page: number): Promise; // The site's shipped curated search aliases (the same /search-aliases.json the // viewer reads). Returns [] when the file is absent or malformed. Used to make // caption search alias-aware, so a query for a term with a curated regex // (e.g. "k cups" → "(k|cake)[ -]?cup") also matches the mis-transcribed // spellings. Result is cached per source. loadAliases(): Promise; // The site's published curated tags (/tags.json) with their per-site counts. // Returns [] when the file is absent or malformed — which is the normal // state for a site with nothing shippable to publish AND for any site built // before corpus spec 4, so a caller must read [] as "this source publishes no // curated tags" and say so, never as an error. Cached per source. // // OPTIONAL on the interface for the same reason duplicateIndex and statsIndex // are: the in-memory test stubs have no such concept, and a tool must degrade // rather than assume. loadTags?(): Promise; // The site's channel-group definitions (summaries/manifest.json). Returns the // empty fallback when absent/malformed, or in hub mode (federated per-site // groups are a different model — deferred). Cached per source. loadGroups(): Promise; // The public origin of the viewer that owns this source's videos, or null // when there isn't one (a local dir on disk). Used to build archilyzer viewer // deep links for cited moments (momentUrl). A single-site remote returns its // base URL; a hub returns null because each video's origin is its member // site's url (carried on the ChannelRef as `siteUrl`) — prefer that per-video. publicOrigin(): string | null; // A channel's live-chat/subs manifest (subs//manifest.json), or null // when the channel ships no subs shards. Same slugToPage/pageCount shape as // the transcripts manifest. Fetched lazily — only the chat search scope needs // it. subsManifest(ch: ChannelRef): Promise; // A page of a channel's subs records (subs//page-NNNN.json). Each record // inlines its per-track cues under `tracks` (e.g. `tracks.live_chat`). subsPage(ch: ChannelRef, page: number): Promise; // A channel's social-posts manifest (posts//manifest.json), or null // when the channel ships no posts shards (i.e. it is a video channel). Same // slugToPage/pageCount shape as the transcripts manifest. This is the MCP's // single abstraction boundary, so adding it here yields the posts corpus on // all three transports (local / remote / hub) at once. postsManifest(ch: ChannelRef): Promise; // A page of a channel's posts (posts//page-NNNN.json). postsPage(ch: ChannelRef, page: number): Promise; // A map of every video's availability (deleted/unlisted), keyed by the // member-local video slug (`/`), built from the summaries // shards. Fetched lazily and cached — only the `fav` availability filter needs // it. Empty when the source ships no summaries. // // ReadonlyMap so the richer videoIndex() can BE this map rather than a // projection of it: ReadonlyMap is covariant in its value type, so one // Map satisfies both and the two can never drift. availabilityMap(): Promise>; // The full summaries-backed index, when this source ships one. OPTIONAL: the // in-memory test stubs don't implement it, and a site that ships no // summaries/ genuinely has no index — callers must degrade to a full scan // rather than assume an empty index means an empty corpus. videoIndex?(): Promise; // How many shard pages this source is willing to have in flight at once. // Optional; callers use `?? DEFAULT_PAGE_CONCURRENCY`. Local is CPU-bound on // JSON.parse (measured: 42 ms read vs 389 ms parse for an 8 MB page), so // concurrency there only overlaps read with parse and saturates quickly. // Remote is latency-bound, where it is the dominant win. readonly pageConcurrency?: number; // The shipped cross-platform duplicate report, folded to slug → membership. // OPTIONAL and empty-when-absent: compose-site only writes duplicates.json // when there is at least one publishable cluster, corpus.json does not // declare it, and the in-memory test stubs have no such concept. A site that // ships none must behave exactly as it does today. duplicateIndex?(): Promise; // The shipped per-video stats index (view/like counts, cueCount, transcript // coverage), slug-keyed. OPTIONAL for the same reasons. Lazy: stats/ is one // ~3.4 MB page here and up to 20 MB elsewhere, so it is only read when a tool // actually asks for it. statsIndex?(): Promise>; // A channel's AI-digest manifest (digests//manifest.json), or null when // the channel has none. Mirrors the posts pair, including the cached negative // — the digest corpus is SPARSE BY DESIGN (a channel with zero digests gets // no manifest at all), so probing it per query must not cost a read per // channel per call. OPTIONAL on the interface for the usual reason. digestsManifest?(ch: ChannelRef): Promise; digestPage?(ch: ChannelRef, page: number): Promise; // What the site publishes (corpus spec 5): "cited" for a report site, whose // channel list is EMPTY BY DESIGN — no channels, no shards, only its reports // and the moments they cite. A caller says so rather than reporting an empty // archive. Learned from corpus.json with the channel list. OPTIONAL: a stub // has no corpus.json, and a hub's members each have their own. scope?(): Promise; // The site's report index (/reports/index.json, spec 5): one entry per // published report. [] when it publishes none — absent, malformed, or a site // built before spec 5. Cached per source. reports?(): Promise; // One report's page view (/reports//page.json), every citation // resolved; null when the site publishes no report of that id. reportPage?(id: string): Promise; } // What a site publishes, as its corpus.json says (spec 5). "full" is the // searchable corpus every site has always been, and what a corpus.json from // before spec 5 reads as; "cited" is a report site. export type CorpusScope = "full" | "cited"; export function corpusScopeOf(corpus: SiteCorpusJson | null | undefined): CorpusScope { return corpus?.site?.scope === "cited" ? "cited" : "full"; } // A report index as served, folded to its entries. Tolerant like the tags // read: a document that is not an index, or an entry without an id and a // title, is dropped rather than thrown — an index the reader cannot use is a // site with no reports it can name. export function coerceReportIndex(raw: unknown): ReportIndexEntry[] { const doc = (raw ?? {}) as { format?: unknown; reports?: unknown }; if (doc.format !== REPORT_INDEX_FORMAT || !Array.isArray(doc.reports)) return []; return doc.reports.filter( (e): e is ReportIndexEntry => !!e && typeof e === "object" && typeof (e as ReportIndexEntry).id === "string" && typeof (e as ReportIndexEntry).title === "string", ); } // A report page view as served, or a throw: unlike the index, a page asked for // by id that does not parse as one is an error worth naming, not an absence. export function coerceReportPage(raw: unknown, id: string): ReportPageView { const v = (raw ?? {}) as Partial; if (typeof v.id !== "string" || !Array.isArray(v.sections) || typeof v.citations !== "object") { throw new Error(`report ${id}: page.json is not a report page view`); } return v as ReportPageView; } // A report id as ONE path segment of a URL. An id is a lowercase slug // (lib/report/schema.ts REPORT_ID_RE); anything else is encoded so it cannot // leave /reports/ — it simply names no report. function reportPagePathFor(id: string): string { return reportViewPath(encodeURIComponent(id)); } // An HTTP STATUS the archive itself returned, as opposed to a transport // failure. A caller that treats absence as data — a site ships no // /search-aliases.json, a channel ships no digests — needs to tell the two // apart: a 404 is an answer and is worth caching, a dropped connection is // neither. `message` is byte-identical to the plain Error this replaced, so // anything that only reads the message is unaffected. export class ArchiveHttpError extends Error { constructor( readonly status: number, readonly url: string, statusText: string, ) { super(`GET ${url} -> ${status} ${statusText}`); this.name = "ArchiveHttpError"; } } // Used when a source states no preference. Deliberately modest: each in-flight // page costs its raw bytes plus ~2.7× that once parsed, and this box is shared. export const DEFAULT_PAGE_CONCURRENCY = 4; // Shape of the channels we read out of a site corpus.json (Layer 1). Kept loose // — we only need slug/name/count/group. export type CorpusJsonChannel = { slug: string; name?: string; videoCount?: number; groupId?: string; }; export type SiteCorpusJson = { channels?: CorpusJsonChannel[]; // The composed site's own declared public origin — the deployed archilyzer // viewer these shards were built for. Present in every spec-3 corpus.json. // `scope: "cited"` (spec 5): a report site, with no channels (corpusScopeOf). site?: { id?: string; title?: string; url?: string; scope?: string }; // Spec 5: the site's reports, when it publishes any. reports?: { index?: string; count?: number }; }; export type HubCorpusJson = { kind?: string; sites?: HubSite[]; }; // One server-only env read, guarded so this module loads in a browser. The // dynamic `process.env?.[…]` form is deliberately NOT the inlinable // `process.env.NAME` member expression — a page cache budget is a server knob, // and a browser reader takes the default. function envVar(name: string): string | undefined { return typeof process !== "undefined" ? process.env?.[name] : undefined; } // ─── Bounded promise caches ─── // // Two different caching problems, so two different structures: // // manifests — ~551 KB for the whole corpus (29 channels). Small, hot, and // re-read constantly: an un-hinted 20-id get_transcripts batch // used to cost ~300 manifest reads because findVideo walks every // channel per id. Cached OUTRIGHT, no bound. // // pages — ~7.4 MB of raw JSON each, several times that once parsed. An // unbounded map of these is gigabytes, so this is a small LRU. // Its job is the 20-id batch that lands on ONE shared page (20 // reads → 1); it is deliberately NOT sized to hold a scan, which // visits each page exactly once and would only be paying memory // for evictions. // // The bound is a RAW-BYTE budget, not an entry count, because page sizes differ // by an order of magnitude across corpora (a 13-record VOD page is 8 MB; a // shorts channel's page is a fraction of that). Default 48 MB, configurable // with TRANSCRIPT_MCP_PAGE_CACHE_MB (0 disables). // // Why 48: a parsed page retains about 2.7× its file bytes (measured — an // 8.09 MB page holds 21.8 MB of JS heap), so 48 MB of raw budget is roughly // 130 MB resident. That is the most I am willing to hold on a box that also // runs a GPU digest sweep and other agents' jobs. It is ~6 pages of this // corpus, which covers the working set this cache exists for (a 20-id batch // from an enumerate worklist arrives in page order and lands on 1–3 pages). // Note what it deliberately does NOT cover: a full-corpus scan is 170 pages ≈ // 3.7 GB retained, so there is no cache size between "6 pages" and "impossible" // that changes the full-scan story. Filter-first scanning changes that instead. const DEFAULT_PAGE_CACHE_MB = 48; // Always keep at least this many entries, so a corpus whose single page exceeds // the whole budget still caches that page rather than thrashing on it. const MIN_CACHED_PAGES = 2; export function pageCacheBudgetBytes(): number { const raw = envVar("TRANSCRIPT_MCP_PAGE_CACHE_MB"); const mb = raw === undefined || raw.trim() === "" ? DEFAULT_PAGE_CACHE_MB : Number(raw); const safe = Number.isFinite(mb) && mb >= 0 ? mb : DEFAULT_PAGE_CACHE_MB; return Math.floor(safe * 1024 * 1024); } // What a cached loader reports back: the parsed value plus the raw byte size it // was parsed from, which is what the budget is denominated in. export type Sized = { value: T; bytes: number }; // An LRU keyed by string, holding PROMISES rather than values so that N // concurrent callers for the same page coalesce onto one read — the pattern // makeChatFetcher already uses. A rejected promise evicts itself, so a // transient failure is never cached as a permanent one. // // Sizes are only known once a read resolves, so an in-flight entry counts as 0 // and the budget is enforced on resolve. An entry evicted while still in flight // resolves normally for whoever already holds its promise; it just isn't // remembered. export class PageCache { private map = new Map; bytes: number }>(); private total = 0; constructor(private readonly maxBytes: number) {} take(key: string, load: () => Promise>): Promise { const hit = this.map.get(key); if (hit !== undefined) { this.map.delete(key); this.map.set(key, hit); // most-recently used goes last return hit.p; } if (this.maxBytes <= 0) return load().then((s) => s.value); const entry: { p: Promise; bytes: number } = { p: null as never, bytes: 0 }; entry.p = load() .then((s) => { // Only account for it if we're still the live entry for this key — // a refresh() between issue and resolve must not resurrect it. if (this.map.get(key) === entry) { entry.bytes = s.bytes; this.total += s.bytes; this.evict(); } return s.value; }) .catch((e: unknown) => { this.drop(key, entry); throw e; }); this.map.set(key, entry); return entry.p; } private drop(key: string, entry: { bytes: number }): void { if (this.map.get(key) === entry) { this.map.delete(key); this.total -= entry.bytes; } } private evict(): void { while (this.total > this.maxBytes && this.map.size > MIN_CACHED_PAGES) { const oldest = this.map.entries().next().value; if (oldest === undefined) break; this.map.delete(oldest[0]); this.total -= oldest[1].bytes; } } clear(): void { this.map.clear(); this.total = 0; } } // The unbounded sibling, for the small-and-hot caches (manifests). Same // don't-memoise-a-failure rule. export class PromiseMap { private map = new Map>(); take(key: string, load: () => Promise): Promise { const hit = this.map.get(key); if (hit !== undefined) return hit; const p = load().catch((e: unknown) => { this.map.delete(key); throw e; }); this.map.set(key, p); return p; } clear(): void { this.map.clear(); } } // ─── Remote: fetch shards from a deployed site origin over HTTP ─── export class RemoteSource implements ArchiveReader { readonly label: string; private base: string; private aliases?: SearchAlias[]; private tags?: PublishedTag[]; private groups?: ChannelGroups; private index?: Promise>; private duplicates?: Promise; private stats?: Promise>; private subsManifests = new PromiseMap(); private digestManifests = new PromiseMap(); private subsPages: PageCache; private postsManifests = new PromiseMap(); private transcriptManifests = new PromiseMap(); private transcriptPages: PageCache; // What corpus.json said the site publishes, learned by readChannels(). private siteScope?: CorpusScope; private reportIndex?: Promise; private reportPages = new PromiseMap(); // Over HTTP the cost is latency, not parse, so a wider window is the dominant // win — this is where bounded concurrency actually pays. readonly pageConcurrency = 8; // `budgetBytes` lets a hub divide one memory ceiling across its members // instead of granting each member the full budget (N members × 48 MB is not a // budget, it's N budgets). constructor(baseUrl: string, budgetBytes = pageCacheBudgetBytes()) { this.base = baseUrl.replace(/\/+$/, ""); this.label = `remote:${this.base}`; this.subsPages = new PageCache(budgetBytes); this.transcriptPages = new PageCache(budgetBytes); } // The deployed site origin — the archilyzer viewer that owns these videos. publicOrigin(): string | null { return this.base; } private resetCaches(): void { this.subsManifests.clear(); this.digestManifests.clear(); this.subsPages.clear(); this.postsManifests.clear(); this.transcriptManifests.clear(); this.transcriptPages.clear(); this.aliases = undefined; this.groups = undefined; this.index = undefined; this.duplicates = undefined; this.stats = undefined; this.siteScope = undefined; this.reportIndex = undefined; this.reportPages.clear(); } async scope(): Promise { await this.listChannels(); return this.siteScope ?? "full"; } // A 404 is the answer "no reports" (a full site without any, or one built // before spec 5) and is cached; any other failure is not memoised. reports(): Promise { this.reportIndex ??= this.readReportIndex() .then(coerceReportIndex) .catch((e: unknown) => { if (e instanceof ArchiveHttpError && e.status === 404) return []; this.reportIndex = undefined; throw e; }); return this.reportIndex; } reportPage(id: string): Promise { return this.reportPages.take(id, async () => { try { return coerceReportPage( await this.getJson(reportPagePathFor(id), "reportPage"), id, ); } catch (e) { if (e instanceof ArchiveHttpError && e.status === 404) return null; throw e; } }); } subsManifest(ch: ChannelRef): Promise { return this.subsManifests.take(ch.slug, async () => { try { return await this.readChannelSubsManifest(ch.slug); } catch { return null; } }); } subsPage(ch: ChannelRef, page: number): Promise { return this.subsPages.take(`${ch.slug}:${page}`, () => this.getJsonSized(pageUrl("subs", ch.slug, page), "subsPage"), ); } // Cached per channel including the negative answer — otherwise every // posts-covering search costs one 404 per video-only channel. postsManifest(ch: ChannelRef): Promise { return this.postsManifests.take(ch.slug, async () => { try { return await this.readChannelPostsManifest(ch.slug); } catch { return null; } }); } postsPage(ch: ChannelRef, page: number): Promise { return this.getJson(pageUrl("posts", ch.slug, page), "postsPage"); } videoIndex(): Promise { // buildVideoIndex already treats a throw from either reader as absence // (empty index / skipped page), so the raw reads' rejections land exactly // where the old inline `res.ok ? … : null` did. this.index ??= buildVideoIndex( () => this.readSummariesManifest(), (page) => this.readSummariesPage(page), ); return this.index; } availabilityMap(): Promise> { return this.videoIndex(); } digestsManifest(ch: ChannelRef): Promise { return this.digestManifests.take(ch.slug, async () => { try { return await this.readChannelDigestsManifest(ch.slug); } catch { return null; } }); } digestPage(ch: ChannelRef, page: number): Promise { return this.getJson(pageUrl("digests", ch.slug, page), "digestPage"); } duplicateIndex(): Promise { this.duplicates ??= (async () => { try { return buildDuplicateIndex(await this.readDuplicates()); } catch { return buildDuplicateIndex(null); } })(); return this.duplicates; } statsIndex(): Promise> { this.stats ??= buildStatsIndex( () => this.readStatsManifest(), (page) => this.readStatsPage(page), ); return this.stats; } async loadAliases(): Promise { if (this.aliases) return this.aliases; try { this.aliases = coerceAliasConfig(await this.readAliasConfig()).aliases; } catch { this.aliases = []; } return this.aliases; } async loadTags(): Promise { if (this.tags) return this.tags; try { this.tags = coercePublishedTags(await this.readTags()); } catch { this.tags = []; // 404 (no shippable tag, or a pre-spec-4 site) or junk } return this.tags; } async loadGroups(): Promise { if (this.groups) return this.groups; try { this.groups = parseGroupsManifest(await this.readSummariesManifest()); } catch { this.groups = EMPTY_GROUPS; } return this.groups; } // Fetches as TEXT so the byte size is knowable — the page cache's budget is // denominated in raw bytes, and res.json() throws the length away. // // `p` is a ROOT-RELATIVE path from the contract builders; the base is joined // here so the thrown error names the full URL, exactly as before. // // `kind` is the io-stats bucket. It is OPTIONAL, and the reads that pass none // are the ones that never recorded a read before this file owned them (the // summaries/stats/duplicates/alias folds all used a bare `fetch`). Adding // them to the ledger would move the bench's structural counters for a reason // that has nothing to do with the walk, so the blind spots are preserved // deliberately rather than quietly closed. private async getJsonSized( p: string, kind?: string, ): Promise<{ value: T; bytes: number }> { const res = await fetch(`${this.base}${p}`); if (!res.ok) { throw new ArchiveHttpError(res.status, `${this.base}${p}`, res.statusText); } const raw = await res.text(); if (kind !== undefined) recordRead(kind, raw.length); return { value: JSON.parse(raw) as T, bytes: raw.length }; } private async getJson(p: string, kind?: string): Promise { return (await this.getJsonSized(p, kind)).value; } // ─── Raw document reads ─── // // One URL, one fetch, one parse, and a THROW on anything but a 200. Every // tolerance policy is built on top of these — this class's own // empty-when-absent folds below, and the viewer's component caches — so each // published document's URL is written once and each caller picks its own // answer to "absent". // // That split is not cosmetic: a tool scanning 30 channels wants `null` for a // missing posts tree, while the viewer wants a rejection, because react-query // retries a rejected query and will never retry a resolved `null`. Folding // both into one tolerant method is how a transient network blip becomes a // permanently empty panel. readCorpus(): Promise { return this.getJson(rootFileUrl("corpus.json"), "corpus"); } readAliasConfig(): Promise { return this.getJson(rootFileUrl("search-aliases.json")); } readReportIndex(): Promise { return this.getJson(REPORTS_INDEX_PATH, "reportIndex"); } readTags(): Promise { return this.getJson(rootFileUrl(TAGS_FILENAME)); } readDuplicates(): Promise { return this.getJson(rootFileUrl(DUPLICATES_FILENAME)); } readSummariesManifest(): Promise { return this.getJson(manifestUrl("summaries")); } readSummariesPage(page: number): Promise { return this.getJson(pageUrl("summaries", undefined, page)); } readStatsManifest(): Promise { return this.getJson(manifestUrl("stats")); } readStatsPage(page: number): Promise { return this.getJson(pageUrl("stats", undefined, page)); } // The SITE-level index of which channels ship a per-channel tree — a // different document from a channel's own manifest (see treeManifestUrl). readSubsSiteManifest(): Promise { return this.getJson(treeManifestUrl("subs")); } readPostsSiteManifest(): Promise { return this.getJson(treeManifestUrl("posts")); } readChannelSubsManifest(slug: string): Promise { return this.getJson(manifestUrl("subs", slug)); } readChannelPostsManifest(slug: string): Promise { return this.getJson(manifestUrl("posts", slug)); } // 404 → null, everything else → throw. The digest corpus is SPARSE BY DESIGN // (a channel with no digests ships no manifest at all), so the 404 is the // document's own answer and belongs in the raw read; a 500 or a dropped // connection is not proof of absence and must stay distinguishable. async readChannelDigestsManifest( slug: string, ): Promise { const p = manifestUrl("digests", slug); const res = await fetch(`${this.base}${p}`); if (res.status === 404) return null; if (!res.ok) { throw new ArchiveHttpError(res.status, `${this.base}${p}`, res.statusText); } return (await res.json()) as ChannelDigestsManifest; } private channelList?: Promise; listChannels(opts: { refresh?: boolean } = {}): Promise { if (opts.refresh) { this.channelList = undefined; this.resetCaches(); } this.channelList ??= this.readChannels().catch((e: unknown) => { this.channelList = undefined; // don't memoise a failure throw e; }); return this.channelList; } // A CITED site's corpus.json lists no channels (spec 5); its scope is // recorded so a caller can say "a report site" rather than "empty", and a // stray channel entry on one is not read as a corpus. private async readChannels(): Promise { const corpus = await this.readCorpus(); this.siteScope = corpusScopeOf(corpus); if (this.siteScope === "cited") return []; return (corpus.channels ?? []).map((c) => ({ key: c.slug, slug: c.slug, name: c.name ?? c.slug, videoCount: c.videoCount, groupId: c.groupId, siteUrl: this.base, })); } transcriptsManifest(ch: ChannelRef): Promise { return this.transcriptManifests.take(ch.slug, () => this.getJson( manifestUrl("transcripts", ch.slug), "transcriptsManifest", ), ); } transcriptPage(ch: ChannelRef, page: number): Promise { return this.transcriptPages.take(`${ch.slug}:${page}`, () => this.getJsonSized( pageUrl("transcripts", ch.slug, page), "transcriptPage", ), ); } }