// ─── Local: read composed shards from a directory on disk ─── // // THE ONLY FILE UNDER lib/archive/ THAT TOUCHES node:*. reader.ts imports // nothing from here (not even a type), so a browser bundle that reaches the // reader cannot drag `node:fs` in behind it. `pnpm --filter export exec next // build` is what actually proves that — tsc cannot tell a `node:fs` import from // any other, and the export app is the bundle where it would blow up. // // `dir` is a COMPOSED PUBLIC DIR — the output of compose-site.ts, laid out // exactly as the deployed origin serves it. It is not a corpus on disk, so // there is no `config.dataDir` to resolve here: the relocation resolver the // one-core plan asks for belongs to the thing that walks // `channels//data/` (umtool's cues.mjs), not to this. import { readFile, readdir } from "node:fs/promises"; import path from "node:path"; import type { ChannelTranscriptsManifest, ChannelSubsManifest, Manifest, } from "../manifest"; import type { TranscriptDetail, DisplaySummary } from "../transcripts"; import type { SubsDetail } from "../subs"; import type { ChannelPostsManifest, Post } from "../posts"; import { coerceAliasConfig, type SearchAlias } from "../searchAliases"; import { TAGS_FILENAME, type PublishedTag } from "../curatedTags"; import { coercePublishedTags } from "../publishedTags"; import { DUPLICATES_FILENAME, type DuplicateReport } from "../duplicates"; import type { ChannelDigestsManifest, VideoDigest } from "../digests"; import type { StatsManifest, VideoStat } from "../stats"; import { pageFileName } from "./contract"; import { isReportId } from "../report/schema"; import { REPORTS_INDEX_PATH, reportViewPath, type ReportIndexEntry, type ReportPageView, } from "../report/views"; import { recordRead } from "./io-stats"; import { DEFAULT_PAGE_CONCURRENCY, EMPTY_GROUPS, PageCache, PromiseMap, buildDuplicateIndex, buildStatsIndex, buildVideoIndex, coerceReportIndex, coerceReportPage, corpusScopeOf, pageCacheBudgetBytes, parseGroupsManifest, type ArchiveReader, type ChannelGroups, type ChannelRef, type CorpusScope, type DuplicateIndex, type IndexedVideo, type SiteCorpusJson, type VideoAvailability, type VideoIndex, } from "./reader"; // Read a local JSON file, counting its bytes when instrumentation is on, and // reporting the raw size so a byte-budgeted cache can account for it. async function readLocalJsonSized( file: string, kind: string, ): Promise<{ value: T; bytes: number }> { const raw = await readFile(file, "utf8"); recordRead(kind, raw.length); return { value: JSON.parse(raw) as T, bytes: raw.length }; } async function readLocalJson(file: string, kind: string): Promise { return (await readLocalJsonSized(file, kind)).value; } // Opt-out for the composed site's declared origin (see LocalSource.publicOrigin). // Set TRANSCRIPT_PLATFORM_LINKS=1 to cite platform watch pages instead, which is // the right answer when a local build's declared site url is not actually // deployed. const PREFER_PLATFORM_LINKS = process.env.TRANSCRIPT_PLATFORM_LINKS === "1"; // Prefers corpus.json for the channel list (names + counts); falls back to // listing the transcripts/ subdirectories so it works even pre-Layer-1. export class LocalSource implements ArchiveReader { readonly label: string; private aliases?: SearchAlias[]; private tags?: PublishedTag[]; private groups?: ChannelGroups; private index?: Promise>; private duplicates?: Promise; private stats?: Promise>; // The composed site's own declared origin, learned from corpus.json the first // time the channel list is read. Undefined = not looked at yet. private siteOrigin: string | null | undefined; // What corpus.json said the site publishes, learned with the channel list. private siteScope?: CorpusScope; private reportIndex?: Promise; constructor(private dir: string) { this.label = `local:${dir}`; } // The deployed archilyzer viewer these shards were composed for, as declared // by the dir's own corpus.json (`site.url`). A composed public dir is not an // anonymous pile of JSON — it names the site it is the build output of — so // citing that viewer is both possible and the right default: a reader // following a citation lands in the archive, at the cited second, with the // transcript around it, rather than on the platform page where the archive's // whole point (that we still have a copy) is invisible. // // Populated by readChannels(), which every read path runs before it renders a // link. Null when the dir ships no corpus.json (the bare directory-listing // fallback), or when TRANSCRIPT_PLATFORM_LINKS=1 asks for platform links — // both fall back to the platform watch page exactly as before. publicOrigin(): string | null { return this.siteOrigin ?? null; } // Cached INCLUDING the negative answer, exactly like postsManifests below: // live-chat search probes every channel in scope, and a video-only channel // would otherwise cost one failed read per query. private subsManifests = new PromiseMap(); private digestManifests = new PromiseMap(); private subsPages = new PageCache(pageCacheBudgetBytes()); private postsManifests = new PromiseMap(); private transcriptManifests = new PromiseMap(); private transcriptPages = new PageCache(pageCacheBudgetBytes()); // Drop every cached read. Reached only through listChannels({refresh:true}) — // the deliberate, explicit staleness escape hatch for a corpus rebuilt under // a long-lived server. Not a TTL, on purpose. private resetCaches(): void { this.subsManifests.clear(); this.digestManifests.clear(); this.subsPages.clear(); this.postsManifests.clear(); this.transcriptManifests.clear(); this.transcriptPages.clear(); this.aliases = undefined; this.groups = undefined; this.index = undefined; this.duplicates = undefined; this.stats = undefined; this.siteOrigin = undefined; this.siteScope = undefined; this.reportIndex = undefined; } async scope(): Promise { await this.listChannels(); return this.siteScope ?? "full"; } // A composed dir with no reports/index.json publishes none (ENOENT is data, // as for tags.json). reports(): Promise { this.reportIndex ??= readLocalJson( path.join(this.dir, REPORTS_INDEX_PATH), "reportIndex", ) .then(coerceReportIndex) .catch(() => []); return this.reportIndex; } // The id is validated before it touches a path: a report id is a slug, and // anything else (a `..`, a slash) names no report rather than another file. async reportPage(id: string): Promise { if (!isReportId(id)) return null; let raw: unknown; try { raw = await readLocalJson( path.join(this.dir, reportViewPath(id)), "reportPage", ); } catch (e) { if ((e as NodeJS.ErrnoException).code === "ENOENT") return null; throw e; } return coerceReportPage(raw, id); } subsManifest(ch: ChannelRef): Promise { return this.subsManifests.take(ch.slug, () => readLocalJson( path.join(this.dir, "subs", ch.slug, "manifest.json"), "subsManifest", ).catch(() => null), // channel ships no subs shards ); } subsPage(ch: ChannelRef, page: number): Promise { return this.subsPages.take(`${ch.slug}:${page}`, () => readLocalJsonSized( path.join(this.dir, "subs", ch.slug, pageFileName(page)), "subsPage", ), ); } // Cached per channel INCLUDING the negative answer: most channels are // video-only, and a posts-covering search would otherwise re-probe every one // of them on every query. postsManifest(ch: ChannelRef): Promise { return this.postsManifests.take(ch.slug, () => readLocalJson( path.join(this.dir, "posts", ch.slug, "manifest.json"), "postsManifest", ).catch(() => null), // channel ships no posts shards ); } async postsPage(ch: ChannelRef, page: number): Promise { return readLocalJson( path.join(this.dir, "posts", ch.slug, pageFileName(page)), "postsPage", ); } async loadAliases(): Promise { if (this.aliases) return this.aliases; try { this.aliases = coerceAliasConfig( await readLocalJson(path.join(this.dir, "search-aliases.json"), "aliases"), ).aliases; } catch { this.aliases = []; // no/invalid file — search stays plain } return this.aliases; } // A composed public dir either carries /tags.json or it does not — the same // 404-is-data rule the remote reader follows, just as ENOENT. async loadTags(): Promise { if (this.tags) return this.tags; try { this.tags = coercePublishedTags( await readLocalJson(path.join(this.dir, TAGS_FILENAME), "tags"), ); } catch { this.tags = []; // no/invalid file — this build publishes no tags } return this.tags; } async loadGroups(): Promise { if (this.groups) return this.groups; try { this.groups = parseGroupsManifest( await readLocalJson( path.join(this.dir, "summaries", "manifest.json"), "summariesManifest", ), ); } catch { this.groups = EMPTY_GROUPS; // no/invalid manifest — groups off } return this.groups; } private channelList?: Promise; listChannels(opts: { refresh?: boolean } = {}): Promise { if (opts.refresh) { this.channelList = undefined; this.resetCaches(); } this.channelList ??= this.readChannels().catch((e: unknown) => { this.channelList = undefined; // don't memoise a failure throw e; }); return this.channelList; } private async readChannels(): Promise { try { const corpus = await readLocalJson( path.join(this.dir, "corpus.json"), "corpus", ); const declared = corpus.site?.url?.trim(); this.siteOrigin = declared && !PREFER_PLATFORM_LINKS ? declared.replace(/\/+$/, "") : null; // A CITED site (spec 5) has no channels by design: never fall back to // listing a transcripts/ directory a stale public dir might still hold. this.siteScope = corpusScopeOf(corpus); if (this.siteScope === "cited") return []; if (Array.isArray(corpus.channels) && corpus.channels.length > 0) { return corpus.channels.map((c) => ({ key: c.slug, slug: c.slug, name: c.name ?? c.slug, videoCount: c.videoCount, groupId: c.groupId, })); } } catch { // no corpus.json — fall back to a directory listing } const transcriptsDir = path.join(this.dir, "transcripts"); let entries: string[] = []; try { entries = await readdir(transcriptsDir); } catch { return []; } const channels: ChannelRef[] = []; for (const slug of entries.sort()) { // A channel dir has a manifest.json; skip stray files. try { await readFile(path.join(transcriptsDir, slug, "manifest.json"), "utf8"); channels.push({ key: slug, slug, name: slug }); } catch { // not a channel dir } } return channels; } transcriptsManifest(ch: ChannelRef): Promise { return this.transcriptManifests.take(ch.slug, () => readLocalJson( path.join(this.dir, "transcripts", ch.slug, "manifest.json"), "transcriptsManifest", ), ); } transcriptPage(ch: ChannelRef, page: number): Promise { return this.transcriptPages.take(`${ch.slug}:${page}`, () => readLocalJsonSized( path.join(this.dir, "transcripts", ch.slug, pageFileName(page)), "transcriptPage", ), ); } // Local reads are CPU-bound on JSON.parse, so a modest window is all that is // available to win: it overlaps the next page's read with this page's parse. readonly pageConcurrency = DEFAULT_PAGE_CONCURRENCY; videoIndex(): Promise { this.index ??= buildVideoIndex( () => readLocalJson( path.join(this.dir, "summaries", "manifest.json"), "summariesManifest", ), (page) => readLocalJson( path.join(this.dir, "summaries", pageFileName(page)), "summariesPage", ), ); return this.index; } availabilityMap(): Promise> { return this.videoIndex(); } digestsManifest(ch: ChannelRef): Promise { return this.digestManifests.take(ch.slug, () => readLocalJson( path.join(this.dir, "digests", ch.slug, "manifest.json"), "digestsManifest", ).catch(() => null), // channel has no digests ); } digestPage(ch: ChannelRef, page: number): Promise { return readLocalJson( path.join(this.dir, "digests", ch.slug, pageFileName(page)), "digestPage", ); } duplicateIndex(): Promise { this.duplicates ??= readLocalJson( path.join(this.dir, DUPLICATES_FILENAME), "duplicates", ) .then(buildDuplicateIndex) .catch(() => buildDuplicateIndex(null)); // no report shipped — no clusters return this.duplicates; } statsIndex(): Promise> { this.stats ??= buildStatsIndex( () => readLocalJson( path.join(this.dir, "stats", "manifest.json"), "statsManifest", ), (page) => readLocalJson( path.join(this.dir, "stats", pageFileName(page)), "statsPage", ), ); return this.stats; } }