commit ea2d3d04d126af72cf52b77dfe0d48a78e791d88 parent f4553b29691883148d64b3eb687ccbee2816e83d Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st> Date: Sat, 12 Sep 2026 02:27:45 -0400 merge: one-core/phase-2-s1 — the contract and one ArchiveReader under common/lib/archive S1 of Phase 2, reviewed clean: mcp's reader moved into common/lib/archive with the node imports isolated in reader-fs.ts, the URL walk as code that buildSiteCorpus now calls, the hub entry types collapsed per direction, the lib/search stubs, and the composed fixture under plans/tools. Gates on the slice tip: tsc clean, common 1077, mcp 205, scripts 71/1, export build clean, bench structural counters identical, export e2e 172/172, hub 5/5. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Diffstat:
43 files changed, 2885 insertions(+), 1455 deletions(-)
diff --git a/common/lib/archive/contract.test.ts b/common/lib/archive/contract.test.ts @@ -0,0 +1,168 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + CONTRACT, + ROOT_FILES, + archiveUrl, + corpusUrl, + isFlatTree, + manifestUrl, + pageFileName, + pageUrl, + rootFileUrl, + shipsPwa, +} from "./contract"; +import { buildSiteCorpus } from "../corpus"; +import type { PublicSiteDescriptor } from "../siteDescriptor"; + +// A minimal descriptor with two channels, one of which ships posts and digests, +// so every conditional manifest pointer buildSiteCorpus can emit is exercised. +function descriptor(siteUrl?: string): PublicSiteDescriptor { + return { + contract: CONTRACT.siteDescriptor, + siteId: "testsite", + siteTitle: "Test Site", + siteDescription: "fixture", + generatedAt: "2026-09-12T00:00:00.000Z", + ...(siteUrl ? { siteUrl } : {}), + channels: [ + { slug: "alpha", name: "Alpha", count: 3 }, + { slug: "beta", name: "Beta", count: 2, groupId: "g1" }, + ], + groups: [], + defaultGroupId: "default", + socialLinks: [], + pwa: false, + } as unknown as PublicSiteDescriptor; +} + +test("page shards are zero-padded to four — page-0.json is a 404", () => { + assert.equal(pageFileName(0), "page-0000.json"); + assert.equal(pageFileName(7), "page-0007.json"); + assert.equal(pageFileName(1234), "page-1234.json"); + assert.equal(pageFileName(12345), "page-12345.json"); + assert.equal(CONTRACT.pagePad, 4); +}); + +test("archiveUrl: absolute with a base, root-relative without", () => { + assert.equal(archiveUrl(undefined, "/corpus.json"), "/corpus.json"); + assert.equal(archiveUrl("https://x.example", "/corpus.json"), "https://x.example/corpus.json"); + // A trailing slash on the base must not double up. + assert.equal(archiveUrl("https://x.example/", "/corpus.json"), "https://x.example/corpus.json"); + assert.equal(archiveUrl("https://x.example///", "/a/b.json"), "https://x.example/a/b.json"); +}); + +test("per-channel trees take a slug, flat trees do not", () => { + assert.equal(isFlatTree("summaries"), true); + assert.equal(isFlatTree("stats"), true); + assert.equal(isFlatTree("transcripts"), false); + + assert.equal(manifestUrl("transcripts", "alpha"), "/transcripts/alpha/manifest.json"); + assert.equal(manifestUrl("summaries"), "/summaries/manifest.json"); + assert.equal(manifestUrl("stats"), "/stats/manifest.json"); + assert.equal(pageUrl("subs", "alpha", 2), "/subs/alpha/page-0002.json"); + assert.equal(pageUrl("summaries", undefined, 2), "/summaries/page-0002.json"); + assert.equal(pageUrl("stats", undefined, 0), "/stats/page-0000.json"); + + // A per-channel tree with no slug is a bug at the call site, not a URL that + // 404s quietly against a real site. + assert.throws(() => manifestUrl("transcripts"), /needs a channel slug/); + assert.throws(() => pageUrl("posts", undefined, 1), /needs a channel slug/); +}); + +test("root files: the four JSON documents a reader fetches by name", () => { + assert.deepEqual([...ROOT_FILES], [ + "corpus.json", + "site.json", + "search-aliases.json", + "duplicates.json", + ]); + assert.equal(corpusUrl(), "/corpus.json"); + assert.equal(corpusUrl("https://x.example"), "https://x.example/corpus.json"); + assert.equal(rootFileUrl("duplicates.json"), "/duplicates.json"); + assert.equal( + rootFileUrl("search-aliases.json", "https://x.example"), + "https://x.example/search-aliases.json", + ); +}); + +// THE reason these functions exist: what a reader builds must be byte-for-byte +// what the writer emitted. Compare against buildSiteCorpus's actual output +// rather than against a second hand-written string. +test("manifestUrl reproduces what buildSiteCorpus emits — absolute, with a siteUrl", () => { + const base = "https://testsite.example"; + const corpus = buildSiteCorpus(descriptor(base), { + hasArchives: false, + postCounts: { beta: 4 }, + digestCounts: { beta: 1 }, + }); + const alpha = corpus.channels[0]; + const beta = corpus.channels[1]; + + assert.equal(alpha.manifests.transcripts, manifestUrl("transcripts", "alpha", base)); + assert.equal(alpha.manifests.subs, manifestUrl("subs", "alpha", base)); + assert.equal(beta.manifests.posts, manifestUrl("posts", "beta", base)); + assert.equal(beta.manifests.digests, manifestUrl("digests", "beta", base)); + assert.equal(alpha.manifests.transcripts, `${base}/transcripts/alpha/manifest.json`); + + // A channel with no posts/digests advertises neither pointer — the layer is + // absent from the wire, not present-and-empty. + assert.equal(alpha.manifests.posts, undefined); + assert.equal(alpha.manifests.digests, undefined); +}); + +test("manifestUrl reproduces what buildSiteCorpus emits — root-relative, no siteUrl", () => { + const corpus = buildSiteCorpus(descriptor(), { hasArchives: false }); + const alpha = corpus.channels[0]; + + assert.equal(alpha.manifests.transcripts, manifestUrl("transcripts", "alpha")); + assert.equal(alpha.manifests.transcripts, "/transcripts/alpha/manifest.json"); + assert.equal(alpha.manifests.subs, "/subs/alpha/manifest.json"); + assert.equal(corpus.site.url, undefined); +}); + +test("the hub corpus's per-member pointers go through the same builders", () => { + const corpus = buildSiteCorpus(descriptor("https://a.example/"), { + hasArchives: false, + }); + // buildSiteCorpus strips the trailing slash exactly as archiveUrl does. + assert.equal( + corpus.channels[0].manifests.transcripts, + "https://a.example/transcripts/alpha/manifest.json", + ); +}); + +test("shipsPwa: the site flag, or hub mode", () => { + const prev = process.env.INSTANCE_MODE; + try { + delete process.env.INSTANCE_MODE; + assert.equal(shipsPwa({}), false); + assert.equal(shipsPwa({ pwa: false }), false); + assert.equal(shipsPwa({ pwa: true }), true); + process.env.INSTANCE_MODE = "hub"; + // Hub mode always ships the PWA, whatever the site says. + assert.equal(shipsPwa({}), true); + assert.equal(shipsPwa({ pwa: false }), true); + process.env.INSTANCE_MODE = "site"; + assert.equal(shipsPwa({}), false); + } finally { + if (prev === undefined) delete process.env.INSTANCE_MODE; + else process.env.INSTANCE_MODE = prev; + } +}); + +test("CONTRACT is frozen where it is published", () => { + // These are on the wire. A change here is a change to every deployed archive's + // machine contract, so it belongs in a slice that says so, not in a refactor. + assert.equal(CONTRACT.corpusSpec, 3); + assert.equal(CONTRACT.siteDescriptor, 1); + assert.equal(CONTRACT.manifest, 3); + assert.equal(CONTRACT.transcriptsManifest, 1); + assert.equal(CONTRACT.subsManifest, 4); + assert.equal(CONTRACT.subsChannelManifest, 1); + assert.equal(CONTRACT.summariesPageSize, 1000); + assert.deepEqual( + [...CONTRACT.layers], + ["transcripts", "subs", "posts", "digests", "summaries"], + ); +}); diff --git a/common/lib/archive/contract.ts b/common/lib/archive/contract.ts @@ -0,0 +1,197 @@ +// THE PUBLISHED CONTRACT, as code. +// +// A published archive is a pile of static JSON whose *shape* is a promise: +// `/corpus.json` names every channel and how to walk manifest -> slugToPage -> +// `page-<NNNN>.json`. Phase 0 collapsed the version constants into one +// `CONTRACT` object; this module is the next step — the URL SHAPES that go with +// them, so a reader and the writer that produced the files agree by +// construction rather than by two people editing the same string twice. +// +// Browser-safe on purpose: every consumer of the contract (the viewer's caches, +// the offline cache, the MCP reader, a project's cue resolver) imports from +// here, and half of them run in a browser. Nothing in this file touches +// `node:*`. +// +// WHY THIS FILE OWNS `CONTRACT` AND `pageFileName` RATHER THAN RE-EXPORTING +// THEM: `lib/corpus.ts` must CALL the URL builders (one definition of the shape +// `buildSiteCorpus` emits), and `lib/manifest.ts` reads `CONTRACT.pagePad` at +// module scope. Importing `CONTRACT` from `corpus.ts` here would close the loop +// corpus -> archive/contract -> manifest -> corpus, and the TDZ read of +// `CONTRACT.manifest` at `manifest.ts`'s top level makes that cycle a hard +// ReferenceError, not a warning. So the contract module sits at the BOTTOM of +// the stack and `corpus.ts` / `manifest.ts` re-export from it: every existing +// import site (`from "./corpus"`, `from "./manifest"`) is unchanged, and the +// dependency graph is a DAG. + +import { DUPLICATES_FILENAME } from "../duplicates"; + +// The machine-readable versions and constants of the published contract. Every +// value here appears on the wire, so a change is a wire change — see the +// per-field notes in lib/corpus.ts for what a bump means to a client. +export const CONTRACT = { + // /corpus.json's own `spec`. `generator` is deliberately unversioned; see the + // note in lib/corpus.ts for why a credit line did not bump the spec. + corpusSpec: 3, + // /site.json's `contract` (siteDescriptor.ts). + siteDescriptor: 1, + // The four manifest versions (manifest.ts). Each is the version field of one + // served document; they move independently and always have. + manifest: 3, + transcriptsManifest: 1, + subsManifest: 4, + subsChannelManifest: 1, + // Records per /summaries/page-NNNN.json. + summariesPageSize: 1000, + // The zero-padding on every page shard's file name. `page-0.json` is a 404 — + // this is the whole reason the constant exists in one place. + pagePad: 4, + // The served trees that follow the manifest -> slugToPage -> page-NNNN walk. + // "summaries" is flat (one manifest, pages, no per-channel level). + layers: ["transcripts", "subs", "posts", "digests", "summaries"], +} as const; + +export type ContractLayer = (typeof CONTRACT.layers)[number]; + +// `stats/` follows the same manifest -> page walk but is NOT a contract layer: +// corpus.json's shardScheme does not document it, so a client that only knows +// the contract must not be told to expect it. It still needs URLs, so the URL +// builders take the wider ArchiveTree and CONTRACT.layers stays frozen. +export type ArchiveTree = ContractLayer | "stats"; + +// The trees with no per-channel level: one manifest at the tree root and pages +// beside it. Everything else is /<tree>/<slug>/…. +const FLAT_TREES: ReadonlySet<string> = new Set(["summaries", "stats"]); + +export function isFlatTree(tree: ArchiveTree): boolean { + return FLAT_TREES.has(tree); +} + +// THE page-shard file name, for every layer. `page-0.json` is a 404 on every +// published archive, so a copy that lost the padding would 404 silently against +// a real site and pass every unit test. lib/manifest.ts re-exports this (and +// the per-layer aliases with it), so the six historical copies stay one. +export function pageFileName(index: number): string { + return `page-${String(index).padStart(CONTRACT.pagePad, "0")}.json`; +} + +// Join an origin base with a root-relative path. When no base is known (a site +// built without a configured siteUrl) the path is left root-relative — still +// correct for a same-origin fetch, just not portable cross-origin. +// +// This is the `join` buildSiteCorpus has always used; every URL builder below +// goes through it, which is what makes "absolute when siteUrl is set, +// root-relative otherwise" one rule instead of a dozen call sites. +export function archiveUrl(base: string | undefined, p: string): string { + if (!base) return p; + return `${base.replace(/\/+$/, "")}${p}`; +} + +// The root-level documents a reader fetches by name. Deliberately only the JSON +// an ArchiveReader (or the viewer's offline cache) actually reads — llms.txt, +// robots.txt and sitemap.xml are human/crawler surfaces with no reader. +// +// duplicates.json is the one that is legitimately absent: compose-site only +// writes it when there is at least one publishable cluster, and corpus.json +// does not declare it, so a 404 here is data, not an error. +export const ROOT_FILES = [ + "corpus.json", + "site.json", + "search-aliases.json", + DUPLICATES_FILENAME, +] as const; + +export type RootFile = (typeof ROOT_FILES)[number]; + +export function rootFileUrl(file: RootFile, base?: string): string { + return archiveUrl(base, `/${file}`); +} + +export function corpusUrl(base?: string): string { + return rootFileUrl("corpus.json", base); +} + +// A tree's manifest URL. `slug` is required for the per-channel trees and +// ignored for the flat ones (summaries, stats). +export function manifestUrl( + tree: ArchiveTree, + slug?: string, + base?: string, +): string { + if (isFlatTree(tree)) return archiveUrl(base, `/${tree}/manifest.json`); + if (!slug) throw new Error(`manifestUrl(${tree}) needs a channel slug`); + return archiveUrl(base, `/${tree}/${slug}/manifest.json`); +} + +// One page shard of a tree. Same slug rule as manifestUrl. +export function pageUrl( + tree: ArchiveTree, + slug: string | undefined, + page: number, + base?: string, +): string { + const file = pageFileName(page); + if (isFlatTree(tree)) return archiveUrl(base, `/${tree}/${file}`); + if (!slug) throw new Error(`pageUrl(${tree}) needs a channel slug`); + return archiveUrl(base, `/${tree}/${slug}/${file}`); +} + +// Whether a build ships an installable PWA — the "dangerous permissions" +// surface: a service worker, a web manifest, and installability. An axis +// INDEPENDENT of the shell: +// - hub mode always ships the PWA (Archilyzer IS the installable app); +// - site mode is a dumb instance by default (federatable JSON only, not +// installable) and opts in per site via the `pwa` config flag. +// +// Was two copies with "keep in sync" comments on each (compose-site.ts and +// export/app/lib/mode.ts); this is the one. `process.env.INSTANCE_MODE` is read +// here exactly as both copies read it — Next inlines that member expression +// into the client bundle at build time, so a `typeof process` guard around it +// would turn hub mode OFF in the browser rather than make it safer. +export function shipsPwa(site: { pwa?: boolean }): boolean { + return site.pwa === true || process.env.INSTANCE_MODE === "hub"; +} + +// ─── The hub's member entry, once ─── +// +// A hub member travelled under four names: `HubSiteEntry` (compose-hub.ts, the +// built-in pool it writes), `HubMemberInput` (buildHubCorpus's input), +// `HubCorpusSite` (the entry it emits into the hub corpus.json) and `HubSite` +// (what mcp's HubSource reads back out of that same file). +// +// They collapse to TWO types, not one, and the reason is on the wire: the input +// spells a member `siteTitle`/`siteUrl` and the emitted document spells it +// `title`/`url` with two derived pointers beside it. corpus.json is frozen, so +// the emitted shape cannot be renamed to match the input shape. What does +// collapse is each DIRECTION: one input type (was two) and one emitted type, +// with the read-back spelling now a Pick of the emitted one so the two can +// never drift. + +// A member site as the hub is TOLD about it: the pool entry compose-hub writes +// and the input buildHubCorpus maps. `accent`, `hubUrl` and `contract` are only +// carried by the pool entry (the client's coerceBuiltin reads them); a member +// without a `siteUrl` cannot be federated and is dropped by both consumers. +export type HubMemberInput = { + siteId: string; + siteTitle: string; + siteUrl: string; + pwa?: boolean; + accent?: string; + hubUrl?: string; + contract?: number; +}; + +// A member site as the hub PUBLISHES it, in the `sites` array of a hub +// corpus.json. Every field is on the wire. +export type HubCorpusSite = { + siteId: string; + title: string; + url: string; + corpus: string; + siteJson: string; + pwa: boolean; +}; + +// What a reader needs off that entry to reach a member: its identity and its +// origin. A Pick rather than a restatement, so adding a field to the published +// entry cannot leave a second copy of its name behind. +export type HubSite = Pick<HubCorpusSite, "siteId" | "title" | "url">; diff --git a/common/lib/archive/io-stats.ts b/common/lib/archive/io-stats.ts @@ -0,0 +1,40 @@ +// ─── I/O instrumentation (opt-in, for mcp/bench) ─── +// +// A process-wide counter of shard reads and parsed bytes, so the benchmark can +// report the STRUCTURAL cost of a query (how many pages, how many bytes) next +// to its wall time. That matters on a shared box specifically: wall time is only +// meaningful when the machine is idle, but read counts and byte counts are +// properties of the query plan and hold under any load. +// +// Off unless MCP_IO_STATS=1, and even then it is two integer adds per read. +// +// Moved here from mcp/src/source.ts with one change: the enable flag is guarded +// so this module can be imported from a browser bundle. `process.env?.X` behind +// a `typeof process` check is a dynamic read Next does NOT inline, which is +// exactly right for a server-only diagnostic — the browser gets `undefined` and +// the counters stay off, rather than the bundle failing on a missing global. + +export type IoStats = { reads: number; bytes: number }; + +const IO_STATS_ON = + typeof process !== "undefined" && process.env?.MCP_IO_STATS === "1"; + +const ioTotals: Record<string, IoStats> = {}; + +export function recordRead(kind: string, bytes: number): void { + if (!IO_STATS_ON) return; + const slot = (ioTotals[kind] ??= { reads: 0, bytes: 0 }); + slot.reads++; + slot.bytes += bytes; +} + +// A snapshot of every counter so far, for diffing across one tool call. +export function ioStatsSnapshot(): Record<string, IoStats> { + const out: Record<string, IoStats> = {}; + for (const [k, v] of Object.entries(ioTotals)) out[k] = { ...v }; + return out; +} + +export function ioStatsEnabled(): boolean { + return IO_STATS_ON; +} diff --git a/common/lib/archive/reader-fs.ts b/common/lib/archive/reader-fs.ts @@ -0,0 +1,335 @@ +// ─── Local: read composed shards from a directory on disk ─── +// +// THE ONLY FILE UNDER lib/archive/ THAT TOUCHES node:*. reader.ts imports +// nothing from here (not even a type), so a browser bundle that reaches the +// reader cannot drag `node:fs` in behind it. `pnpm --filter export exec next +// build` is what actually proves that — tsc cannot tell a `node:fs` import from +// any other, and the export app is the bundle where it would blow up. +// +// `dir` is a COMPOSED PUBLIC DIR — the output of compose-site.ts, laid out +// exactly as the deployed origin serves it. It is not a corpus on disk, so +// there is no `config.dataDir` to resolve here: the relocation resolver the +// one-core plan asks for belongs to the thing that walks +// `channels/<slug>/data/` (umtool's cues.mjs), not to this. + +import { readFile, readdir } from "node:fs/promises"; +import path from "node:path"; +import type { + ChannelTranscriptsManifest, + ChannelSubsManifest, + Manifest, +} from "../manifest"; +import type { TranscriptDetail, DisplaySummary } from "../transcripts"; +import type { SubsDetail } from "../subs"; +import type { ChannelPostsManifest, Post } from "../posts"; +import { coerceAliasConfig, type SearchAlias } from "../searchAliases"; +import { DUPLICATES_FILENAME, type DuplicateReport } from "../duplicates"; +import type { ChannelDigestsManifest, VideoDigest } from "../digests"; +import type { StatsManifest, VideoStat } from "../stats"; +import { pageFileName } from "./contract"; +import { recordRead } from "./io-stats"; +import { + DEFAULT_PAGE_CONCURRENCY, + EMPTY_GROUPS, + PageCache, + PromiseMap, + buildDuplicateIndex, + buildStatsIndex, + buildVideoIndex, + pageCacheBudgetBytes, + parseGroupsManifest, + type ArchiveReader, + type ChannelGroups, + type ChannelRef, + type DuplicateIndex, + type IndexedVideo, + type SiteCorpusJson, + type VideoAvailability, + type VideoIndex, +} from "./reader"; + +// Read a local JSON file, counting its bytes when instrumentation is on, and +// reporting the raw size so a byte-budgeted cache can account for it. +async function readLocalJsonSized<T>( + file: string, + kind: string, +): Promise<{ value: T; bytes: number }> { + const raw = await readFile(file, "utf8"); + recordRead(kind, raw.length); + return { value: JSON.parse(raw) as T, bytes: raw.length }; +} + +async function readLocalJson<T>(file: string, kind: string): Promise<T> { + return (await readLocalJsonSized<T>(file, kind)).value; +} + +// Opt-out for the composed site's declared origin (see LocalSource.publicOrigin). +// Set TRANSCRIPT_PLATFORM_LINKS=1 to cite platform watch pages instead, which is +// the right answer when a local build's declared site url is not actually +// deployed. +const PREFER_PLATFORM_LINKS = process.env.TRANSCRIPT_PLATFORM_LINKS === "1"; + +// Prefers corpus.json for the channel list (names + counts); falls back to +// listing the transcripts/ subdirectories so it works even pre-Layer-1. +export class LocalSource implements ArchiveReader { + readonly label: string; + private aliases?: SearchAlias[]; + private groups?: ChannelGroups; + private index?: Promise<Map<string, IndexedVideo>>; + private duplicates?: Promise<DuplicateIndex>; + private stats?: Promise<ReadonlyMap<string, VideoStat>>; + // The composed site's own declared origin, learned from corpus.json the first + // time the channel list is read. Undefined = not looked at yet. + private siteOrigin: string | null | undefined; + + constructor(private dir: string) { + this.label = `local:${dir}`; + } + + // The deployed archilyzer viewer these shards were composed for, as declared + // by the dir's own corpus.json (`site.url`). A composed public dir is not an + // anonymous pile of JSON — it names the site it is the build output of — so + // citing that viewer is both possible and the right default: a reader + // following a citation lands in the archive, at the cited second, with the + // transcript around it, rather than on the platform page where the archive's + // whole point (that we still have a copy) is invisible. + // + // Populated by readChannels(), which every read path runs before it renders a + // link. Null when the dir ships no corpus.json (the bare directory-listing + // fallback), or when TRANSCRIPT_PLATFORM_LINKS=1 asks for platform links — + // both fall back to the platform watch page exactly as before. + publicOrigin(): string | null { + return this.siteOrigin ?? null; + } + + // Cached INCLUDING the negative answer, exactly like postsManifests below: + // live-chat search probes every channel in scope, and a video-only channel + // would otherwise cost one failed read per query. + private subsManifests = new PromiseMap<ChannelSubsManifest | null>(); + private digestManifests = new PromiseMap<ChannelDigestsManifest | null>(); + private subsPages = new PageCache<SubsDetail[]>(pageCacheBudgetBytes()); + private postsManifests = new PromiseMap<ChannelPostsManifest | null>(); + private transcriptManifests = new PromiseMap<ChannelTranscriptsManifest>(); + private transcriptPages = new PageCache<TranscriptDetail[]>(pageCacheBudgetBytes()); + + // Drop every cached read. Reached only through listChannels({refresh:true}) — + // the deliberate, explicit staleness escape hatch for a corpus rebuilt under + // a long-lived server. Not a TTL, on purpose. + private resetCaches(): void { + this.subsManifests.clear(); + this.digestManifests.clear(); + this.subsPages.clear(); + this.postsManifests.clear(); + this.transcriptManifests.clear(); + this.transcriptPages.clear(); + this.aliases = undefined; + this.groups = undefined; + this.index = undefined; + this.duplicates = undefined; + this.stats = undefined; + this.siteOrigin = undefined; + } + + subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { + return this.subsManifests.take(ch.slug, () => + readLocalJson<ChannelSubsManifest>( + path.join(this.dir, "subs", ch.slug, "manifest.json"), + "subsManifest", + ).catch(() => null), // channel ships no subs shards + ); + } + + subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { + return this.subsPages.take(`${ch.slug}:${page}`, () => + readLocalJsonSized<SubsDetail[]>( + path.join(this.dir, "subs", ch.slug, pageFileName(page)), + "subsPage", + ), + ); + } + + // Cached per channel INCLUDING the negative answer: most channels are + // video-only, and a posts-covering search would otherwise re-probe every one + // of them on every query. + postsManifest(ch: ChannelRef): Promise<ChannelPostsManifest | null> { + return this.postsManifests.take(ch.slug, () => + readLocalJson<ChannelPostsManifest>( + path.join(this.dir, "posts", ch.slug, "manifest.json"), + "postsManifest", + ).catch(() => null), // channel ships no posts shards + ); + } + + async postsPage(ch: ChannelRef, page: number): Promise<Post[]> { + return readLocalJson<Post[]>( + path.join(this.dir, "posts", ch.slug, pageFileName(page)), + "postsPage", + ); + } + + async loadAliases(): Promise<SearchAlias[]> { + if (this.aliases) return this.aliases; + try { + this.aliases = coerceAliasConfig( + await readLocalJson(path.join(this.dir, "search-aliases.json"), "aliases"), + ).aliases; + } catch { + this.aliases = []; // no/invalid file — search stays plain + } + return this.aliases; + } + + async loadGroups(): Promise<ChannelGroups> { + if (this.groups) return this.groups; + try { + this.groups = parseGroupsManifest( + await readLocalJson( + path.join(this.dir, "summaries", "manifest.json"), + "summariesManifest", + ), + ); + } catch { + this.groups = EMPTY_GROUPS; // no/invalid manifest — groups off + } + return this.groups; + } + + private channelList?: Promise<ChannelRef[]>; + + listChannels(opts: { refresh?: boolean } = {}): Promise<ChannelRef[]> { + if (opts.refresh) { + this.channelList = undefined; + this.resetCaches(); + } + this.channelList ??= this.readChannels().catch((e: unknown) => { + this.channelList = undefined; // don't memoise a failure + throw e; + }); + return this.channelList; + } + + private async readChannels(): Promise<ChannelRef[]> { + try { + const corpus = await readLocalJson<SiteCorpusJson>( + path.join(this.dir, "corpus.json"), + "corpus", + ); + const declared = corpus.site?.url?.trim(); + this.siteOrigin = + declared && !PREFER_PLATFORM_LINKS ? declared.replace(/\/+$/, "") : null; + if (Array.isArray(corpus.channels) && corpus.channels.length > 0) { + return corpus.channels.map((c) => ({ + key: c.slug, + slug: c.slug, + name: c.name ?? c.slug, + videoCount: c.videoCount, + groupId: c.groupId, + })); + } + } catch { + // no corpus.json — fall back to a directory listing + } + const transcriptsDir = path.join(this.dir, "transcripts"); + let entries: string[] = []; + try { + entries = await readdir(transcriptsDir); + } catch { + return []; + } + const channels: ChannelRef[] = []; + for (const slug of entries.sort()) { + // A channel dir has a manifest.json; skip stray files. + try { + await readFile(path.join(transcriptsDir, slug, "manifest.json"), "utf8"); + channels.push({ key: slug, slug, name: slug }); + } catch { + // not a channel dir + } + } + return channels; + } + + transcriptsManifest(ch: ChannelRef): Promise<ChannelTranscriptsManifest> { + return this.transcriptManifests.take(ch.slug, () => + readLocalJson<ChannelTranscriptsManifest>( + path.join(this.dir, "transcripts", ch.slug, "manifest.json"), + "transcriptsManifest", + ), + ); + } + + transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]> { + return this.transcriptPages.take(`${ch.slug}:${page}`, () => + readLocalJsonSized<TranscriptDetail[]>( + path.join(this.dir, "transcripts", ch.slug, pageFileName(page)), + "transcriptPage", + ), + ); + } + + // Local reads are CPU-bound on JSON.parse, so a modest window is all that is + // available to win: it overlaps the next page's read with this page's parse. + readonly pageConcurrency = DEFAULT_PAGE_CONCURRENCY; + + videoIndex(): Promise<VideoIndex> { + this.index ??= buildVideoIndex( + () => + readLocalJson<Manifest>( + path.join(this.dir, "summaries", "manifest.json"), + "summariesManifest", + ), + (page) => + readLocalJson<DisplaySummary[]>( + path.join(this.dir, "summaries", pageFileName(page)), + "summariesPage", + ), + ); + return this.index; + } + + availabilityMap(): Promise<ReadonlyMap<string, VideoAvailability>> { + return this.videoIndex(); + } + + digestsManifest(ch: ChannelRef): Promise<ChannelDigestsManifest | null> { + return this.digestManifests.take(ch.slug, () => + readLocalJson<ChannelDigestsManifest>( + path.join(this.dir, "digests", ch.slug, "manifest.json"), + "digestsManifest", + ).catch(() => null), // channel has no digests + ); + } + + digestPage(ch: ChannelRef, page: number): Promise<VideoDigest[]> { + return readLocalJson<VideoDigest[]>( + path.join(this.dir, "digests", ch.slug, pageFileName(page)), + "digestPage", + ); + } + + duplicateIndex(): Promise<DuplicateIndex> { + this.duplicates ??= readLocalJson<DuplicateReport>( + path.join(this.dir, DUPLICATES_FILENAME), + "duplicates", + ) + .then(buildDuplicateIndex) + .catch(() => buildDuplicateIndex(null)); // no report shipped — no clusters + return this.duplicates; + } + + statsIndex(): Promise<ReadonlyMap<string, VideoStat>> { + this.stats ??= buildStatsIndex( + () => + readLocalJson<StatsManifest>( + path.join(this.dir, "stats", "manifest.json"), + "statsManifest", + ), + (page) => + readLocalJson<VideoStat[]>( + path.join(this.dir, "stats", pageFileName(page)), + "statsPage", + ), + ); + return this.stats; + } +} diff --git a/common/lib/archive/reader-hub.ts b/common/lib/archive/reader-hub.ts @@ -0,0 +1,316 @@ +// ─── Hub: federate over every member site listed in the hub corpus.json ─── +// +// Each member is its own RemoteSource; channels are namespaced by site so keys +// stay unique, and manifest/page calls dispatch to the owning member. Browser- +// safe for the same reason reader.ts is: every read is `fetch`. + +import type { ChannelTranscriptsManifest, ChannelSubsManifest } from "../manifest"; +import type { TranscriptDetail } from "../transcripts"; +import type { SubsDetail } from "../subs"; +import type { ChannelPostsManifest, Post } from "../posts"; +import { coerceAliasConfig, type SearchAlias } from "../searchAliases"; +import type { ChannelDigestsManifest, VideoDigest } from "../digests"; +import type { VideoStat } from "../stats"; +import { corpusUrl, rootFileUrl, type HubSite } from "./contract"; +import { + EMPTY_GROUPS, + RemoteSource, + pageCacheBudgetBytes, + type ArchiveReader, + type ChannelGroups, + type ChannelRef, + type DuplicateIndex, + type HubCorpusJson, + type IndexedVideo, + type VideoAvailability, + type VideoIndex, +} from "./reader"; + +export class HubSource implements ArchiveReader { + readonly label: string; + readonly hubBase: string; + private members = new Map<string, RemoteSource>(); // siteId -> source + private aliases?: SearchAlias[]; + private index?: Promise<Map<string, IndexedVideo>>; + private duplicates?: Promise<DuplicateIndex>; + private stats?: Promise<ReadonlyMap<string, VideoStat>>; + private sites?: Promise<HubSite[]>; + + // A hub is N HTTP origins, so the latency argument for a wide window applies + // even harder than for a single remote. + readonly pageConcurrency = 8; + // Optional subset allowlist of member siteIds. Undefined = federate every + // member; a set restricts listChannels() to those members (site discovery via + // listSites() stays unfiltered so a picker can still see all members). + private allowSiteIds?: Set<string>; + + constructor(hubUrl: string, allowSiteIds?: string[]) { + this.hubBase = hubUrl.replace(/\/+$/, ""); + this.allowSiteIds = + allowSiteIds && allowSiteIds.length > 0 + ? new Set(allowSiteIds) + : undefined; + this.label = this.allowSiteIds + ? `hub:${this.hubBase} (${this.allowSiteIds.size} site(s))` + : `hub:${this.hubBase}`; + } + + // Fetch the hub's corpus.json and return its member sites — UNFILTERED (the + // full membership), even when this source is scoped to a subset, so a picker + // (list_sources / use_source) can show every member. + // + // Memoised on the promise: readChannels() and videoIndex() both need it, so + // an un-memoised version fetched the hub roster twice on a cold hub search. + // A failure is not memoised. + listSites(): Promise<HubSite[]> { + this.sites ??= this.readSites().catch((e: unknown) => { + this.sites = undefined; + throw e; + }); + return this.sites; + } + + private async readSites(): Promise<HubSite[]> { + const url = corpusUrl(this.hubBase); + const res = await fetch(url); + if (!res.ok) { + throw new Error(`GET ${url} -> ${res.status} ${res.statusText}`); + } + const hub = (await res.json()) as HubCorpusJson; + return hub.sites ?? []; + } + + // One page-cache budget for the whole hub, divided across its members — N + // members must not each get the full ceiling. Floored so a large federation + // still caches something per member. + private memberBudget(memberCount: number): number { + const total = pageCacheBudgetBytes(); + const floor = 8 * 1024 * 1024; + return Math.max(floor, Math.floor(total / Math.max(1, memberCount))); + } + + // A hub can ship its own /search-aliases.json (the merged federation-wide + // dictionary); if it doesn't, aliases are simply off for hub-wide search. + async loadAliases(): Promise<SearchAlias[]> { + if (this.aliases) return this.aliases; + try { + const res = await fetch(rootFileUrl("search-aliases.json", this.hubBase)); + this.aliases = res.ok + ? coerceAliasConfig(await res.json()).aliases + : []; + } catch { + this.aliases = []; + } + return this.aliases; + } + + // In hub mode each group is itself a federated member site (a different model + // — accent-per-origin), so hub-wide channel-group tokens are deferred: return + // the empty fallback. Multi-channel scoping still works (member groupIds carry + // through listChannels, they just don't resolve against hub-level groups). + async loadGroups(): Promise<ChannelGroups> { + return EMPTY_GROUPS; + } + + private memberFor(siteId: string): RemoteSource { + const m = this.members.get(siteId); + if (!m) throw new Error(`unknown hub member site: ${siteId}`); + return m; + } + + private channelList?: Promise<ChannelRef[]>; + + listChannels(opts: { refresh?: boolean } = {}): Promise<ChannelRef[]> { + if (opts.refresh) { + this.channelList = undefined; + this.sites = undefined; + this.index = undefined; + this.duplicates = undefined; + this.stats = undefined; + this.aliases = undefined; + // Members hold their own manifest/page caches; drop them wholesale so a + // refresh means the same thing federation-wide as it does locally. + this.members.clear(); + } + this.channelList ??= this.readChannels().catch((e: unknown) => { + this.channelList = undefined; // don't memoise a failure + throw e; + }); + return this.channelList; + } + + // Populating `members` must stay INSIDE the memoised call: memberFor() + // depends on it, so a memo that skipped this would leave every + // transcriptPage/postsPage dispatch throwing "unknown hub member site". + private async readChannels(): Promise<ChannelRef[]> { + const sites = (await this.listSites()).filter( + (s) => !this.allowSiteIds || this.allowSiteIds.has(s.siteId), + ); + const budget = this.memberBudget(sites.length); + const all: ChannelRef[] = []; + // Sequential member fetches keep it simple and polite; the channel count is + // small. A failing member is skipped rather than failing the whole list. + for (const site of sites) { + const remote = + this.members.get(site.siteId) ?? new RemoteSource(site.url, budget); + this.members.set(site.siteId, remote); + try { + const channels = await remote.listChannels(); + for (const c of channels) { + all.push({ + ...c, + key: `${site.siteId}/${c.slug}`, + siteId: site.siteId, + siteTitle: site.title, + siteUrl: site.url, + }); + } + } catch { + // skip an unreachable member + } + } + return all; + } + + transcriptsManifest(ch: ChannelRef): Promise<ChannelTranscriptsManifest> { + if (!ch.siteId) throw new Error("hub channel ref missing siteId"); + return this.memberFor(ch.siteId).transcriptsManifest(ch); + } + + transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]> { + if (!ch.siteId) throw new Error("hub channel ref missing siteId"); + return this.memberFor(ch.siteId).transcriptPage(ch, page); + } + + // A hub has no single viewer origin — each video's origin is its member + // site's url (carried on the ChannelRef as `siteUrl`), which momentUrl prefers + // per-video. Return null so we never mint a wrong-origin viewer link. + publicOrigin(): string | null { + return null; + } + + async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { + if (!ch.siteId) return null; + try { + return await this.memberFor(ch.siteId).subsManifest(ch); + } catch { + return null; // member not yet registered / unreachable + } + } + + subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { + if (!ch.siteId) throw new Error("hub channel ref missing siteId"); + return this.memberFor(ch.siteId).subsPage(ch, page); + } + + async postsManifest(ch: ChannelRef): Promise<ChannelPostsManifest | null> { + if (!ch.siteId) return null; + try { + return await this.memberFor(ch.siteId).postsManifest(ch); + } catch { + return null; // member not yet registered / unreachable + } + } + + postsPage(ch: ChannelRef, page: number): Promise<Post[]> { + if (!ch.siteId) throw new Error("hub channel ref missing siteId"); + return this.memberFor(ch.siteId).postsPage(ch, page); + } + + async digestsManifest(ch: ChannelRef): Promise<ChannelDigestsManifest | null> { + if (!ch.siteId) return null; + try { + return await this.memberFor(ch.siteId).digestsManifest(ch); + } catch { + return null; // member not yet registered / unreachable + } + } + + digestPage(ch: ChannelRef, page: number): Promise<VideoDigest[]> { + if (!ch.siteId) throw new Error("hub channel ref missing siteId"); + return this.memberFor(ch.siteId).digestPage(ch, page); + } + + // Merge each member's video index. Keys are member-local slugs + // (`<channelSlug>/<id>`) — the same slug a member's transcript page records + // carry — so a per-record lookup joins correctly. Built lazily/cached. + videoIndex(): Promise<VideoIndex> { + this.index ??= this.buildMergedIndex(); + return this.index; + } + + private async buildMergedIndex(): Promise<Map<string, IndexedVideo>> { + const merged = new Map<string, IndexedVideo>(); + let sites: HubSite[]; + try { + sites = (await this.listSites()).filter( + (s) => !this.allowSiteIds || this.allowSiteIds.has(s.siteId), + ); + } catch { + return merged; + } + const budget = this.memberBudget(sites.length); + for (const site of sites) { + const remote = + this.members.get(site.siteId) ?? new RemoteSource(site.url, budget); + this.members.set(site.siteId, remote); + try { + for (const [slug, rec] of await remote.videoIndex()) { + merged.set(slug, rec); + } + } catch { + // skip an unreachable member + } + } + return merged; + } + + availabilityMap(): Promise<ReadonlyMap<string, VideoAvailability>> { + return this.videoIndex(); + } + + // Duplicate clusters are detected WITHIN a site, so federating them is a + // merge of per-member maps and nothing more — this deliberately does not try + // to detect mirrors ACROSS member sites. Two sites holding the same recording + // is a real thing, but nothing has compared their transcripts, and inventing + // a cross-site cluster here would be asserting a duplicate no detector ever + // confirmed. + duplicateIndex(): Promise<DuplicateIndex> { + this.duplicates ??= this.mergeMembers((m) => m.duplicateIndex()); + return this.duplicates; + } + + statsIndex(): Promise<ReadonlyMap<string, VideoStat>> { + this.stats ??= this.mergeMembers((m) => m.statsIndex()); + return this.stats; + } + + // Merge one lazily-read layer across every member site, keyed by the + // member-local slug — the same shape and the same tolerance as + // buildMergedIndex (an unreachable member is skipped, not fatal). + private async mergeMembers<T>( + read: (m: RemoteSource) => Promise<ReadonlyMap<string, T>>, + ): Promise<Map<string, T>> { + const merged = new Map<string, T>(); + let sites: HubSite[]; + try { + sites = (await this.listSites()).filter( + (s) => !this.allowSiteIds || this.allowSiteIds.has(s.siteId), + ); + } catch { + return merged; + } + const budget = this.memberBudget(sites.length); + for (const site of sites) { + const remote = + this.members.get(site.siteId) ?? new RemoteSource(site.url, budget); + this.members.set(site.siteId, remote); + try { + for (const [k, v] of await read(remote)) merged.set(k, v); + } catch { + // skip an unreachable member + } + } + return merged; + } +} diff --git a/common/lib/archive/reader.test.ts b/common/lib/archive/reader.test.ts @@ -0,0 +1,454 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import type { ChannelTranscriptsManifest } from "../manifest"; +import type { TranscriptDetail } from "../transcripts"; +import type { ChannelDigestsManifest, VideoDigest } from "../digests"; +import { manifestUrl, pageUrl } from "./contract"; +import { + PageCache, + PromiseMap, + RemoteSource, + type ArchiveReader, + type ChannelRef, +} from "./reader"; + +// ─── an in-memory archive ─── +// +// A Map from URL to JSON body, handed to RemoteSource as `fetch`. That makes +// the thing under test the REAL reader (its caches, its walk, its tolerance) +// rather than a stub that agrees with it — and it makes every assertion below +// about fetches countable, which is the only way to state "one read, not N". + +const ORIGIN = "https://fixture.example"; + +type Archive = { + fetches: string[]; + body: Map<string, unknown>; + // Resolvers for URLs deliberately held open, so two concurrent callers can be + // observed sharing one in-flight read. + gate?: { url: string; release: () => void }; +}; + +function makeArchive(entries: Record<string, unknown>): Archive { + const a: Archive = { fetches: [], body: new Map(Object.entries(entries)) }; + return a; +} + +function installFetch(a: Archive, hold?: string): () => void { + const real = globalThis.fetch; + let release: (() => void) | undefined; + globalThis.fetch = (async (input: RequestInfo | URL) => { + const url = String(input); + a.fetches.push(url); + if (!a.body.has(url)) { + return new Response("not found", { status: 404, statusText: "Not Found" }); + } + const text = JSON.stringify(a.body.get(url)); + if (hold !== undefined && url === hold) { + await new Promise<void>((r) => { + release = r; + }); + } + return new Response(text, { status: 200 }); + }) as typeof fetch; + a.gate = { url: hold ?? "", release: () => release?.() }; + return () => { + globalThis.fetch = real; + }; +} + +function pageRecords(slug: string, ids: string[]): TranscriptDetail[] { + return ids.map( + (id) => + ({ + id, + slug: `${slug}/${id}`, + title: id, + uploadDate: "20240101", + duration: 60, + channel: slug, + channelSlug: slug, + cues: [{ start: 0, end: 1, text: "hello" }], + }) as unknown as TranscriptDetail, + ); +} + +const CORPUS = { + spec: 3, + kind: "site", + site: { id: "fixture", title: "Fixture", url: ORIGIN }, + channels: [{ slug: "alpha", name: "Alpha", videoCount: 3 }], +}; + +const ALPHA_MANIFEST: ChannelTranscriptsManifest = { + version: 1, + channelSlug: "alpha", + pageCount: 2, + maxPageBytes: 1000, + generatedAt: "2026-09-12T00:00:00.000Z", + slugToPage: { "alpha/a1": 0, "alpha/a2": 0, "alpha/a3": 1 }, +}; + +function fullArchive(): Archive { + return makeArchive({ + [`${ORIGIN}/corpus.json`]: CORPUS, + [manifestUrl("transcripts", "alpha", ORIGIN)]: ALPHA_MANIFEST, + [pageUrl("transcripts", "alpha", 0, ORIGIN)]: pageRecords("alpha", ["a1", "a2"]), + [pageUrl("transcripts", "alpha", 1, ORIGIN)]: pageRecords("alpha", ["a3"]), + }); +} + +const ALPHA: ChannelRef = { key: "alpha", slug: "alpha", name: "Alpha" }; + +// ─── the walk ─── + +test("the documented walk: corpus.json -> manifest -> slugToPage -> page", async () => { + const a = fullArchive(); + const restore = installFetch(a); + try { + const reader: ArchiveReader = new RemoteSource(ORIGIN); + const channels = await reader.listChannels(); + assert.deepEqual( + channels.map((c) => c.slug), + ["alpha"], + ); + // The channel ref carries its owning origin, which is what lets a hub + // attribute a hit to the member site it came from. + assert.equal(channels[0].siteUrl, ORIGIN); + + const manifest = await reader.transcriptsManifest(channels[0]); + const page = manifest.slugToPage["alpha/a3"]; + assert.equal(page, 1); + + const records = await reader.transcriptPage(channels[0], page); + assert.deepEqual( + records.map((r) => r.id), + ["a3"], + ); + + assert.deepEqual(a.fetches, [ + `${ORIGIN}/corpus.json`, + `${ORIGIN}/transcripts/alpha/manifest.json`, + `${ORIGIN}/transcripts/alpha/page-0001.json`, + ]); + } finally { + restore(); + } +}); + +test("the reader reproduces the contract's URL shapes exactly", async () => { + const a = fullArchive(); + const restore = installFetch(a); + try { + const reader = new RemoteSource(ORIGIN); + await reader.transcriptsManifest(ALPHA); + await reader.transcriptPage(ALPHA, 0); + assert.deepEqual(a.fetches, [ + manifestUrl("transcripts", "alpha", ORIGIN), + pageUrl("transcripts", "alpha", 0, ORIGIN), + ]); + // …and the shapes really are the published ones, not just self-consistent. + assert.deepEqual(a.fetches, [ + `${ORIGIN}/transcripts/alpha/manifest.json`, + `${ORIGIN}/transcripts/alpha/page-0000.json`, + ]); + } finally { + restore(); + } +}); + +test("a trailing slash on the origin does not double up", async () => { + const a = fullArchive(); + const restore = installFetch(a); + try { + const reader = new RemoteSource(`${ORIGIN}/`); + await reader.transcriptsManifest(ALPHA); + assert.deepEqual(a.fetches, [`${ORIGIN}/transcripts/alpha/manifest.json`]); + } finally { + restore(); + } +}); + +// ─── caching ─── + +test("the manifest memo is the promise: two concurrent callers, one fetch", async () => { + const a = fullArchive(); + const held = manifestUrl("transcripts", "alpha", ORIGIN); + const restore = installFetch(a, held); + try { + const reader = new RemoteSource(ORIGIN); + const first = reader.transcriptsManifest(ALPHA); + const second = reader.transcriptsManifest(ALPHA); + // Both callers are waiting on the SAME in-flight read — not two reads that + // happen to land on the same answer. + assert.equal(a.fetches.length, 1); + a.gate?.release(); + const [m1, m2] = await Promise.all([first, second]); + assert.equal(m1, m2); + assert.equal(a.fetches.length, 1); + + // A third, after it resolved, is still free. + await reader.transcriptsManifest(ALPHA); + assert.equal(a.fetches.length, 1); + } finally { + restore(); + } +}); + +test("listChannels({refresh}) is the only way to re-read, and it drops the page caches too", async () => { + const a = fullArchive(); + const restore = installFetch(a); + try { + const reader = new RemoteSource(ORIGIN); + await reader.listChannels(); + await reader.transcriptsManifest(ALPHA); + await reader.transcriptPage(ALPHA, 0); + const before = a.fetches.length; + await reader.listChannels(); + await reader.transcriptsManifest(ALPHA); + await reader.transcriptPage(ALPHA, 0); + assert.equal(a.fetches.length, before, "everything served from cache"); + + await reader.listChannels({ refresh: true }); + await reader.transcriptsManifest(ALPHA); + await reader.transcriptPage(ALPHA, 0); + assert.equal(a.fetches.length, before * 2); + } finally { + restore(); + } +}); + +test("a failed read is not memoised", async () => { + const a = makeArchive({}); + const restore = installFetch(a); + try { + const reader = new RemoteSource(ORIGIN); + await assert.rejects(() => reader.listChannels(), /404/); + // The corpus appears on the second try; a memoised failure would hide it. + a.body.set(`${ORIGIN}/corpus.json`, CORPUS); + const channels = await reader.listChannels(); + assert.equal(channels.length, 1); + } finally { + restore(); + } +}); + +// ─── the sparse layers: a 404 is data, and the negative is cached ─── + +test("a channel with no digests caches the absence — one probe, not one per query", async () => { + const a = fullArchive(); + const restore = installFetch(a); + try { + const reader = new RemoteSource(ORIGIN); + assert.equal(await reader.digestsManifest?.(ALPHA), null); + assert.equal(await reader.digestsManifest?.(ALPHA), null); + assert.equal(await reader.digestsManifest?.(ALPHA), null); + assert.deepEqual(a.fetches, [manifestUrl("digests", "alpha", ORIGIN)]); + } finally { + restore(); + } +}); + +test("a channel WITH digests reads them through the same walk", async () => { + const digests: ChannelDigestsManifest = { + version: 1, + channelSlug: "alpha", + pageCount: 1, + maxPageBytes: 1000, + generatedAt: "2026-09-12T00:00:00.000Z", + slugToPage: { "alpha/a1": 0 }, + } as unknown as ChannelDigestsManifest; + const a = fullArchive(); + a.body.set(manifestUrl("digests", "alpha", ORIGIN), digests); + a.body.set(pageUrl("digests", "alpha", 0, ORIGIN), [ + { slug: "alpha/a1", chapters: [] } as unknown as VideoDigest, + ]); + const restore = installFetch(a); + try { + const reader = new RemoteSource(ORIGIN); + const m = await reader.digestsManifest?.(ALPHA); + assert.equal(m?.slugToPage["alpha/a1"], 0); + const page = await reader.digestPage?.(ALPHA, 0); + assert.equal(page?.length, 1); + } finally { + restore(); + } +}); + +test("subs and posts are absent-tolerant in the same way", async () => { + const a = fullArchive(); + const restore = installFetch(a); + try { + const reader = new RemoteSource(ORIGIN); + assert.equal(await reader.subsManifest(ALPHA), null); + assert.equal(await reader.postsManifest(ALPHA), null); + assert.equal(await reader.subsManifest(ALPHA), null); + assert.equal(await reader.postsManifest(ALPHA), null); + assert.deepEqual(a.fetches, [ + manifestUrl("subs", "alpha", ORIGIN), + manifestUrl("posts", "alpha", ORIGIN), + ]); + } finally { + restore(); + } +}); + +test("an absent summaries index is an empty map, not an error", async () => { + const a = fullArchive(); + const restore = installFetch(a); + try { + const reader = new RemoteSource(ORIGIN); + const index = await reader.videoIndex?.(); + assert.equal(index?.size, 0); + // And the availability map IS that index — one read, not two. + const avail = await reader.availabilityMap(); + assert.equal(avail.size, 0); + } finally { + restore(); + } +}); + +test("an absent duplicates.json is no clusters, not a failure", async () => { + const a = fullArchive(); + const restore = installFetch(a); + try { + const reader = new RemoteSource(ORIGIN); + const dupes = await reader.duplicateIndex?.(); + assert.equal(dupes?.size, 0); + assert.deepEqual(a.fetches, [`${ORIGIN}/duplicates.json`]); + } finally { + restore(); + } +}); + +test("an absent search-aliases.json leaves search plain", async () => { + const a = fullArchive(); + const restore = installFetch(a); + try { + const reader = new RemoteSource(ORIGIN); + assert.deepEqual(await reader.loadAliases(), []); + assert.deepEqual(await reader.loadAliases(), []); + assert.deepEqual(a.fetches, [`${ORIGIN}/search-aliases.json`]); + } finally { + restore(); + } +}); + +// ─── PromiseMap ─── + +test("PromiseMap: one load per key, and a rejection evicts itself", async () => { + const map = new PromiseMap<number>(); + let loads = 0; + const load = () => { + loads++; + return Promise.resolve(7); + }; + const [a, b] = await Promise.all([map.take("k", load), map.take("k", load)]); + assert.equal(a, 7); + assert.equal(b, 7); + assert.equal(loads, 1); + + let fails = 0; + const boom = () => { + fails++; + return Promise.reject(new Error("nope")); + }; + await assert.rejects(() => map.take("bad", boom)); + await assert.rejects(() => map.take("bad", boom)); + assert.equal(fails, 2, "a transient failure is never cached as a permanent one"); + + map.clear(); + await map.take("k", load); + assert.equal(loads, 2); +}); + +// ─── PageCache ─── + +test("PageCache evicts by RAW BYTES, least-recently-used first", async () => { + // Budget for exactly three 10-byte pages. + const cache = new PageCache<string>(30); + const loaded: string[] = []; + const put = (k: string) => + cache.take(k, async () => { + loaded.push(k); + return { value: k, bytes: 10 }; + }); + + await put("p1"); + await put("p2"); + await put("p3"); + assert.deepEqual(loaded, ["p1", "p2", "p3"], "three pages fit the budget exactly"); + + // A hit moves its entry to the most-recently-used end. Touch p1 so the + // least-recently-used is p2, not the insertion order's p1 — that is the whole + // difference between an LRU and a queue. + await put("p1"); + assert.deepEqual(loaded, ["p1", "p2", "p3"], "all three still resident"); + + // A fourth page takes the total to 40 > 30, so one goes: p2. + await put("p4"); + assert.deepEqual(loaded, ["p1", "p2", "p3", "p4"]); + await put("p3"); + await put("p1"); + assert.equal(loaded.length, 4, "p3 and p1 survived"); + await put("p2"); + assert.deepEqual(loaded, ["p1", "p2", "p3", "p4", "p2"], "p2 was the victim"); +}); + +test("PageCache keeps MIN_CACHED_PAGES even when one page blows the budget", async () => { + // One page is four times the whole budget: thrashing on it would mean never + // caching anything, which is worse than going over. + const cache = new PageCache<string>(10); + const loaded: string[] = []; + const put = (k: string) => + cache.take(k, async () => { + loaded.push(k); + return { value: k, bytes: 40 }; + }); + await put("big1"); + await put("big2"); + await put("big1"); + await put("big2"); + assert.deepEqual(loaded, ["big1", "big2"]); +}); + +test("PageCache with a zero budget caches nothing but still serves", async () => { + const cache = new PageCache<string>(0); + const loaded: string[] = []; + const put = (k: string) => + cache.take(k, async () => { + loaded.push(k); + return { value: k, bytes: 10 }; + }); + assert.equal(await put("p"), "p"); + assert.equal(await put("p"), "p"); + assert.deepEqual(loaded, ["p", "p"]); +}); + +test("PageCache coalesces concurrent callers and drops a rejection", async () => { + const cache = new PageCache<string>(1000); + let loads = 0; + let release!: () => void; + const gate = new Promise<void>((r) => { + release = r; + }); + const load = async () => { + loads++; + await gate; + return { value: "v", bytes: 10 }; + }; + const a = cache.take("k", load); + const b = cache.take("k", load); + assert.equal(loads, 1); + release(); + assert.deepEqual(await Promise.all([a, b]), ["v", "v"]); + + let fails = 0; + const boom = async () => { + fails++; + throw new Error("nope"); + }; + await assert.rejects(() => cache.take("bad", boom)); + await assert.rejects(() => cache.take("bad", boom)); + assert.equal(fails, 2); +}); diff --git a/common/lib/archive/reader.ts b/common/lib/archive/reader.ts @@ -0,0 +1,779 @@ +// THE ARCHIVE READER — one walk of the published shard scheme. +// +// Moved verbatim from mcp/src/source.ts, which was the best of the five +// hand-rolled readers in the repo (three transports, promise-coalescing caches, +// a byte-budgeted LRU, tolerant of every layer a site legitimately does not +// ship). `mcp/src/source.ts` is now a re-export of this file, so nothing in the +// MCP server changed shape; what changed is that the viewer, the offline cache +// and umtool can reach the same implementation instead of re-deriving it. +// +// ZERO node imports. This file and reader-hub.ts run in a browser; the only +// node-touching transport is reader-fs.ts (LocalSource), which imports FROM +// here and is never value-imported BY here — a `next build` of the export app +// is the test of that, because tsc cannot tell a `node:fs` import from any +// other. +// +// `process` appears once, behind a `typeof process` guard, to read the page +// cache budget knob. Everything else is transport-agnostic. + +import { + type ChannelTranscriptsManifest, + type ChannelSubsManifest, + type Manifest, +} from "../manifest"; +import type { TranscriptDetail, DisplaySummary } from "../transcripts"; +import { summaryState, type VideoState } from "../availability"; +import type { SubsDetail } from "../subs"; +import type { ChannelPostsManifest, Post } from "../posts"; +import { coerceAliasConfig, type SearchAlias } from "../searchAliases"; +import { + resolveCanonicalSlug, + DUPLICATES_FILENAME, + type DuplicateReport, +} from "../duplicates"; +import type { ChannelDigestsManifest, VideoDigest } from "../digests"; +import type { StatsManifest, VideoStat } from "../stats"; +import { + parseChannelGroups, + resolveDefaultGroupId, + DEFAULT_GROUP_FALLBACK_ID, + type ChannelGroup, +} from "../channelGroups"; +import { manifestUrl, pageUrl, rootFileUrl, type HubSite } from "./contract"; +import { recordRead } from "./io-stats"; + +// A channel the source can serve. `siteId`/`siteUrl` are only populated in hub +// mode (so results can be attributed to the owning member site); `key` is the +// stable, source-unique handle a tool passes back to fetch this channel's data. +// `groupId` is the channel's raw group membership as shipped in corpus.json (it +// may be unknown/absent — resolve it against loadGroups() with +// resolveChannelGroupId before using it). +export type ChannelRef = { + key: string; + slug: string; + name: string; + videoCount?: number; + groupId?: string; + siteId?: string; + siteTitle?: string; + siteUrl?: string; +}; + +// The site's channel-group definitions, as read from summaries/manifest.json — +// the same groups the viewer's channel filter renders. `defaultGroupId` is the +// bucket unknown/absent channel groupIds fold onto (resolveChannelGroupId). +export type ChannelGroups = { + groups: ChannelGroup[]; + defaultGroupId: string; +}; + +// What a source returns when it has no group definitions (absent/malformed +// manifest, or hub mode where per-site groups are a different model). +export const EMPTY_GROUPS: ChannelGroups = { + groups: [], + defaultGroupId: DEFAULT_GROUP_FALLBACK_ID, +}; + +// Per-video availability, joined in from the summaries shards (a +// TranscriptDetail record does not carry presence state). Keyed by the +// member-local video slug (`<channelSlug>/<id>`) — the same slug a transcript +// page record carries — so the search engine can apply the `fav` filter. +export type VideoAvailability = { state: VideoState }; + +// One video as the summaries shards describe it. A strict superset of +// VideoAvailability, so the same map serves both the availability join and the +// filter-first page planner — there is one index, not two parallel reads of the +// same files. +// +// Everything here comes from `summaries/`, which is a GLOBAL index of every +// video in the corpus: 1.4 MB and ~0.5 s to parse, against 1.3 GB and ~48 s for +// the transcripts. That ratio is the whole basis of filter-first scanning — the +// exact page set a filtered query needs is computable from this plus each +// channel manifest's `slugToPage`, before a single transcript byte is read. +export type IndexedVideo = { + state: VideoState; + id: string; + channelSlug: string; + title: string; + uploadDate: string; + isLivestream: boolean; + ageRestricted: boolean; +}; + +export type VideoIndex = ReadonlyMap<string, IndexedVideo>; + +// One video's membership in a cross-platform duplicate cluster, as shipped in +// duplicates.json. Keyed by the member-local slug, like the video index. +// +// `aligned` is carried per SIBLING, not per cluster, and is the gate on ever +// translating a timestamp from one copy to another. Absent means NOT MEASURED, +// which must be read as not aligned — a mirror with a longer intro matches on +// text at shifted times, so a plausible-looking citation would land in the +// wrong place in the wrong upload. That is the failure mode that looks like +// success, and the only defence is refusing to guess. +export type ClusterMembership = { + clusterId: string; + // The member that owns derived work for this cluster (resolveCanonicalSlug), + // or null when a human marked the cluster not-a-duplicate. + canonicalSlug: string | null; + isCanonical: boolean; + // Every OTHER member of the cluster. + siblings: { + slug: string; + id: string; + channelSlug: string; + channel: string; + platform: string; + title: string; + duration: number; + uploadDate: string; + hasTranscript: boolean; + aligned: boolean; + offsetSeconds: number | null; + }[]; + // The cluster is a clip-of-a-longer-video relationship: the members overlap + // only partially, so nothing may be mapped across wholesale. + contained: boolean; + // Title+duration only — nothing compared the actual content. An unconfirmed + // suspect, not an established duplicate. + needsReview: boolean; +}; + +export type DuplicateIndex = ReadonlyMap<string, ClusterMembership>; + +// Fold a duplicates.json report into a slug → membership map. Tolerant of +// absence throughout: `compose-site.ts` only writes the file when there is at +// least one publishable cluster, and corpus.json doesn't even declare it, so a +// site legitimately ships none. +export function buildDuplicateIndex( + report: DuplicateReport | null, +): Map<string, ClusterMembership> { + const map = new Map<string, ClusterMembership>(); + if (!report || !Array.isArray(report.clusters)) return map; + for (const cluster of report.clusters) { + const refs = cluster.videoRefs ?? []; + if (refs.length < 2) continue; + const canonicalSlug = resolveCanonicalSlug(cluster); + // A human marked it not-a-duplicate — it is not a cluster any more. + if (canonicalSlug === null) continue; + for (const ref of refs) { + map.set(ref.slug, { + clusterId: cluster.clusterId, + canonicalSlug, + isCanonical: ref.slug === canonicalSlug, + contained: cluster.contained === true, + needsReview: cluster.needsReview === true, + siblings: refs + .filter((o) => o.slug !== ref.slug) + .map((o) => ({ + slug: o.slug, + id: o.id, + channelSlug: o.channelSlug, + channel: o.channel, + platform: o.platform, + title: o.title, + duration: o.duration, + uploadDate: o.uploadDate, + hasTranscript: o.hasTranscript === true, + // Alignment is a property of the PAIR, and the report records it + // against each member relative to the cluster's canonical. Both + // sides must be measured-and-aligned before a timestamp may cross. + aligned: ref.aligned === true && o.aligned === true, + offsetSeconds: o.offsetSeconds ?? null, + })), + }); + } + } + return map; +} + +// Fold the shipped stats shards into a slug → stat map. Same tolerant shape as +// the summaries read: absent or malformed is an empty map, never an error. +// +// Note the shape difference that forces a full read rather than a targeted one: +// StatsManifest carries `channels` and `pageCount` but NO `slugToPage`, so +// there is no way to jump to the page holding one video. Pages are large (up to +// STATS_MAX_PAGE_BYTES = 20 MB), which is exactly why this is lazy — nothing +// reads it until a tool asks for a stat. +export async function buildStatsIndex( + readManifest: () => Promise<StatsManifest | null>, + readPage: (page: number) => Promise<VideoStat[] | null>, +): Promise<Map<string, VideoStat>> { + const map = new Map<string, VideoStat>(); + let manifest: StatsManifest | null; + try { + manifest = await readManifest(); + } catch { + return map; + } + if (!manifest || typeof manifest.pageCount !== "number") return map; + for (let page = 0; page < manifest.pageCount; page++) { + let records: VideoStat[] | null; + try { + records = await readPage(page); + } catch { + continue; + } + if (!records) continue; + for (const r of records) { + if (typeof r.slug === "string") map.set(r.slug, r); + } + } + return map; +} + +// Read a site's global summaries shards (summaries/manifest.json + +// summaries/page-NNNN.json) via `readPage` and fold them into a slug → video +// map. Tolerant: an absent/malformed manifest yields an empty map, and a page +// that fails to read is skipped. `readManifest`/`readPage` throw or return null +// on absence per the source's transport. +// +// Tolerance is load-bearing for the page planner, not just politeness: a +// summaries set that is missing, partial, or older than the transcripts must +// degrade to "I don't know about this video", and the planner's rule for +// don't-know is to scan the page anyway. +export async function buildVideoIndex( + readManifest: () => Promise<Manifest | null>, + readPage: (page: number) => Promise<DisplaySummary[] | null>, +): Promise<Map<string, IndexedVideo>> { + const map = new Map<string, IndexedVideo>(); + let manifest: Manifest | null; + try { + manifest = await readManifest(); + } catch { + return map; + } + if (!manifest || typeof manifest.pageCount !== "number") return map; + for (let page = 0; page < manifest.pageCount; page++) { + let records: DisplaySummary[] | null; + try { + records = await readPage(page); + } catch { + continue; + } + if (!records) continue; + for (const r of records) { + if (typeof r.slug !== "string") continue; + // summaryState falls back to the legacy isDeleted/isUnlisted booleans, + // which matters here more than anywhere: a hub reads summaries pages + // from member origins it does not control, so some of them will have + // been built before `state` existed. + map.set(r.slug, { + state: summaryState(r), + id: r.id, + channelSlug: r.channelSlug, + title: r.title ?? "", + uploadDate: r.uploadDate ?? "", + isLivestream: r.isLivestream === true, + ageRestricted: r.ageRestricted === true, + }); + } + } + return map; +} + +// Parse a summaries/manifest.json blob into channel-group defs, tolerating any +// missing/malformed shape (→ empty fallback). +export function parseGroupsManifest(raw: unknown): ChannelGroups { + const m = (raw ?? {}) as { groups?: unknown; defaultGroupId?: unknown }; + const groups = parseChannelGroups(m.groups); + return { groups, defaultGroupId: resolveDefaultGroupId(m.defaultGroupId, groups) }; +} + +// A read-only view over a transcript corpus's paginated JSON shards. Three +// implementations (local dir / remote origin / federated hub) all speak the +// same three-call contract, which mirrors the documented shard scheme in +// corpus.json: list channels, get a channel's manifest (slug -> page map), get +// a page of full transcript records. +// +// NOTE the method that is deliberately absent: there is no `record(layer, slug, +// id)`. The contract's walk is manifest -> shard -> record and a caller that +// wants one record already holds the shard, so a per-record fetch would be a +// structural regression (N reads where the plan needs one) dressed up as +// convenience. +export interface ArchiveReader { + readonly label: string; + // The channel list, memoised per source instance. searchTranscripts, + // findVideo and findPost all call it, so a 20-id get_transcripts batch used + // to cost 20 corpus.json fetches over HTTP. The memo is the promise, so + // concurrent callers share one fetch. Pass `refresh` to drop it and re-read — + // an explicit staleness escape hatch, deliberately not a TTL. + listChannels(opts?: { refresh?: boolean }): Promise<ChannelRef[]>; + transcriptsManifest(ch: ChannelRef): Promise<ChannelTranscriptsManifest>; + transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]>; + // The site's shipped curated search aliases (the same /search-aliases.json the + // viewer reads). Returns [] when the file is absent or malformed. Used to make + // caption search alias-aware, so a query for a term with a curated regex + // (e.g. "k cups" → "(k|cake)[ -]?cup") also matches the mis-transcribed + // spellings. Result is cached per source. + loadAliases(): Promise<SearchAlias[]>; + // The site's channel-group definitions (summaries/manifest.json). Returns the + // empty fallback when absent/malformed, or in hub mode (federated per-site + // groups are a different model — deferred). Cached per source. + loadGroups(): Promise<ChannelGroups>; + // The public origin of the viewer that owns this source's videos, or null + // when there isn't one (a local dir on disk). Used to build archilyzer viewer + // deep links for cited moments (momentUrl). A single-site remote returns its + // base URL; a hub returns null because each video's origin is its member + // site's url (carried on the ChannelRef as `siteUrl`) — prefer that per-video. + publicOrigin(): string | null; + // A channel's live-chat/subs manifest (subs/<slug>/manifest.json), or null + // when the channel ships no subs shards. Same slugToPage/pageCount shape as + // the transcripts manifest. Fetched lazily — only the chat search scope needs + // it. + subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null>; + // A page of a channel's subs records (subs/<slug>/page-NNNN.json). Each record + // inlines its per-track cues under `tracks` (e.g. `tracks.live_chat`). + subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]>; + // A channel's social-posts manifest (posts/<slug>/manifest.json), or null + // when the channel ships no posts shards (i.e. it is a video channel). Same + // slugToPage/pageCount shape as the transcripts manifest. This is the MCP's + // single abstraction boundary, so adding it here yields the posts corpus on + // all three transports (local / remote / hub) at once. + postsManifest(ch: ChannelRef): Promise<ChannelPostsManifest | null>; + // A page of a channel's posts (posts/<slug>/page-NNNN.json). + postsPage(ch: ChannelRef, page: number): Promise<Post[]>; + // A map of every video's availability (deleted/unlisted), keyed by the + // member-local video slug (`<channelSlug>/<id>`), built from the summaries + // shards. Fetched lazily and cached — only the `fav` availability filter needs + // it. Empty when the source ships no summaries. + // + // ReadonlyMap so the richer videoIndex() can BE this map rather than a + // projection of it: ReadonlyMap is covariant in its value type, so one + // Map<string, IndexedVideo> satisfies both and the two can never drift. + availabilityMap(): Promise<ReadonlyMap<string, VideoAvailability>>; + // The full summaries-backed index, when this source ships one. OPTIONAL: the + // in-memory test stubs don't implement it, and a site that ships no + // summaries/ genuinely has no index — callers must degrade to a full scan + // rather than assume an empty index means an empty corpus. + videoIndex?(): Promise<VideoIndex>; + // How many shard pages this source is willing to have in flight at once. + // Optional; callers use `?? DEFAULT_PAGE_CONCURRENCY`. Local is CPU-bound on + // JSON.parse (measured: 42 ms read vs 389 ms parse for an 8 MB page), so + // concurrency there only overlaps read with parse and saturates quickly. + // Remote is latency-bound, where it is the dominant win. + readonly pageConcurrency?: number; + // The shipped cross-platform duplicate report, folded to slug → membership. + // OPTIONAL and empty-when-absent: compose-site only writes duplicates.json + // when there is at least one publishable cluster, corpus.json does not + // declare it, and the in-memory test stubs have no such concept. A site that + // ships none must behave exactly as it does today. + duplicateIndex?(): Promise<DuplicateIndex>; + // The shipped per-video stats index (view/like counts, cueCount, transcript + // coverage), slug-keyed. OPTIONAL for the same reasons. Lazy: stats/ is one + // ~3.4 MB page here and up to 20 MB elsewhere, so it is only read when a tool + // actually asks for it. + statsIndex?(): Promise<ReadonlyMap<string, VideoStat>>; + // A channel's AI-digest manifest (digests/<slug>/manifest.json), or null when + // the channel has none. Mirrors the posts pair, including the cached negative + // — the digest corpus is SPARSE BY DESIGN (a channel with zero digests gets + // no manifest at all), so probing it per query must not cost a read per + // channel per call. OPTIONAL on the interface for the usual reason. + digestsManifest?(ch: ChannelRef): Promise<ChannelDigestsManifest | null>; + digestPage?(ch: ChannelRef, page: number): Promise<VideoDigest[]>; +} + +// Used when a source states no preference. Deliberately modest: each in-flight +// page costs its raw bytes plus ~2.7× that once parsed, and this box is shared. +export const DEFAULT_PAGE_CONCURRENCY = 4; + +// Shape of the channels we read out of a site corpus.json (Layer 1). Kept loose +// — we only need slug/name/count/group. +export type CorpusJsonChannel = { + slug: string; + name?: string; + videoCount?: number; + groupId?: string; +}; +export type SiteCorpusJson = { + channels?: CorpusJsonChannel[]; + // The composed site's own declared public origin — the deployed archilyzer + // viewer these shards were built for. Present in every spec-3 corpus.json. + site?: { id?: string; title?: string; url?: string }; +}; + +export type HubCorpusJson = { + kind?: string; + sites?: HubSite[]; +}; + +// One server-only env read, guarded so this module loads in a browser. The +// dynamic `process.env?.[…]` form is deliberately NOT the inlinable +// `process.env.NAME` member expression — a page cache budget is a server knob, +// and a browser reader takes the default. +function envVar(name: string): string | undefined { + return typeof process !== "undefined" ? process.env?.[name] : undefined; +} + +// ─── Bounded promise caches ─── +// +// Two different caching problems, so two different structures: +// +// manifests — ~551 KB for the whole corpus (29 channels). Small, hot, and +// re-read constantly: an un-hinted 20-id get_transcripts batch +// used to cost ~300 manifest reads because findVideo walks every +// channel per id. Cached OUTRIGHT, no bound. +// +// pages — ~7.4 MB of raw JSON each, several times that once parsed. An +// unbounded map of these is gigabytes, so this is a small LRU. +// Its job is the 20-id batch that lands on ONE shared page (20 +// reads → 1); it is deliberately NOT sized to hold a scan, which +// visits each page exactly once and would only be paying memory +// for evictions. +// +// The bound is a RAW-BYTE budget, not an entry count, because page sizes differ +// by an order of magnitude across corpora (a 13-record VOD page is 8 MB; a +// shorts channel's page is a fraction of that). Default 48 MB, configurable +// with TRANSCRIPT_MCP_PAGE_CACHE_MB (0 disables). +// +// Why 48: a parsed page retains about 2.7× its file bytes (measured — an +// 8.09 MB page holds 21.8 MB of JS heap), so 48 MB of raw budget is roughly +// 130 MB resident. That is the most I am willing to hold on a box that also +// runs a GPU digest sweep and other agents' jobs. It is ~6 pages of this +// corpus, which covers the working set this cache exists for (a 20-id batch +// from an enumerate worklist arrives in page order and lands on 1–3 pages). +// Note what it deliberately does NOT cover: a full-corpus scan is 170 pages ≈ +// 3.7 GB retained, so there is no cache size between "6 pages" and "impossible" +// that changes the full-scan story. Filter-first scanning changes that instead. +const DEFAULT_PAGE_CACHE_MB = 48; +// Always keep at least this many entries, so a corpus whose single page exceeds +// the whole budget still caches that page rather than thrashing on it. +const MIN_CACHED_PAGES = 2; + +export function pageCacheBudgetBytes(): number { + const raw = envVar("TRANSCRIPT_MCP_PAGE_CACHE_MB"); + const mb = + raw === undefined || raw.trim() === "" ? DEFAULT_PAGE_CACHE_MB : Number(raw); + const safe = Number.isFinite(mb) && mb >= 0 ? mb : DEFAULT_PAGE_CACHE_MB; + return Math.floor(safe * 1024 * 1024); +} + +// What a cached loader reports back: the parsed value plus the raw byte size it +// was parsed from, which is what the budget is denominated in. +export type Sized<T> = { value: T; bytes: number }; + +// An LRU keyed by string, holding PROMISES rather than values so that N +// concurrent callers for the same page coalesce onto one read — the pattern +// makeChatFetcher already uses. A rejected promise evicts itself, so a +// transient failure is never cached as a permanent one. +// +// Sizes are only known once a read resolves, so an in-flight entry counts as 0 +// and the budget is enforced on resolve. An entry evicted while still in flight +// resolves normally for whoever already holds its promise; it just isn't +// remembered. +export class PageCache<T> { + private map = new Map<string, { p: Promise<T>; bytes: number }>(); + private total = 0; + constructor(private readonly maxBytes: number) {} + + take(key: string, load: () => Promise<Sized<T>>): Promise<T> { + const hit = this.map.get(key); + if (hit !== undefined) { + this.map.delete(key); + this.map.set(key, hit); // most-recently used goes last + return hit.p; + } + if (this.maxBytes <= 0) return load().then((s) => s.value); + + const entry: { p: Promise<T>; bytes: number } = { p: null as never, bytes: 0 }; + entry.p = load() + .then((s) => { + // Only account for it if we're still the live entry for this key — + // a refresh() between issue and resolve must not resurrect it. + if (this.map.get(key) === entry) { + entry.bytes = s.bytes; + this.total += s.bytes; + this.evict(); + } + return s.value; + }) + .catch((e: unknown) => { + this.drop(key, entry); + throw e; + }); + this.map.set(key, entry); + return entry.p; + } + + private drop(key: string, entry: { bytes: number }): void { + if (this.map.get(key) === entry) { + this.map.delete(key); + this.total -= entry.bytes; + } + } + + private evict(): void { + while (this.total > this.maxBytes && this.map.size > MIN_CACHED_PAGES) { + const oldest = this.map.entries().next().value; + if (oldest === undefined) break; + this.map.delete(oldest[0]); + this.total -= oldest[1].bytes; + } + } + + clear(): void { + this.map.clear(); + this.total = 0; + } +} + +// The unbounded sibling, for the small-and-hot caches (manifests). Same +// don't-memoise-a-failure rule. +export class PromiseMap<T> { + private map = new Map<string, Promise<T>>(); + + take(key: string, load: () => Promise<T>): Promise<T> { + const hit = this.map.get(key); + if (hit !== undefined) return hit; + const p = load().catch((e: unknown) => { + this.map.delete(key); + throw e; + }); + this.map.set(key, p); + return p; + } + + clear(): void { + this.map.clear(); + } +} + +// ─── Remote: fetch shards from a deployed site origin over HTTP ─── +export class RemoteSource implements ArchiveReader { + readonly label: string; + private base: string; + private aliases?: SearchAlias[]; + private groups?: ChannelGroups; + private index?: Promise<Map<string, IndexedVideo>>; + private duplicates?: Promise<DuplicateIndex>; + private stats?: Promise<ReadonlyMap<string, VideoStat>>; + private subsManifests = new PromiseMap<ChannelSubsManifest | null>(); + private digestManifests = new PromiseMap<ChannelDigestsManifest | null>(); + private subsPages: PageCache<SubsDetail[]>; + private postsManifests = new PromiseMap<ChannelPostsManifest | null>(); + private transcriptManifests = new PromiseMap<ChannelTranscriptsManifest>(); + private transcriptPages: PageCache<TranscriptDetail[]>; + + // Over HTTP the cost is latency, not parse, so a wider window is the dominant + // win — this is where bounded concurrency actually pays. + readonly pageConcurrency = 8; + + // `budgetBytes` lets a hub divide one memory ceiling across its members + // instead of granting each member the full budget (N members × 48 MB is not a + // budget, it's N budgets). + constructor(baseUrl: string, budgetBytes = pageCacheBudgetBytes()) { + this.base = baseUrl.replace(/\/+$/, ""); + this.label = `remote:${this.base}`; + this.subsPages = new PageCache<SubsDetail[]>(budgetBytes); + this.transcriptPages = new PageCache<TranscriptDetail[]>(budgetBytes); + } + + // The deployed site origin — the archilyzer viewer that owns these videos. + publicOrigin(): string | null { + return this.base; + } + + private resetCaches(): void { + this.subsManifests.clear(); + this.digestManifests.clear(); + this.subsPages.clear(); + this.postsManifests.clear(); + this.transcriptManifests.clear(); + this.transcriptPages.clear(); + this.aliases = undefined; + this.groups = undefined; + this.index = undefined; + this.duplicates = undefined; + this.stats = undefined; + } + + subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { + return this.subsManifests.take(ch.slug, async () => { + try { + const res = await fetch(manifestUrl("subs", ch.slug, this.base)); + return res.ok ? ((await res.json()) as ChannelSubsManifest) : null; + } catch { + return null; + } + }); + } + + subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { + return this.subsPages.take(`${ch.slug}:${page}`, () => + this.getJsonSized<SubsDetail[]>(pageUrl("subs", ch.slug, page), "subsPage"), + ); + } + + // Cached per channel including the negative answer — otherwise every + // posts-covering search costs one 404 per video-only channel. + postsManifest(ch: ChannelRef): Promise<ChannelPostsManifest | null> { + return this.postsManifests.take(ch.slug, async () => { + try { + const res = await fetch(manifestUrl("posts", ch.slug, this.base)); + return res.ok ? ((await res.json()) as ChannelPostsManifest) : null; + } catch { + return null; + } + }); + } + + postsPage(ch: ChannelRef, page: number): Promise<Post[]> { + return this.getJson(pageUrl("posts", ch.slug, page), "postsPage"); + } + + videoIndex(): Promise<VideoIndex> { + this.index ??= buildVideoIndex( + async () => { + const res = await fetch(manifestUrl("summaries", undefined, this.base)); + return res.ok ? ((await res.json()) as Manifest) : null; + }, + async (page) => { + const res = await fetch(pageUrl("summaries", undefined, page, this.base)); + return res.ok ? ((await res.json()) as DisplaySummary[]) : null; + }, + ); + return this.index; + } + + availabilityMap(): Promise<ReadonlyMap<string, VideoAvailability>> { + return this.videoIndex(); + } + + digestsManifest(ch: ChannelRef): Promise<ChannelDigestsManifest | null> { + return this.digestManifests.take(ch.slug, async () => { + try { + const res = await fetch(manifestUrl("digests", ch.slug, this.base)); + return res.ok ? ((await res.json()) as ChannelDigestsManifest) : null; + } catch { + return null; + } + }); + } + + digestPage(ch: ChannelRef, page: number): Promise<VideoDigest[]> { + return this.getJson(pageUrl("digests", ch.slug, page), "digestPage"); + } + + duplicateIndex(): Promise<DuplicateIndex> { + this.duplicates ??= (async () => { + try { + const res = await fetch(rootFileUrl(DUPLICATES_FILENAME, this.base)); + return buildDuplicateIndex( + res.ok ? ((await res.json()) as DuplicateReport) : null, + ); + } catch { + return buildDuplicateIndex(null); + } + })(); + return this.duplicates; + } + + statsIndex(): Promise<ReadonlyMap<string, VideoStat>> { + this.stats ??= buildStatsIndex( + async () => { + const res = await fetch(manifestUrl("stats", undefined, this.base)); + return res.ok ? ((await res.json()) as StatsManifest) : null; + }, + async (page) => { + const res = await fetch(pageUrl("stats", undefined, page, this.base)); + return res.ok ? ((await res.json()) as VideoStat[]) : null; + }, + ); + return this.stats; + } + + async loadAliases(): Promise<SearchAlias[]> { + if (this.aliases) return this.aliases; + try { + const res = await fetch(rootFileUrl("search-aliases.json", this.base)); + this.aliases = res.ok + ? coerceAliasConfig(await res.json()).aliases + : []; + } catch { + this.aliases = []; + } + return this.aliases; + } + + async loadGroups(): Promise<ChannelGroups> { + if (this.groups) return this.groups; + try { + const res = await fetch(manifestUrl("summaries", undefined, this.base)); + this.groups = res.ok + ? parseGroupsManifest(await res.json()) + : EMPTY_GROUPS; + } catch { + this.groups = EMPTY_GROUPS; + } + return this.groups; + } + + // Fetches as TEXT so the byte size is knowable — the page cache's budget is + // denominated in raw bytes, and res.json() throws the length away. + // + // `p` is a ROOT-RELATIVE path from the contract builders; the base is joined + // here so the thrown error names the full URL, exactly as before. + private async getJsonSized<T>( + p: string, + kind: string, + ): Promise<{ value: T; bytes: number }> { + const res = await fetch(`${this.base}${p}`); + if (!res.ok) { + throw new Error(`GET ${this.base}${p} -> ${res.status} ${res.statusText}`); + } + const raw = await res.text(); + recordRead(kind, raw.length); + return { value: JSON.parse(raw) as T, bytes: raw.length }; + } + + private async getJson<T>(p: string, kind = "json"): Promise<T> { + return (await this.getJsonSized<T>(p, kind)).value; + } + + private channelList?: Promise<ChannelRef[]>; + + listChannels(opts: { refresh?: boolean } = {}): Promise<ChannelRef[]> { + if (opts.refresh) { + this.channelList = undefined; + this.resetCaches(); + } + this.channelList ??= this.readChannels().catch((e: unknown) => { + this.channelList = undefined; // don't memoise a failure + throw e; + }); + return this.channelList; + } + + private async readChannels(): Promise<ChannelRef[]> { + const corpus = await this.getJson<SiteCorpusJson>( + rootFileUrl("corpus.json"), + "corpus", + ); + return (corpus.channels ?? []).map((c) => ({ + key: c.slug, + slug: c.slug, + name: c.name ?? c.slug, + videoCount: c.videoCount, + groupId: c.groupId, + siteUrl: this.base, + })); + } + + transcriptsManifest(ch: ChannelRef): Promise<ChannelTranscriptsManifest> { + return this.transcriptManifests.take(ch.slug, () => + this.getJson<ChannelTranscriptsManifest>( + manifestUrl("transcripts", ch.slug), + "transcriptsManifest", + ), + ); + } + + transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]> { + return this.transcriptPages.take(`${ch.slug}:${page}`, () => + this.getJsonSized<TranscriptDetail[]>( + pageUrl("transcripts", ch.slug, page), + "transcriptPage", + ), + ); + } +} diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts @@ -1,5 +1,15 @@ import type { PublicSiteDescriptor } from "./siteDescriptor"; import { PROJECT_GENERATOR } from "./project"; +import { + CONTRACT, + archiveUrl, + corpusUrl, + manifestUrl, + rootFileUrl, + type ContractLayer, + type HubCorpusSite, + type HubMemberInput, +} from "./archive/contract"; // The machine-readable corpus index emitted at `/corpus.json` on every export // bundle (and an aggregate variant on a hub). It does NOT contain transcripts — @@ -22,31 +32,20 @@ import { PROJECT_GENERATOR } from "./project"; // miss data it could otherwise have retrieved. `generator` is an informational // string that points at the software, not at any content; no reader's behaviour // changes by not knowing about it, and the only in-repo consumer -// (mcp/src/source.ts) casts the parsed corpus loosely. Bumping the spec would +// (the archive reader) casts the parsed corpus loosely. Bumping the spec would // force every client to re-evaluate compatibility for a credit line. -export const CONTRACT = { - // /corpus.json's own `spec`. See the note above for why `generator` did not - // bump it. - corpusSpec: 3, - // /site.json's `contract` (siteDescriptor.ts). - siteDescriptor: 1, - // The four manifest versions (manifest.ts). Each is the version field of one - // served document; they move independently and always have. - manifest: 3, - transcriptsManifest: 1, - subsManifest: 4, - subsChannelManifest: 1, - // Records per /summaries/page-NNNN.json. - summariesPageSize: 1000, - // The zero-padding on every page shard's file name. `page-0.json` is a 404 — - // this is the whole reason the constant exists in one place. - pagePad: 4, - // The served trees that follow the manifest -> slugToPage -> page-NNNN walk. - // "summaries" is flat (one manifest, pages, no per-channel level). - layers: ["transcripts", "subs", "posts", "digests", "summaries"], -} as const; - -export type ContractLayer = (typeof CONTRACT.layers)[number]; +// +// CONTRACT ITSELF NOW LIVES IN `lib/archive/contract.ts`, beside the URL +// builders this file calls, and is re-exported here so every `from "./corpus"` +// import site is unchanged. It had to move rather than be imported: `manifest.ts` +// reads `CONTRACT.pagePad` at module scope, so corpus -> archive/contract -> +// manifest -> corpus would be a TDZ ReferenceError, not a lint warning. The +// contract module imports nothing from either, which makes the graph a DAG. +export { CONTRACT }; +export type { ContractLayer }; +// The hub member types moved with it (they are contract shapes, not corpus +// builders) and are re-exported for the same reason. +export type { HubCorpusSite, HubMemberInput }; // Every version constant below is a field of CONTRACT, re-exported under the // name its callers already use. Nothing on the wire changes by moving one. @@ -183,15 +182,6 @@ export type SiteCorpus = { useWithAi: string; }; -export type HubCorpusSite = { - siteId: string; - title: string; - url: string; - corpus: string; - siteJson: string; - pwa: boolean; -}; - export type HubCorpus = { spec: number; kind: "hub"; @@ -204,21 +194,10 @@ export type HubCorpus = { useWithAi: string; }; -// Minimal member shape a hub knows about (mirror of compose-hub.ts HubSiteEntry). -export type HubMemberInput = { - siteId: string; - siteTitle: string; - siteUrl: string; - pwa?: boolean; -}; - -// Join an origin base with a root-relative path. When no base is known (a site -// built without a configured siteUrl) the path is left root-relative — still -// correct for a same-origin fetch, just not portable cross-origin. -function join(base: string | undefined, p: string): string { - if (!base) return p; - return `${base.replace(/\/+$/, "")}${p}`; -} +// The join every URL below goes through — "absolute when siteUrl is set, +// root-relative otherwise" — now defined once in archive/contract.ts, where the +// readers that must reproduce these URLs can reach it. +const join = archiveUrl; // Build the per-site corpus index from the public site descriptor (`/site.json`) // plus whether this build emitted bulk archives. @@ -247,15 +226,14 @@ export function buildSiteCorpus( ...(c.groupId ? { groupId: c.groupId } : {}), ...(postCount ? { postCount } : {}), ...(digestCount ? { digestCount } : {}), + // One definition of the manifest URL shape, shared with every reader. + // The conditional spreads stay here: whether a layer is ADVERTISED is a + // property of this site's build, not of the contract. manifests: { - transcripts: join(base, `/transcripts/${c.slug}/manifest.json`), - subs: join(base, `/subs/${c.slug}/manifest.json`), - ...(postCount - ? { posts: join(base, `/posts/${c.slug}/manifest.json`) } - : {}), - ...(digestCount - ? { digests: join(base, `/digests/${c.slug}/manifest.json`) } - : {}), + transcripts: manifestUrl("transcripts", c.slug, base), + subs: manifestUrl("subs", c.slug, base), + ...(postCount ? { posts: manifestUrl("posts", c.slug, base) } : {}), + ...(digestCount ? { digests: manifestUrl("digests", c.slug, base) } : {}), }, }; }); @@ -311,8 +289,8 @@ export function buildHubCorpus( siteId: m.siteId, title: m.siteTitle, url, - corpus: `${url}/corpus.json`, - siteJson: `${url}/site.json`, + corpus: corpusUrl(url), + siteJson: rootFileUrl("site.json", url), pwa: m.pwa === true, }; }); @@ -361,7 +339,7 @@ export function renderSiteLlmsTxt(corpus: SiteCorpus): string { out.push(""); out.push("## Corpus"); out.push( - `- [corpus.json](${join(base, "/corpus.json")}): machine-readable index — ` + + `- [corpus.json](${corpusUrl(base)}): machine-readable index — ` + `channels and how to fetch any transcript from the paginated JSON shards.`, ); if (corpus.digestScheme) { @@ -417,7 +395,7 @@ export function renderHubLlmsTxt(corpus: HubCorpus): string { out.push(""); out.push("## Federation"); out.push( - `- [corpus.json](${join(base, "/corpus.json")}): machine-readable directory ` + + `- [corpus.json](${corpusUrl(base)}): machine-readable directory ` + `of every member site and its corpus endpoint.`, ); out.push(""); diff --git a/common/lib/manifest.ts b/common/lib/manifest.ts @@ -1,5 +1,6 @@ import type { ChannelGroup } from "./channelGroups"; import { CONTRACT } from "./corpus"; +import { pageFileName } from "./archive/contract"; export type ChannelEntry = { name: string; @@ -41,9 +42,12 @@ export const SUMMARIES_PAGE_SIZE = CONTRACT.summariesPageSize; // in this file (summaries, transcripts, subs) plus one per layer module; // `page-0.json` is a 404 on every published archive, so a copy that lost the // padding would 404 silently against a real site and pass every unit test. -export function pageFileName(index: number): string { - return `page-${String(index).padStart(CONTRACT.pagePad, "0")}.json`; -} +// +// Defined in `archive/contract.ts` — beside CONTRACT.pagePad and the URL +// builders that use it — and re-exported here so the twelve call sites that +// import it from `./manifest` (and the three per-layer aliases in stats.ts / +// posts.ts / digests.ts) are unchanged. +export { pageFileName }; export const TRANSCRIPTS_MANIFEST_VERSION = CONTRACT.transcriptsManifest; diff --git a/common/lib/search/collapse.ts b/common/lib/search/collapse.ts @@ -0,0 +1,29 @@ +// ONE SEARCH PIPELINE — placeholder. Filled by one-core phase 2 slice S3 +// (`plans/one-core-phase-2.md` §S3); created here by S1 so no two parallel +// slices race to create the same file. +// +// Intended contents: cross-platform mirror collapsing, from +// `mcp/src/search.ts` (collapseDuplicates) and `components/SearchResults.tsx`. +// +// The two rules that make it safe on by default, and that must survive the +// move verbatim: +// +// 1. The kept row is the cluster's canonical member WHEN that member is +// itself among the matches — otherwise simply the first match. A mirror is +// frequently the only surviving copy of a deleted upload, and preferring +// an absent canonical would delete exactly the evidence a "what did the +// removed videos say" question is asking for. +// 2. The collapsed copies are NAMED on the row they folded into. Nothing +// vanishes; the count stops double-counting. +// +// And the one it must never break: timestamps are NEVER mapped between copies +// here. That requires the per-pair `aligned` gate, and this function does not +// move a single second of anything. +// +// export function collapseDuplicates<T extends CollapsibleHit>( +// hits: T[], +// dupes: DuplicateIndex, +// enabled: boolean, +// ): { collapsed: number; kept: T[] }; + +export {}; diff --git a/common/lib/search/evalTree.ts b/common/lib/search/evalTree.ts @@ -0,0 +1,31 @@ +// ONE SEARCH PIPELINE — placeholder. Filled by one-core phase 2 slice S3 +// (`plans/one-core-phase-2.md` §S3); created here by S1 so no two parallel +// slices race to create the same file. +// +// Intended contents: the query-tree evaluator that exists twice today — +// `mcp/src/search.ts:1189-1338` (evalLeaf / evalNode / passesFilters) and +// `common/lib/searchEval.ts`'s copy — as one implementation over one record +// shape. `passesFilters` stays SINGULAR: a second "cheap" predicate for +// planning is exactly how a pruner starts silently disagreeing with the scanner +// about what matches. +// +// export type LeafMatcher = { scope: LayerScope; test: (text: string) => boolean }; +// export type RecordCtx = { +// cues: Cue[]; +// title: string; +// description?: string; +// tags?: string[]; +// includeSnippets: boolean; +// snippetsPerVideo: number; +// }; +// export type LeafOutcome = { matched: boolean; count: number; hits: ScopedSnippet[] }; +// +// export function evalLeaf(leaf: QueryNode, m: LeafMatcher, ctx: RecordCtx): LeafOutcome; +// export function evalNode(node: QueryNode, ms: LeafMatcher[], ctx: RecordCtx): LeafOutcome; +// export function passesFilters( +// rec: { isLivestream?: boolean; ageRestricted?: boolean; uploadDate: string }, +// f: SearchFilters, +// avail: VideoAvailability | undefined, +// ): boolean; + +export {}; diff --git a/common/lib/search/policy.ts b/common/lib/search/policy.ts @@ -0,0 +1,30 @@ +// ONE SEARCH PIPELINE — placeholder. Filled by one-core phase 2 slice S3 +// (`plans/one-core-phase-2.md` §S3); created here by S1 so no two parallel +// slices race to create the same file. +// +// Intended contents: the caps that are today private constants of +// `mcp/src/search.ts`, named and passed rather than re-derived, so the bench's +// structural counts are unchanged by construction. +// +// export type SearchPolicy = { +// // A hard ceiling on shard pages fetched per query, so a rare term over a +// // large (or hub-wide) corpus cannot run away. Reaching it sets +// // `truncated`. search.ts:259 MAX_PAGES 400 +// maxPages: number; +// // A ceiling on matched videos collected before counting stops, so +// // `total` stays bounded for a very common term. Reaching it also sets +// // `truncated`. search.ts:263 HARD_VIDEO_CAP 2000 +// hardVideoCap: number; +// // Cap on windowed excerpt lines per video, so a video with hundreds of +// // matches cannot blow a batch's token budget. +// // search.ts:268 WINDOW_LINE_CAP 200 +// windowLineCap: number; +// // Snippet truncation width. search.ts:456 truncate(…, 240) +// snippetChars: number; +// }; +// +// // What the MCP server passes. The viewer passes its own, UNCAPPED: a human +// // scrolling a page is not spending an agent's token budget. +// export const MCP_POLICY: SearchPolicy; + +export {}; diff --git a/common/lib/search/rank.ts b/common/lib/search/rank.ts @@ -0,0 +1,16 @@ +// ONE SEARCH PIPELINE — placeholder. Filled by one-core phase 2 slice S3 +// (`plans/one-core-phase-2.md` §S3); created here by S1 so no two parallel +// slices race to create the same file. +// +// Intended contents: result ordering, once. `mcp/src/search.ts:477` and the +// viewer's ordering in `components/SearchResults.tsx` are the same intent +// written twice. +// +// export type RankMode = "relevance" | "date" | "duration" | …; +// export function rankHits<T extends RankableHit>(hits: T[], mode: RankMode): T[]; +// +// Fetchers are INJECTED into this module's callers ({ reader, policy, +// onProgress }) so react-query stays in `components/`; reuse +// `lib/concurrency.ts`'s `mapConcurrent` rather than a second limiter. + +export {}; diff --git a/common/lib/search/window.ts b/common/lib/search/window.ts @@ -0,0 +1,18 @@ +// ONE SEARCH PIPELINE — placeholder. Filled by one-core phase 2 slice S3 +// (`plans/one-core-phase-2.md` §S3); created here by S1 so no two parallel +// slices race to create the same file. +// +// Intended contents: the windowing + snippet layer, over the existing +// `lib/transcriptWindow.ts` primitives (`windowCues`, `cuesToSnippets`, +// `mergeSnippets`), so the viewer's excerpt and the MCP's excerpt are the same +// excerpt. +// +// export function windowedTranscript( +// cues: Cue[], +// match: Matcher, +// opts: { before: number; after: number; lineCap: number }, +// ): WindowSnippet[]; +// +// export function truncate(text: string, max: number): string; + +export {}; diff --git a/common/package.json b/common/package.json @@ -41,7 +41,7 @@ "./styles/*": "./styles/*.ts" }, "scripts": { - "test": "tsx --test \"*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components}/*.test.ts\"" + "test": "tsx --test \"*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components}/*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components}/*/*.test.ts\"" }, "dependencies": { "@sindresorhus/slugify": "^3.0.0", diff --git a/mcp/src/source.ts b/mcp/src/source.ts @@ -1,1386 +1,47 @@ -import { readFile, readdir } from "node:fs/promises"; -import path from "node:path"; -import { - pageFileName, - type ChannelTranscriptsManifest, - type ChannelSubsManifest, - type Manifest, -} from "yt-dlp-transcript-common/lib/manifest"; -import type { - TranscriptDetail, - DisplaySummary, -} from "yt-dlp-transcript-common/lib/transcripts"; -import { - summaryState, - type VideoState, -} from "yt-dlp-transcript-common/lib/availability"; -import type { SubsDetail } from "yt-dlp-transcript-common/lib/subs"; -import { - postsPageFileName, - type ChannelPostsManifest, - type Post, -} from "yt-dlp-transcript-common/lib/posts"; -import { - coerceAliasConfig, - type SearchAlias, -} from "yt-dlp-transcript-common/lib/searchAliases"; -import { - resolveCanonicalSlug, - DUPLICATES_FILENAME, - type DuplicateReport, -} from "yt-dlp-transcript-common/lib/duplicates"; -import { - digestPageFileName, - type ChannelDigestsManifest, - type VideoDigest, -} from "yt-dlp-transcript-common/lib/digests"; -import { - statsPageFileName, - type StatsManifest, - type VideoStat, -} from "yt-dlp-transcript-common/lib/stats"; -import { - parseChannelGroups, - resolveDefaultGroupId, - DEFAULT_GROUP_FALLBACK_ID, - type ChannelGroup, -} from "yt-dlp-transcript-common/lib/channelGroups"; - -// A channel the source can serve. `siteId`/`siteUrl` are only populated in hub -// mode (so results can be attributed to the owning member site); `key` is the -// stable, source-unique handle a tool passes back to fetch this channel's data. -// `groupId` is the channel's raw group membership as shipped in corpus.json (it -// may be unknown/absent — resolve it against loadGroups() with -// resolveChannelGroupId before using it). -export type ChannelRef = { - key: string; - slug: string; - name: string; - videoCount?: number; - groupId?: string; - siteId?: string; - siteTitle?: string; - siteUrl?: string; -}; - -// The site's channel-group definitions, as read from summaries/manifest.json — -// the same groups the viewer's channel filter renders. `defaultGroupId` is the -// bucket unknown/absent channel groupIds fold onto (resolveChannelGroupId). -export type ChannelGroups = { - groups: ChannelGroup[]; - defaultGroupId: string; -}; - -// What a source returns when it has no group definitions (absent/malformed -// manifest, or hub mode where per-site groups are a different model). -const EMPTY_GROUPS: ChannelGroups = { - groups: [], - defaultGroupId: DEFAULT_GROUP_FALLBACK_ID, -}; - -// Per-video availability, joined in from the summaries shards (a -// TranscriptDetail record does not carry presence state). Keyed by the -// member-local video slug (`<channelSlug>/<id>`) — the same slug a transcript -// page record carries — so the search engine can apply the `fav` filter. -export type VideoAvailability = { state: VideoState }; - -// One video as the summaries shards describe it. A strict superset of -// VideoAvailability, so the same map serves both the availability join and the -// filter-first page planner — there is one index, not two parallel reads of the -// same files. -// -// Everything here comes from `summaries/`, which is a GLOBAL index of every -// video in the corpus: 1.4 MB and ~0.5 s to parse, against 1.3 GB and ~48 s for -// the transcripts. That ratio is the whole basis of filter-first scanning — the -// exact page set a filtered query needs is computable from this plus each -// channel manifest's `slugToPage`, before a single transcript byte is read. -export type IndexedVideo = { - state: VideoState; - id: string; - channelSlug: string; - title: string; - uploadDate: string; - isLivestream: boolean; - ageRestricted: boolean; -}; - -export type VideoIndex = ReadonlyMap<string, IndexedVideo>; - -// One video's membership in a cross-platform duplicate cluster, as shipped in -// duplicates.json. Keyed by the member-local slug, like the video index. -// -// `aligned` is carried per SIBLING, not per cluster, and is the gate on ever -// translating a timestamp from one copy to another. Absent means NOT MEASURED, -// which must be read as not aligned — a mirror with a longer intro matches on -// text at shifted times, so a plausible-looking citation would land in the -// wrong place in the wrong upload. That is the failure mode that looks like -// success, and the only defence is refusing to guess. -export type ClusterMembership = { - clusterId: string; - // The member that owns derived work for this cluster (resolveCanonicalSlug), - // or null when a human marked the cluster not-a-duplicate. - canonicalSlug: string | null; - isCanonical: boolean; - // Every OTHER member of the cluster. - siblings: { - slug: string; - id: string; - channelSlug: string; - channel: string; - platform: string; - title: string; - duration: number; - uploadDate: string; - hasTranscript: boolean; - aligned: boolean; - offsetSeconds: number | null; - }[]; - // The cluster is a clip-of-a-longer-video relationship: the members overlap - // only partially, so nothing may be mapped across wholesale. - contained: boolean; - // Title+duration only — nothing compared the actual content. An unconfirmed - // suspect, not an established duplicate. - needsReview: boolean; -}; - -export type DuplicateIndex = ReadonlyMap<string, ClusterMembership>; - -// Fold a duplicates.json report into a slug → membership map. Tolerant of -// absence throughout: `compose-site.ts` only writes the file when there is at -// least one publishable cluster, and corpus.json doesn't even declare it, so a -// site legitimately ships none. -export function buildDuplicateIndex( - report: DuplicateReport | null, -): Map<string, ClusterMembership> { - const map = new Map<string, ClusterMembership>(); - if (!report || !Array.isArray(report.clusters)) return map; - for (const cluster of report.clusters) { - const refs = cluster.videoRefs ?? []; - if (refs.length < 2) continue; - const canonicalSlug = resolveCanonicalSlug(cluster); - // A human marked it not-a-duplicate — it is not a cluster any more. - if (canonicalSlug === null) continue; - for (const ref of refs) { - map.set(ref.slug, { - clusterId: cluster.clusterId, - canonicalSlug, - isCanonical: ref.slug === canonicalSlug, - contained: cluster.contained === true, - needsReview: cluster.needsReview === true, - siblings: refs - .filter((o) => o.slug !== ref.slug) - .map((o) => ({ - slug: o.slug, - id: o.id, - channelSlug: o.channelSlug, - channel: o.channel, - platform: o.platform, - title: o.title, - duration: o.duration, - uploadDate: o.uploadDate, - hasTranscript: o.hasTranscript === true, - // Alignment is a property of the PAIR, and the report records it - // against each member relative to the cluster's canonical. Both - // sides must be measured-and-aligned before a timestamp may cross. - aligned: ref.aligned === true && o.aligned === true, - offsetSeconds: o.offsetSeconds ?? null, - })), - }); - } - } - return map; -} - -// Fold the shipped stats shards into a slug → stat map. Same tolerant shape as -// the summaries read: absent or malformed is an empty map, never an error. -// -// Note the shape difference that forces a full read rather than a targeted one: -// StatsManifest carries `channels` and `pageCount` but NO `slugToPage`, so -// there is no way to jump to the page holding one video. Pages are large (up to -// STATS_MAX_PAGE_BYTES = 20 MB), which is exactly why this is lazy — nothing -// reads it until a tool asks for a stat. -async function buildStatsIndex( - readManifest: () => Promise<StatsManifest | null>, - readPage: (page: number) => Promise<VideoStat[] | null>, -): Promise<Map<string, VideoStat>> { - const map = new Map<string, VideoStat>(); - let manifest: StatsManifest | null; - try { - manifest = await readManifest(); - } catch { - return map; - } - if (!manifest || typeof manifest.pageCount !== "number") return map; - for (let page = 0; page < manifest.pageCount; page++) { - let records: VideoStat[] | null; - try { - records = await readPage(page); - } catch { - continue; - } - if (!records) continue; - for (const r of records) { - if (typeof r.slug === "string") map.set(r.slug, r); - } - } - return map; -} - -// Read a site's global summaries shards (summaries/manifest.json + -// summaries/page-NNNN.json) via `readPage` and fold them into a slug → video -// map. Tolerant: an absent/malformed manifest yields an empty map, and a page -// that fails to read is skipped. `readManifest`/`readPage` throw or return null -// on absence per the source's transport. -// -// Tolerance is load-bearing for the page planner, not just politeness: a -// summaries set that is missing, partial, or older than the transcripts must -// degrade to "I don't know about this video", and the planner's rule for -// don't-know is to scan the page anyway. -async function buildVideoIndex( - readManifest: () => Promise<Manifest | null>, - readPage: (page: number) => Promise<DisplaySummary[] | null>, -): Promise<Map<string, IndexedVideo>> { - const map = new Map<string, IndexedVideo>(); - let manifest: Manifest | null; - try { - manifest = await readManifest(); - } catch { - return map; - } - if (!manifest || typeof manifest.pageCount !== "number") return map; - for (let page = 0; page < manifest.pageCount; page++) { - let records: DisplaySummary[] | null; - try { - records = await readPage(page); - } catch { - continue; - } - if (!records) continue; - for (const r of records) { - if (typeof r.slug !== "string") continue; - // summaryState falls back to the legacy isDeleted/isUnlisted booleans, - // which matters here more than anywhere: a hub reads summaries pages - // from member origins it does not control, so some of them will have - // been built before `state` existed. - map.set(r.slug, { - state: summaryState(r), - id: r.id, - channelSlug: r.channelSlug, - title: r.title ?? "", - uploadDate: r.uploadDate ?? "", - isLivestream: r.isLivestream === true, - ageRestricted: r.ageRestricted === true, - }); - } - } - return map; -} - -// Parse a summaries/manifest.json blob into channel-group defs, tolerating any -// missing/malformed shape (→ empty fallback). -function parseGroupsManifest(raw: unknown): ChannelGroups { - const m = (raw ?? {}) as { groups?: unknown; defaultGroupId?: unknown }; - const groups = parseChannelGroups(m.groups); - return { groups, defaultGroupId: resolveDefaultGroupId(m.defaultGroupId, groups) }; -} - -// A read-only view over a transcript corpus's paginated JSON shards. Three -// implementations (local dir / remote origin / federated hub) all speak the -// same three-call contract, which mirrors the documented shard scheme in -// corpus.json: list channels, get a channel's manifest (slug -> page map), get -// a page of full transcript records. -export interface ShardSource { - readonly label: string; - // The channel list, memoised per source instance. searchTranscripts, - // findVideo and findPost all call it, so a 20-id get_transcripts batch used - // to cost 20 corpus.json fetches over HTTP. The memo is the promise, so - // concurrent callers share one fetch. Pass `refresh` to drop it and re-read — - // an explicit staleness escape hatch, deliberately not a TTL. - listChannels(opts?: { refresh?: boolean }): Promise<ChannelRef[]>; - transcriptsManifest(ch: ChannelRef): Promise<ChannelTranscriptsManifest>; - transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]>; - // The site's shipped curated search aliases (the same /search-aliases.json the - // viewer reads). Returns [] when the file is absent or malformed. Used to make - // caption search alias-aware, so a query for a term with a curated regex - // (e.g. "k cups" → "(k|cake)[ -]?cup") also matches the mis-transcribed - // spellings. Result is cached per source. - loadAliases(): Promise<SearchAlias[]>; - // The site's channel-group definitions (summaries/manifest.json). Returns the - // empty fallback when absent/malformed, or in hub mode (federated per-site - // groups are a different model — deferred). Cached per source. - loadGroups(): Promise<ChannelGroups>; - // The public origin of the viewer that owns this source's videos, or null - // when there isn't one (a local dir on disk). Used to build archilyzer viewer - // deep links for cited moments (momentUrl). A single-site remote returns its - // base URL; a hub returns null because each video's origin is its member - // site's url (carried on the ChannelRef as `siteUrl`) — prefer that per-video. - publicOrigin(): string | null; - // A channel's live-chat/subs manifest (subs/<slug>/manifest.json), or null - // when the channel ships no subs shards. Same slugToPage/pageCount shape as - // the transcripts manifest. Fetched lazily — only the chat search scope needs - // it. - subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null>; - // A page of a channel's subs records (subs/<slug>/page-NNNN.json). Each record - // inlines its per-track cues under `tracks` (e.g. `tracks.live_chat`). - subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]>; - // A channel's social-posts manifest (posts/<slug>/manifest.json), or null - // when the channel ships no posts shards (i.e. it is a video channel). Same - // slugToPage/pageCount shape as the transcripts manifest. This is the MCP's - // single abstraction boundary, so adding it here yields the posts corpus on - // all three transports (local / remote / hub) at once. - postsManifest(ch: ChannelRef): Promise<ChannelPostsManifest | null>; - // A page of a channel's posts (posts/<slug>/page-NNNN.json). - postsPage(ch: ChannelRef, page: number): Promise<Post[]>; - // A map of every video's availability (deleted/unlisted), keyed by the - // member-local video slug (`<channelSlug>/<id>`), built from the summaries - // shards. Fetched lazily and cached — only the `fav` availability filter needs - // it. Empty when the source ships no summaries. - // - // ReadonlyMap so the richer videoIndex() can BE this map rather than a - // projection of it: ReadonlyMap is covariant in its value type, so one - // Map<string, IndexedVideo> satisfies both and the two can never drift. - availabilityMap(): Promise<ReadonlyMap<string, VideoAvailability>>; - // The full summaries-backed index, when this source ships one. OPTIONAL: the - // in-memory test stubs don't implement it, and a site that ships no - // summaries/ genuinely has no index — callers must degrade to a full scan - // rather than assume an empty index means an empty corpus. - videoIndex?(): Promise<VideoIndex>; - // How many shard pages this source is willing to have in flight at once. - // Optional; callers use `?? DEFAULT_PAGE_CONCURRENCY`. Local is CPU-bound on - // JSON.parse (measured: 42 ms read vs 389 ms parse for an 8 MB page), so - // concurrency there only overlaps read with parse and saturates quickly. - // Remote is latency-bound, where it is the dominant win. - readonly pageConcurrency?: number; - // The shipped cross-platform duplicate report, folded to slug → membership. - // OPTIONAL and empty-when-absent: compose-site only writes duplicates.json - // when there is at least one publishable cluster, corpus.json does not - // declare it, and the in-memory test stubs have no such concept. A site that - // ships none must behave exactly as it does today. - duplicateIndex?(): Promise<DuplicateIndex>; - // The shipped per-video stats index (view/like counts, cueCount, transcript - // coverage), slug-keyed. OPTIONAL for the same reasons. Lazy: stats/ is one - // ~3.4 MB page here and up to 20 MB elsewhere, so it is only read when a tool - // actually asks for it. - statsIndex?(): Promise<ReadonlyMap<string, VideoStat>>; - // A channel's AI-digest manifest (digests/<slug>/manifest.json), or null when - // the channel has none. Mirrors the posts pair, including the cached negative - // — the digest corpus is SPARSE BY DESIGN (a channel with zero digests gets - // no manifest at all), so probing it per query must not cost a read per - // channel per call. OPTIONAL on the interface for the usual reason. - digestsManifest?(ch: ChannelRef): Promise<ChannelDigestsManifest | null>; - digestPage?(ch: ChannelRef, page: number): Promise<VideoDigest[]>; -} - -// Used when a source states no preference. Deliberately modest: each in-flight -// page costs its raw bytes plus ~2.7× that once parsed, and this box is shared. -export const DEFAULT_PAGE_CONCURRENCY = 4; - -// ─── I/O instrumentation (opt-in, for mcp/bench) ─── -// -// A process-wide counter of shard reads and parsed bytes, so the benchmark can -// report the STRUCTURAL cost of a query (how many pages, how many bytes) next -// to its wall time. That matters on this box specifically: wall time is only -// meaningful when the machine is idle, but read counts and byte counts are -// properties of the query plan and hold under any load. -// -// Off unless MCP_IO_STATS=1, and even then it is two integer adds per read. -export type IoStats = { reads: number; bytes: number }; - -const IO_STATS_ON = process.env.MCP_IO_STATS === "1"; - -const ioTotals: Record<string, IoStats> = {}; - -export function recordRead(kind: string, bytes: number): void { - if (!IO_STATS_ON) return; - const slot = (ioTotals[kind] ??= { reads: 0, bytes: 0 }); - slot.reads++; - slot.bytes += bytes; -} - -// A snapshot of every counter so far, for diffing across one tool call. -export function ioStatsSnapshot(): Record<string, IoStats> { - const out: Record<string, IoStats> = {}; - for (const [k, v] of Object.entries(ioTotals)) out[k] = { ...v }; - return out; -} - -export function ioStatsEnabled(): boolean { - return IO_STATS_ON; -} - -// Read a local JSON file, counting its bytes when instrumentation is on, and -// reporting the raw size so a byte-budgeted cache can account for it. -async function readLocalJsonSized<T>( - file: string, - kind: string, -): Promise<{ value: T; bytes: number }> { - const raw = await readFile(file, "utf8"); - recordRead(kind, raw.length); - return { value: JSON.parse(raw) as T, bytes: raw.length }; -} - -async function readLocalJson<T>(file: string, kind: string): Promise<T> { - return (await readLocalJsonSized<T>(file, kind)).value; -} - -// Shape of the channels we read out of a site corpus.json (Layer 1). Kept loose -// — we only need slug/name/count/group. -type CorpusJsonChannel = { - slug: string; - name?: string; - videoCount?: number; - groupId?: string; -}; -type SiteCorpusJson = { - channels?: CorpusJsonChannel[]; - // The composed site's own declared public origin — the deployed archilyzer - // viewer these shards were built for. Present in every spec-3 corpus.json. - site?: { id?: string; title?: string; url?: string }; -}; - -// ─── Bounded promise caches ─── -// -// Two different caching problems, so two different structures: -// -// manifests — ~551 KB for the whole corpus (29 channels). Small, hot, and -// re-read constantly: an un-hinted 20-id get_transcripts batch -// used to cost ~300 manifest reads because findVideo walks every -// channel per id. Cached OUTRIGHT, no bound. -// -// pages — ~7.4 MB of raw JSON each, several times that once parsed. An -// unbounded map of these is gigabytes, so this is a small LRU. -// Its job is the 20-id batch that lands on ONE shared page (20 -// reads → 1); it is deliberately NOT sized to hold a scan, which -// visits each page exactly once and would only be paying memory -// for evictions. -// -// The bound is a RAW-BYTE budget, not an entry count, because page sizes differ -// by an order of magnitude across corpora (a 13-record VOD page is 8 MB; a -// shorts channel's page is a fraction of that). Default 48 MB, configurable -// with TRANSCRIPT_MCP_PAGE_CACHE_MB (0 disables). -// -// Why 48: a parsed page retains about 2.7× its file bytes (measured — an -// 8.09 MB page holds 21.8 MB of JS heap), so 48 MB of raw budget is roughly -// 130 MB resident. That is the most I am willing to hold on a box that also -// runs a GPU digest sweep and other agents' jobs. It is ~6 pages of this -// corpus, which covers the working set this cache exists for (a 20-id batch -// from an enumerate worklist arrives in page order and lands on 1–3 pages). -// Note what it deliberately does NOT cover: a full-corpus scan is 170 pages ≈ -// 3.7 GB retained, so there is no cache size between "6 pages" and "impossible" -// that changes the full-scan story. Filter-first scanning changes that instead. -const DEFAULT_PAGE_CACHE_MB = 48; -// Always keep at least this many entries, so a corpus whose single page exceeds -// the whole budget still caches that page rather than thrashing on it. -const MIN_CACHED_PAGES = 2; - -function pageCacheBudgetBytes(): number { - const raw = process.env.TRANSCRIPT_MCP_PAGE_CACHE_MB; - const mb = - raw === undefined || raw.trim() === "" ? DEFAULT_PAGE_CACHE_MB : Number(raw); - const safe = Number.isFinite(mb) && mb >= 0 ? mb : DEFAULT_PAGE_CACHE_MB; - return Math.floor(safe * 1024 * 1024); -} - -// What a cached loader reports back: the parsed value plus the raw byte size it -// was parsed from, which is what the budget is denominated in. -type Sized<T> = { value: T; bytes: number }; - -// An LRU keyed by string, holding PROMISES rather than values so that N -// concurrent callers for the same page coalesce onto one read — the pattern -// makeChatFetcher already uses. A rejected promise evicts itself, so a -// transient failure is never cached as a permanent one. -// -// Sizes are only known once a read resolves, so an in-flight entry counts as 0 -// and the budget is enforced on resolve. An entry evicted while still in flight -// resolves normally for whoever already holds its promise; it just isn't -// remembered. -class PageCache<T> { - private map = new Map<string, { p: Promise<T>; bytes: number }>(); - private total = 0; - constructor(private readonly maxBytes: number) {} - - take(key: string, load: () => Promise<Sized<T>>): Promise<T> { - const hit = this.map.get(key); - if (hit !== undefined) { - this.map.delete(key); - this.map.set(key, hit); // most-recently used goes last - return hit.p; - } - if (this.maxBytes <= 0) return load().then((s) => s.value); - - const entry: { p: Promise<T>; bytes: number } = { p: null as never, bytes: 0 }; - entry.p = load() - .then((s) => { - // Only account for it if we're still the live entry for this key — - // a refresh() between issue and resolve must not resurrect it. - if (this.map.get(key) === entry) { - entry.bytes = s.bytes; - this.total += s.bytes; - this.evict(); - } - return s.value; - }) - .catch((e: unknown) => { - this.drop(key, entry); - throw e; - }); - this.map.set(key, entry); - return entry.p; - } - - private drop(key: string, entry: { bytes: number }): void { - if (this.map.get(key) === entry) { - this.map.delete(key); - this.total -= entry.bytes; - } - } - - private evict(): void { - while (this.total > this.maxBytes && this.map.size > MIN_CACHED_PAGES) { - const oldest = this.map.entries().next().value; - if (oldest === undefined) break; - this.map.delete(oldest[0]); - this.total -= oldest[1].bytes; - } - } - - clear(): void { - this.map.clear(); - this.total = 0; - } -} - -// The unbounded sibling, for the small-and-hot caches (manifests). Same -// don't-memoise-a-failure rule. -class PromiseMap<T> { - private map = new Map<string, Promise<T>>(); - - take(key: string, load: () => Promise<T>): Promise<T> { - const hit = this.map.get(key); - if (hit !== undefined) return hit; - const p = load().catch((e: unknown) => { - this.map.delete(key); - throw e; - }); - this.map.set(key, p); - return p; - } - - clear(): void { - this.map.clear(); - } -} - -// Opt-out for the composed site's declared origin (see LocalSource.publicOrigin). -// Set TRANSCRIPT_PLATFORM_LINKS=1 to cite platform watch pages instead, which is -// the right answer when a local build's declared site url is not actually -// deployed. -const PREFER_PLATFORM_LINKS = process.env.TRANSCRIPT_PLATFORM_LINKS === "1"; -type HubCorpusJson = { - kind?: string; - sites?: { siteId: string; title: string; url: string }[]; -}; +// The archive reader, as this server has always known it. +// +// This file WAS the reader: 1,386 lines holding the only good walk of the +// published shard scheme in the repo, plus three transports and two caches. It +// now lives in `yt-dlp-transcript-common/lib/archive/`, where the viewer, the +// offline cache and umtool can reach the same implementation instead of each +// re-deriving the walk (one-core plan, phase 2). Nothing about it changed in +// the move — the interface is the same 18 members under a new name, and there +// is deliberately still no per-record fetch. +// +// `ShardSource` stays as the name every importer here already spells, so this +// server's seven `./source` importers and `mcp/bench` (which spawns the server +// and never imports this file) are untouched. + +export type { + ArchiveReader as ShardSource, + ChannelGroups, + ChannelRef, + ClusterMembership, + DuplicateIndex, + IndexedVideo, + VideoAvailability, + VideoIndex, +} from "yt-dlp-transcript-common/lib/archive/reader"; + +export { + DEFAULT_PAGE_CONCURRENCY, + RemoteSource, + buildDuplicateIndex, +} from "yt-dlp-transcript-common/lib/archive/reader"; + +export { LocalSource } from "yt-dlp-transcript-common/lib/archive/reader-fs"; +export { HubSource } from "yt-dlp-transcript-common/lib/archive/reader-hub"; // One member site of a hub, as listed in the hub's corpus.json. Exposed so the -// source controller can resolve a `site`/`sites` token to a member origin. -export type HubSite = { siteId: string; title: string; url: string }; - -// ─── Local: read composed shards from a directory on disk ─── -// `dir` is a composed public dir (or any dir containing transcripts/<slug>/…). -// Prefers corpus.json for the channel list (names + counts); falls back to -// listing the transcripts/ subdirectories so it works even pre-Layer-1. -export class LocalSource implements ShardSource { - readonly label: string; - private aliases?: SearchAlias[]; - private groups?: ChannelGroups; - private index?: Promise<Map<string, IndexedVideo>>; - private duplicates?: Promise<DuplicateIndex>; - private stats?: Promise<ReadonlyMap<string, VideoStat>>; - // The composed site's own declared origin, learned from corpus.json the first - // time the channel list is read. Undefined = not looked at yet. - private siteOrigin: string | null | undefined; - - constructor(private dir: string) { - this.label = `local:${dir}`; - } - - // The deployed archilyzer viewer these shards were composed for, as declared - // by the dir's own corpus.json (`site.url`). A composed public dir is not an - // anonymous pile of JSON — it names the site it is the build output of — so - // citing that viewer is both possible and the right default: a reader - // following a citation lands in the archive, at the cited second, with the - // transcript around it, rather than on the platform page where the archive's - // whole point (that we still have a copy) is invisible. - // - // Populated by readChannels(), which every read path runs before it renders a - // link. Null when the dir ships no corpus.json (the bare directory-listing - // fallback), or when TRANSCRIPT_PLATFORM_LINKS=1 asks for platform links — - // both fall back to the platform watch page exactly as before. - publicOrigin(): string | null { - return this.siteOrigin ?? null; - } - - // Cached INCLUDING the negative answer, exactly like postsManifests below: - // live-chat search probes every channel in scope, and a video-only channel - // would otherwise cost one failed read per query. - private subsManifests = new PromiseMap<ChannelSubsManifest | null>(); - private digestManifests = new PromiseMap<ChannelDigestsManifest | null>(); - private subsPages = new PageCache<SubsDetail[]>(pageCacheBudgetBytes()); - private postsManifests = new PromiseMap<ChannelPostsManifest | null>(); - private transcriptManifests = new PromiseMap<ChannelTranscriptsManifest>(); - private transcriptPages = new PageCache<TranscriptDetail[]>(pageCacheBudgetBytes()); - - // Drop every cached read. Reached only through listChannels({refresh:true}) — - // the deliberate, explicit staleness escape hatch for a corpus rebuilt under - // a long-lived server. Not a TTL, on purpose. - private resetCaches(): void { - this.subsManifests.clear(); - this.digestManifests.clear(); - this.subsPages.clear(); - this.postsManifests.clear(); - this.transcriptManifests.clear(); - this.transcriptPages.clear(); - this.aliases = undefined; - this.groups = undefined; - this.index = undefined; - this.duplicates = undefined; - this.stats = undefined; - this.siteOrigin = undefined; - } - - subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { - return this.subsManifests.take(ch.slug, () => - readLocalJson<ChannelSubsManifest>( - path.join(this.dir, "subs", ch.slug, "manifest.json"), - "subsManifest", - ).catch(() => null), // channel ships no subs shards - ); - } - - subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { - return this.subsPages.take(`${ch.slug}:${page}`, () => - readLocalJsonSized<SubsDetail[]>( - path.join(this.dir, "subs", ch.slug, pageFileName(page)), - "subsPage", - ), - ); - } - - // Cached per channel INCLUDING the negative answer: most channels are - // video-only, and a posts-covering search would otherwise re-probe every one - // of them on every query. - postsManifest(ch: ChannelRef): Promise<ChannelPostsManifest | null> { - return this.postsManifests.take(ch.slug, () => - readLocalJson<ChannelPostsManifest>( - path.join(this.dir, "posts", ch.slug, "manifest.json"), - "postsManifest", - ).catch(() => null), // channel ships no posts shards - ); - } - - async postsPage(ch: ChannelRef, page: number): Promise<Post[]> { - return readLocalJson<Post[]>( - path.join(this.dir, "posts", ch.slug, postsPageFileName(page)), - "postsPage", - ); - } - - async loadAliases(): Promise<SearchAlias[]> { - if (this.aliases) return this.aliases; - try { - this.aliases = coerceAliasConfig( - await readLocalJson(path.join(this.dir, "search-aliases.json"), "aliases"), - ).aliases; - } catch { - this.aliases = []; // no/invalid file — search stays plain - } - return this.aliases; - } - - async loadGroups(): Promise<ChannelGroups> { - if (this.groups) return this.groups; - try { - this.groups = parseGroupsManifest( - await readLocalJson( - path.join(this.dir, "summaries", "manifest.json"), - "summariesManifest", - ), - ); - } catch { - this.groups = EMPTY_GROUPS; // no/invalid manifest — groups off - } - return this.groups; - } - - private channelList?: Promise<ChannelRef[]>; - - listChannels(opts: { refresh?: boolean } = {}): Promise<ChannelRef[]> { - if (opts.refresh) { - this.channelList = undefined; - this.resetCaches(); - } - this.channelList ??= this.readChannels().catch((e: unknown) => { - this.channelList = undefined; // don't memoise a failure - throw e; - }); - return this.channelList; - } - - private async readChannels(): Promise<ChannelRef[]> { - try { - const corpus = await readLocalJson<SiteCorpusJson>( - path.join(this.dir, "corpus.json"), - "corpus", - ); - const declared = corpus.site?.url?.trim(); - this.siteOrigin = - declared && !PREFER_PLATFORM_LINKS ? declared.replace(/\/+$/, "") : null; - if (Array.isArray(corpus.channels) && corpus.channels.length > 0) { - return corpus.channels.map((c) => ({ - key: c.slug, - slug: c.slug, - name: c.name ?? c.slug, - videoCount: c.videoCount, - groupId: c.groupId, - })); - } - } catch { - // no corpus.json — fall back to a directory listing - } - const transcriptsDir = path.join(this.dir, "transcripts"); - let entries: string[] = []; - try { - entries = await readdir(transcriptsDir); - } catch { - return []; - } - const channels: ChannelRef[] = []; - for (const slug of entries.sort()) { - // A channel dir has a manifest.json; skip stray files. - try { - await readFile(path.join(transcriptsDir, slug, "manifest.json"), "utf8"); - channels.push({ key: slug, slug, name: slug }); - } catch { - // not a channel dir - } - } - return channels; - } - - transcriptsManifest(ch: ChannelRef): Promise<ChannelTranscriptsManifest> { - return this.transcriptManifests.take(ch.slug, () => - readLocalJson<ChannelTranscriptsManifest>( - path.join(this.dir, "transcripts", ch.slug, "manifest.json"), - "transcriptsManifest", - ), - ); - } - - transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]> { - return this.transcriptPages.take(`${ch.slug}:${page}`, () => - readLocalJsonSized<TranscriptDetail[]>( - path.join(this.dir, "transcripts", ch.slug, pageFileName(page)), - "transcriptPage", - ), - ); - } - - // Local reads are CPU-bound on JSON.parse, so a modest window is all that is - // available to win: it overlaps the next page's read with this page's parse. - readonly pageConcurrency = DEFAULT_PAGE_CONCURRENCY; - - videoIndex(): Promise<VideoIndex> { - this.index ??= buildVideoIndex( - () => - readLocalJson<Manifest>( - path.join(this.dir, "summaries", "manifest.json"), - "summariesManifest", - ), - (page) => - readLocalJson<DisplaySummary[]>( - path.join(this.dir, "summaries", pageFileName(page)), - "summariesPage", - ), - ); - return this.index; - } - - availabilityMap(): Promise<ReadonlyMap<string, VideoAvailability>> { - return this.videoIndex(); - } - - digestsManifest(ch: ChannelRef): Promise<ChannelDigestsManifest | null> { - return this.digestManifests.take(ch.slug, () => - readLocalJson<ChannelDigestsManifest>( - path.join(this.dir, "digests", ch.slug, "manifest.json"), - "digestsManifest", - ).catch(() => null), // channel has no digests - ); - } - - digestPage(ch: ChannelRef, page: number): Promise<VideoDigest[]> { - return readLocalJson<VideoDigest[]>( - path.join(this.dir, "digests", ch.slug, digestPageFileName(page)), - "digestPage", - ); - } - - duplicateIndex(): Promise<DuplicateIndex> { - this.duplicates ??= readLocalJson<DuplicateReport>( - path.join(this.dir, DUPLICATES_FILENAME), - "duplicates", - ) - .then(buildDuplicateIndex) - .catch(() => buildDuplicateIndex(null)); // no report shipped — no clusters - return this.duplicates; - } - - statsIndex(): Promise<ReadonlyMap<string, VideoStat>> { - this.stats ??= buildStatsIndex( - () => - readLocalJson<StatsManifest>( - path.join(this.dir, "stats", "manifest.json"), - "statsManifest", - ), - (page) => - readLocalJson<VideoStat[]>( - path.join(this.dir, "stats", statsPageFileName(page)), - "statsPage", - ), - ); - return this.stats; - } -} - -// ─── Remote: fetch shards from a deployed site origin over HTTP ─── -export class RemoteSource implements ShardSource { - readonly label: string; - private base: string; - private aliases?: SearchAlias[]; - private groups?: ChannelGroups; - private index?: Promise<Map<string, IndexedVideo>>; - private duplicates?: Promise<DuplicateIndex>; - private stats?: Promise<ReadonlyMap<string, VideoStat>>; - private subsManifests = new PromiseMap<ChannelSubsManifest | null>(); - private digestManifests = new PromiseMap<ChannelDigestsManifest | null>(); - private subsPages: PageCache<SubsDetail[]>; - private postsManifests = new PromiseMap<ChannelPostsManifest | null>(); - private transcriptManifests = new PromiseMap<ChannelTranscriptsManifest>(); - private transcriptPages: PageCache<TranscriptDetail[]>; - - // Over HTTP the cost is latency, not parse, so a wider window is the dominant - // win — this is where bounded concurrency actually pays. - readonly pageConcurrency = 8; - - // `budgetBytes` lets a hub divide one memory ceiling across its members - // instead of granting each member the full budget (N members × 48 MB is not a - // budget, it's N budgets). - constructor(baseUrl: string, budgetBytes = pageCacheBudgetBytes()) { - this.base = baseUrl.replace(/\/+$/, ""); - this.label = `remote:${this.base}`; - this.subsPages = new PageCache<SubsDetail[]>(budgetBytes); - this.transcriptPages = new PageCache<TranscriptDetail[]>(budgetBytes); - } - - // The deployed site origin — the archilyzer viewer that owns these videos. - publicOrigin(): string | null { - return this.base; - } - - private resetCaches(): void { - this.subsManifests.clear(); - this.digestManifests.clear(); - this.subsPages.clear(); - this.postsManifests.clear(); - this.transcriptManifests.clear(); - this.transcriptPages.clear(); - this.aliases = undefined; - this.groups = undefined; - this.index = undefined; - this.duplicates = undefined; - this.stats = undefined; - } - - subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { - return this.subsManifests.take(ch.slug, async () => { - try { - const res = await fetch(`${this.base}/subs/${ch.slug}/manifest.json`); - return res.ok ? ((await res.json()) as ChannelSubsManifest) : null; - } catch { - return null; - } - }); - } - - subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { - return this.subsPages.take(`${ch.slug}:${page}`, () => - this.getJsonSized<SubsDetail[]>( - `/subs/${ch.slug}/${pageFileName(page)}`, - "subsPage", - ), - ); - } - - // Cached per channel including the negative answer — otherwise every - // posts-covering search costs one 404 per video-only channel. - postsManifest(ch: ChannelRef): Promise<ChannelPostsManifest | null> { - return this.postsManifests.take(ch.slug, async () => { - try { - const res = await fetch(`${this.base}/posts/${ch.slug}/manifest.json`); - return res.ok ? ((await res.json()) as ChannelPostsManifest) : null; - } catch { - return null; - } - }); - } - - postsPage(ch: ChannelRef, page: number): Promise<Post[]> { - return this.getJson(`/posts/${ch.slug}/${postsPageFileName(page)}`, "postsPage"); - } - - videoIndex(): Promise<VideoIndex> { - this.index ??= buildVideoIndex( - async () => { - const res = await fetch(`${this.base}/summaries/manifest.json`); - return res.ok ? ((await res.json()) as Manifest) : null; - }, - async (page) => { - const res = await fetch(`${this.base}/summaries/${pageFileName(page)}`); - return res.ok ? ((await res.json()) as DisplaySummary[]) : null; - }, - ); - return this.index; - } - - availabilityMap(): Promise<ReadonlyMap<string, VideoAvailability>> { - return this.videoIndex(); - } - - digestsManifest(ch: ChannelRef): Promise<ChannelDigestsManifest | null> { - return this.digestManifests.take(ch.slug, async () => { - try { - const res = await fetch(`${this.base}/digests/${ch.slug}/manifest.json`); - return res.ok ? ((await res.json()) as ChannelDigestsManifest) : null; - } catch { - return null; - } - }); - } - - digestPage(ch: ChannelRef, page: number): Promise<VideoDigest[]> { - return this.getJson( - `/digests/${ch.slug}/${digestPageFileName(page)}`, - "digestPage", - ); - } - - duplicateIndex(): Promise<DuplicateIndex> { - this.duplicates ??= (async () => { - try { - const res = await fetch(`${this.base}/${DUPLICATES_FILENAME}`); - return buildDuplicateIndex( - res.ok ? ((await res.json()) as DuplicateReport) : null, - ); - } catch { - return buildDuplicateIndex(null); - } - })(); - return this.duplicates; - } - - statsIndex(): Promise<ReadonlyMap<string, VideoStat>> { - this.stats ??= buildStatsIndex( - async () => { - const res = await fetch(`${this.base}/stats/manifest.json`); - return res.ok ? ((await res.json()) as StatsManifest) : null; - }, - async (page) => { - const res = await fetch(`${this.base}/stats/${statsPageFileName(page)}`); - return res.ok ? ((await res.json()) as VideoStat[]) : null; - }, - ); - return this.stats; - } - - async loadAliases(): Promise<SearchAlias[]> { - if (this.aliases) return this.aliases; - try { - const res = await fetch(`${this.base}/search-aliases.json`); - this.aliases = res.ok - ? coerceAliasConfig(await res.json()).aliases - : []; - } catch { - this.aliases = []; - } - return this.aliases; - } - - async loadGroups(): Promise<ChannelGroups> { - if (this.groups) return this.groups; - try { - const res = await fetch(`${this.base}/summaries/manifest.json`); - this.groups = res.ok - ? parseGroupsManifest(await res.json()) - : EMPTY_GROUPS; - } catch { - this.groups = EMPTY_GROUPS; - } - return this.groups; - } - - // Fetches as TEXT so the byte size is knowable — the page cache's budget is - // denominated in raw bytes, and res.json() throws the length away. - private async getJsonSized<T>( - p: string, - kind: string, - ): Promise<{ value: T; bytes: number }> { - const res = await fetch(`${this.base}${p}`); - if (!res.ok) { - throw new Error(`GET ${this.base}${p} -> ${res.status} ${res.statusText}`); - } - const raw = await res.text(); - recordRead(kind, raw.length); - return { value: JSON.parse(raw) as T, bytes: raw.length }; - } - - private async getJson<T>(p: string, kind = "json"): Promise<T> { - return (await this.getJsonSized<T>(p, kind)).value; - } - - private channelList?: Promise<ChannelRef[]>; - - listChannels(opts: { refresh?: boolean } = {}): Promise<ChannelRef[]> { - if (opts.refresh) { - this.channelList = undefined; - this.resetCaches(); - } - this.channelList ??= this.readChannels().catch((e: unknown) => { - this.channelList = undefined; // don't memoise a failure - throw e; - }); - return this.channelList; - } - - private async readChannels(): Promise<ChannelRef[]> { - const corpus = await this.getJson<SiteCorpusJson>("/corpus.json", "corpus"); - return (corpus.channels ?? []).map((c) => ({ - key: c.slug, - slug: c.slug, - name: c.name ?? c.slug, - videoCount: c.videoCount, - groupId: c.groupId, - siteUrl: this.base, - })); - } - - transcriptsManifest(ch: ChannelRef): Promise<ChannelTranscriptsManifest> { - return this.transcriptManifests.take(ch.slug, () => - this.getJson<ChannelTranscriptsManifest>( - `/transcripts/${ch.slug}/manifest.json`, - "transcriptsManifest", - ), - ); - } - - transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]> { - return this.transcriptPages.take(`${ch.slug}:${page}`, () => - this.getJsonSized<TranscriptDetail[]>( - `/transcripts/${ch.slug}/${pageFileName(page)}`, - "transcriptPage", - ), - ); - } -} - -// ─── Hub: federate over every member site listed in the hub corpus.json ─── -// Each member is its own RemoteSource; channels are namespaced by site so keys -// stay unique, and manifest/page calls dispatch to the owning member. -export class HubSource implements ShardSource { - readonly label: string; - readonly hubBase: string; - private members = new Map<string, RemoteSource>(); // siteId -> source - private aliases?: SearchAlias[]; - private index?: Promise<Map<string, IndexedVideo>>; - private duplicates?: Promise<DuplicateIndex>; - private stats?: Promise<ReadonlyMap<string, VideoStat>>; - private sites?: Promise<HubSite[]>; - - // A hub is N HTTP origins, so the latency argument for a wide window applies - // even harder than for a single remote. - readonly pageConcurrency = 8; - // Optional subset allowlist of member siteIds. Undefined = federate every - // member; a set restricts listChannels() to those members (site discovery via - // listSites() stays unfiltered so a picker can still see all members). - private allowSiteIds?: Set<string>; - - constructor(hubUrl: string, allowSiteIds?: string[]) { - this.hubBase = hubUrl.replace(/\/+$/, ""); - this.allowSiteIds = - allowSiteIds && allowSiteIds.length > 0 - ? new Set(allowSiteIds) - : undefined; - this.label = this.allowSiteIds - ? `hub:${this.hubBase} (${this.allowSiteIds.size} site(s))` - : `hub:${this.hubBase}`; - } - - // Fetch the hub's corpus.json and return its member sites — UNFILTERED (the - // full membership), even when this source is scoped to a subset, so a picker - // (list_sources / use_source) can show every member. - // - // Memoised on the promise: readChannels() and videoIndex() both need it, so - // an un-memoised version fetched the hub roster twice on a cold hub search. - // A failure is not memoised. - listSites(): Promise<HubSite[]> { - this.sites ??= this.readSites().catch((e: unknown) => { - this.sites = undefined; - throw e; - }); - return this.sites; - } - - private async readSites(): Promise<HubSite[]> { - const res = await fetch(`${this.hubBase}/corpus.json`); - if (!res.ok) { - throw new Error( - `GET ${this.hubBase}/corpus.json -> ${res.status} ${res.statusText}`, - ); - } - const hub = (await res.json()) as HubCorpusJson; - return hub.sites ?? []; - } - - // One page-cache budget for the whole hub, divided across its members — N - // members must not each get the full ceiling. Floored so a large federation - // still caches something per member. - private memberBudget(memberCount: number): number { - const total = pageCacheBudgetBytes(); - const floor = 8 * 1024 * 1024; - return Math.max(floor, Math.floor(total / Math.max(1, memberCount))); - } - - // A hub can ship its own /search-aliases.json (the merged federation-wide - // dictionary); if it doesn't, aliases are simply off for hub-wide search. - async loadAliases(): Promise<SearchAlias[]> { - if (this.aliases) return this.aliases; - try { - const res = await fetch(`${this.hubBase}/search-aliases.json`); - this.aliases = res.ok - ? coerceAliasConfig(await res.json()).aliases - : []; - } catch { - this.aliases = []; - } - return this.aliases; - } - - // In hub mode each group is itself a federated member site (a different model - // — accent-per-origin), so hub-wide channel-group tokens are deferred: return - // the empty fallback. Multi-channel scoping still works (member groupIds carry - // through listChannels, they just don't resolve against hub-level groups). - async loadGroups(): Promise<ChannelGroups> { - return EMPTY_GROUPS; - } - - private memberFor(siteId: string): RemoteSource { - const m = this.members.get(siteId); - if (!m) throw new Error(`unknown hub member site: ${siteId}`); - return m; - } - - private channelList?: Promise<ChannelRef[]>; - - listChannels(opts: { refresh?: boolean } = {}): Promise<ChannelRef[]> { - if (opts.refresh) { - this.channelList = undefined; - this.sites = undefined; - this.index = undefined; - this.duplicates = undefined; - this.stats = undefined; - this.aliases = undefined; - // Members hold their own manifest/page caches; drop them wholesale so a - // refresh means the same thing federation-wide as it does locally. - this.members.clear(); - } - this.channelList ??= this.readChannels().catch((e: unknown) => { - this.channelList = undefined; // don't memoise a failure - throw e; - }); - return this.channelList; - } - - // Populating `members` must stay INSIDE the memoised call: memberFor() - // depends on it, so a memo that skipped this would leave every - // transcriptPage/postsPage dispatch throwing "unknown hub member site". - private async readChannels(): Promise<ChannelRef[]> { - const sites = (await this.listSites()).filter( - (s) => !this.allowSiteIds || this.allowSiteIds.has(s.siteId), - ); - const budget = this.memberBudget(sites.length); - const all: ChannelRef[] = []; - // Sequential member fetches keep it simple and polite; the channel count is - // small. A failing member is skipped rather than failing the whole list. - for (const site of sites) { - const remote = - this.members.get(site.siteId) ?? new RemoteSource(site.url, budget); - this.members.set(site.siteId, remote); - try { - const channels = await remote.listChannels(); - for (const c of channels) { - all.push({ - ...c, - key: `${site.siteId}/${c.slug}`, - siteId: site.siteId, - siteTitle: site.title, - siteUrl: site.url, - }); - } - } catch { - // skip an unreachable member - } - } - return all; - } - - transcriptsManifest(ch: ChannelRef): Promise<ChannelTranscriptsManifest> { - if (!ch.siteId) throw new Error("hub channel ref missing siteId"); - return this.memberFor(ch.siteId).transcriptsManifest(ch); - } - - transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]> { - if (!ch.siteId) throw new Error("hub channel ref missing siteId"); - return this.memberFor(ch.siteId).transcriptPage(ch, page); - } - - // A hub has no single viewer origin — each video's origin is its member - // site's url (carried on the ChannelRef as `siteUrl`), which momentUrl prefers - // per-video. Return null so we never mint a wrong-origin viewer link. - publicOrigin(): string | null { - return null; - } - - async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { - if (!ch.siteId) return null; - try { - return await this.memberFor(ch.siteId).subsManifest(ch); - } catch { - return null; // member not yet registered / unreachable - } - } - - subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { - if (!ch.siteId) throw new Error("hub channel ref missing siteId"); - return this.memberFor(ch.siteId).subsPage(ch, page); - } - - async postsManifest(ch: ChannelRef): Promise<ChannelPostsManifest | null> { - if (!ch.siteId) return null; - try { - return await this.memberFor(ch.siteId).postsManifest(ch); - } catch { - return null; // member not yet registered / unreachable - } - } - - postsPage(ch: ChannelRef, page: number): Promise<Post[]> { - if (!ch.siteId) throw new Error("hub channel ref missing siteId"); - return this.memberFor(ch.siteId).postsPage(ch, page); - } - - async digestsManifest(ch: ChannelRef): Promise<ChannelDigestsManifest | null> { - if (!ch.siteId) return null; - try { - return await this.memberFor(ch.siteId).digestsManifest(ch); - } catch { - return null; // member not yet registered / unreachable - } - } - - digestPage(ch: ChannelRef, page: number): Promise<VideoDigest[]> { - if (!ch.siteId) throw new Error("hub channel ref missing siteId"); - return this.memberFor(ch.siteId).digestPage(ch, page); - } - - // Merge each member's video index. Keys are member-local slugs - // (`<channelSlug>/<id>`) — the same slug a member's transcript page records - // carry — so a per-record lookup joins correctly. Built lazily/cached. - videoIndex(): Promise<VideoIndex> { - this.index ??= this.buildMergedIndex(); - return this.index; - } - - private async buildMergedIndex(): Promise<Map<string, IndexedVideo>> { - const merged = new Map<string, IndexedVideo>(); - let sites: HubSite[]; - try { - sites = (await this.listSites()).filter( - (s) => !this.allowSiteIds || this.allowSiteIds.has(s.siteId), - ); - } catch { - return merged; - } - const budget = this.memberBudget(sites.length); - for (const site of sites) { - const remote = - this.members.get(site.siteId) ?? new RemoteSource(site.url, budget); - this.members.set(site.siteId, remote); - try { - for (const [slug, rec] of await remote.videoIndex()) { - merged.set(slug, rec); - } - } catch { - // skip an unreachable member - } - } - return merged; - } - - availabilityMap(): Promise<ReadonlyMap<string, VideoAvailability>> { - return this.videoIndex(); - } - - // Duplicate clusters are detected WITHIN a site, so federating them is a - // merge of per-member maps and nothing more — this deliberately does not try - // to detect mirrors ACROSS member sites. Two sites holding the same recording - // is a real thing, but nothing has compared their transcripts, and inventing - // a cross-site cluster here would be asserting a duplicate no detector ever - // confirmed. - duplicateIndex(): Promise<DuplicateIndex> { - this.duplicates ??= this.mergeMembers((m) => m.duplicateIndex()); - return this.duplicates; - } - - statsIndex(): Promise<ReadonlyMap<string, VideoStat>> { - this.stats ??= this.mergeMembers((m) => m.statsIndex()); - return this.stats; - } - - // Merge one lazily-read layer across every member site, keyed by the - // member-local slug — the same shape and the same tolerance as - // buildMergedIndex (an unreachable member is skipped, not fatal). - private async mergeMembers<T>( - read: (m: RemoteSource) => Promise<ReadonlyMap<string, T>>, - ): Promise<Map<string, T>> { - const merged = new Map<string, T>(); - let sites: HubSite[]; - try { - sites = (await this.listSites()).filter( - (s) => !this.allowSiteIds || this.allowSiteIds.has(s.siteId), - ); - } catch { - return merged; - } - const budget = this.memberBudget(sites.length); - for (const site of sites) { - const remote = - this.members.get(site.siteId) ?? new RemoteSource(site.url, budget); - this.members.set(site.siteId, remote); - try { - for (const [k, v] of await read(remote)) merged.set(k, v); - } catch { - // skip an unreachable member - } - } - return merged; - } -} +// source controller can resolve a `site`/`sites` token to a member origin. Now +// a projection of the entry the hub actually publishes (`HubCorpusSite`), so +// the read and write spellings cannot drift. +export type { HubSite } from "yt-dlp-transcript-common/lib/archive/contract"; + +// The opt-in read/byte counters mcp/bench reads off stderr. +export { + ioStatsEnabled, + ioStatsSnapshot, + recordRead, + type IoStats, +} from "yt-dlp-transcript-common/lib/archive/io-stats"; diff --git a/plans/one-core-phase-2.md b/plans/one-core-phase-2.md @@ -27,9 +27,9 @@ original wording. `common/components/{transcript,subs,posts,digest,summaries,stats,duplicates,aliases}Cache.ts` — 825 lines, all keyed by `OriginId` (`components/originId.ts`). Seven more walk sites: `export/app/lib/offlineCache.ts`, `umtool/report-to-video/cues.mjs`, - `common/lib/siteRegistry.ts:287`, `export/app/lib/SearchDataContext.tsx`, - `export/app/lib/SearchSessionContext.tsx`, `export/app/lib/useAskChat.ts`, and the - deliberate copy in `export/app/lib/searchIndex.worker.ts:39-41` (a worker cannot import + `common/components/siteRegistry.ts:287`, `common/components/SearchDataContext.tsx`, + `common/components/SearchSessionContext.tsx`, `export/app/ask/useAskChat.ts`, and the + deliberate copy in `common/components/searchIndex.worker.ts:39-41` (a worker cannot import the reader; it **stays**, guarded by the SW/contract equality test in S2a). 2. **The reader interface is mcp's 18-member `ShardSource`** (`mcp/src/source.ts:292-372`), not the umbrella's four methods. `record(layer, slug, id)` must **not** be added: a @@ -119,10 +119,11 @@ parallel slices reuse it instead of each rebuilding it. The eight caches become `memo(originId, () => reader.X(...))` — the memo stays, the walk goes. `offlineCache.ts` enumerates every layer that has a manifest plus `ROOT_FILES` (`/duplicates.json`, stats) — this closes the duplicates-page-dead-offline gap. -`export/public/site-sw.js:27` `SHARD_RE` and the evict prefixes (`:138-142`) widen to -match. The SW layer list stays hand-written (a service worker cannot import), guarded by a -`contract.test.ts` case that reads both SW files and asserts equality with -`CONTRACT.layers`. +`export/service-worker/site-sw.js:27` `SHARD_RE` and the evict prefixes (`:138-142`) widen +to match, and so does `sw-hub.js:26`, which today lacks `digests` altogether. The SW layer +list stays hand-written (a service worker cannot import), guarded by a `contract.test.ts` +case that reads both files under `export/service-worker/` and asserts equality with +`CONTRACT.layers`. (`export/public/sw.js` is the gitignored composed copy — never edit it.) Not in scope: refactoring `SearchDataContext` / `SearchSessionContext` / `siteRegistry` state. They keep their state; only the fetch walk moves. @@ -238,3 +239,146 @@ Monitor, commit incrementally, add by path, never `next build` in the primary ch ## Record Filled in as slices ship: sha range, actual gate numbers, every divergence from this plan. + +### S1 — shipped 2026-09-12 + +Branch `one-core/phase-2-s1`, off `c7f7b90` (the `integrate/2026-09-storage-priority` +tip). Six commits, `8d60ad9` → `85891df` (this note), unmerged. **No URL shape moved, no +`corpus.json` byte moved, no CONTRACT version moved, no architecture allow-list +entry added or burned.** + +| commit | what | +|---|---| +| `8d60ad9` | `archive/contract.ts` + `archive/io-stats.ts`; `buildSiteCorpus` calls the URL builders | +| `3c631a2` | the reader: `reader.ts` / `reader-fs.ts` / `reader-hub.ts` + `reader.test.ts` | +| `43be5f9` | `mcp/src/source.ts`: 1,386 lines → 47, a re-export | +| `f2f688d` | the five `lib/search/*` stubs S3 fills | +| `f085667` | `plans/tools/compose-fixture-one-youtube-channel/` | +| `85891df` | this note | + +#### Four divergences from the plan above, each because the code said so + +**1. `contract.ts` OWNS `CONTRACT` and `pageFileName`; it does not re-export +them.** The plan (and the umbrella) said re-export from `lib/corpus.ts` / +`lib/manifest.ts`. That is a hard ESM failure, not a style question: `corpus.ts` +must import the URL builders (that is the point — one definition of the shape it +emits), `contract.ts` needs `CONTRACT.pagePad` for `pageFileName`, and +`manifest.ts` reads `CONTRACT.manifest` **at module scope**. Any arrangement that +leaves `CONTRACT` in `corpus.ts` closes the loop `corpus → archive/contract → +manifest → corpus`, and the first module to be imported gets +`ReferenceError: Cannot access 'CONTRACT' before initialization`. So the contract +module is the BOTTOM of the stack — it imports only `lib/duplicates.ts` — and +`corpus.ts` / `manifest.ts` re-export the names. Every existing import site +(`from "./corpus"`, `from "./manifest"`, the three `*PageFileName` aliases) is +byte-identical. Verified by importing each of the ten modules in the cycle under +`tsx`, entry-point by entry-point. + +**2. The four hub spellings collapse to TWO types plus one `Pick`, not one.** +The reason is on the wire. `HubMemberInput` and compose-hub's `HubSiteEntry` are +the same direction and the same vocabulary (`siteTitle`/`siteUrl`) — those +genuinely become one, and S2c deletes the local copy. But `HubCorpusSite` is the +entry **published** in a hub `corpus.json`, and it spells the same member +`title`/`url` with two derived pointers beside it. `corpus.json` is frozen, so +the emitted shape cannot be renamed to match the input shape. What the slice does +instead is make mcp's `HubSite` a `Pick<HubCorpusSite, "siteId"|"title"|"url">`, +so the read-back spelling can no longer drift from the published one. + +**3. `shipsPwa` reads `process.env.INSTANCE_MODE` bare, with no `typeof process` +guard** — unlike `io-stats.ts` and the page-cache knob, which are guarded. The +S1 note first claimed Next inlines that expression into the client bundle; the +S1 review checked and it does not (`INSTANCE_MODE` is neither `NEXT_PUBLIC_` +nor in a `next.config.ts` `env:` block). The real reason the bare read is fine: +the predicate is byte-equivalent to `mode.ts:26-29` and `compose-site.ts:207-209`, +and its only caller today is `export/app/layout.tsx` — a **server** component — +so no browser ever evaluates it. `contract.ts` is not reachable from any +browser-but-not-Next context (the service workers import nothing; the search +worker keeps its own copy). **S2c: keep `shipsPwa` server-called when it deletes +the two copies**, or add the guard then; do not reason from the inlining claim. + +**4. `stats/` gets URLs without joining `CONTRACT.layers`.** It follows the same +manifest → page walk, but `corpus.json`'s `shardScheme` does not document it, so +adding it to the published layer list would be a wire change. The builders take a +wider `ArchiveTree = ContractLayer | "stats"` instead and the layer list stays +frozen. + +One thing the plan did not mention and that had to change: **common's test glob +only reached one directory deep**, so `lib/archive/*.test.ts` would have been +collected by nobody. `common/package.json`'s `test` script now also globs +`{lib,…}/*/*.test.ts`. Nothing else lives two deep today, so no existing test +moved in or out. + +#### Gates + +- `pnpm -r exec tsc --noEmit` — clean in all six packages, after every commit. +- `pnpm --filter yt-dlp-transcript-common test` — **1077 passed / 0 failed** + (baseline 1051 at `c7f7b90`, + 9 `contract.test.ts`, + 17 `reader.test.ts`; + none lost). The architecture test passes with its allow-list untouched. +- `pnpm --filter yt-dlp-transcript-mcp test` — **205 passed / 0 failed**, + unchanged. +- `pnpm test:scripts` — **71 passed / 1 skipped**, unchanged. +- `pnpm --filter export exec next build` — **compiled successfully**, 11 static + pages. This is the only test that `reader-fs.ts` is unreachable from a client + module. (The one warning is the pre-existing NFT trace on + `next.config.ts → channelMedia.ts → controller/channels.ts → app/offline/page.tsx` + — the documented reason this app does not use `output: "standalone"`.) +- `pnpm --filter yt-dlp-transcript-mcp bench --repeat 1 --force`, before at + `c7f7b90` and after, both `--local` the same composed 1.3 GB site, identical + fingerprint (3,358 summaries videos; stats, duplicates and digests all + present). **Every structural counter identical, byte for byte:** + + | case | reads | bytes parsed | + |---|---|---| + | cold-channels | 0 → 0 | 0 → 0 | + | rare, whole corpus | 170 → 170 | 1,335,885,512 → 1,335,885,512 | + | common, whole corpus | 170 → 170 | 1,364,679,296 → 1,364,679,296 | + | channel-scoped | 4 → 4 | 28,847,054 → 28,847,054 | + | date-scoped (filter-first) | 32 → 32 | 213,451,503 → 213,451,503 | + | state-scoped (filter-first) | 9 → 9 | 73,399,988 → 73,399,988 | + | enumerate, whole corpus | 170 → 170 | 1,364,679,296 → 1,364,679,296 | + | get_transcripts × 20 ids | 4 → 4 | 28,847,054 → 28,847,054 | + + The per-case scan notes match too, `pruned` flags included (date-scoped 28 + pages pruned, state-scoped 9). Wall ms is noise and is not quoted. +- **compose-site byte-identity** over the FACTS.md fixture recipe, at `c7f7b90` + and at the tip: 16 files under `public/`, 7 under `index/`, identical modulo + the build clock. The artifact and the normalising diff are committed at + `plans/tools/compose-fixture-one-youtube-channel/`; re-running the README's own + commands into two fresh temp dirs reproduces it. +- **e2e**, behind the queue lock from the worktree (port block #9 — + 3901/3911/3920): export `e2e` **172 passed / 0 failed**, `e2e:hub` + **5 passed / 0 failed**. `e2e:2origin` **could not run**, and the reason is + not this slice — see below. + +#### `e2e:2origin` is red on the base commit, for a reason S1 does not touch + +It never reaches a spec. `playwright.2origin.config.ts:16` calls +`e2e-2origin/globalSetup.ts:142`, which shells `pnpm run build:hub`, and the hub +build dies prerendering `/ask`: + +``` +Error: useSearchSession must be used within a SearchSessionProvider +Export encountered an error on /(workspace)/ask/page: /ask, exiting the build. +``` + +**Verified on `c7f7b90` itself** — `git switch --detach c7f7b90` in this +worktree, `pnpm --filter export run build:hub`: the same error, on the same +page, from the same two chunks (`SearchSessionContext`, `AskChat`). This is the +"hub `/ask` broken on main blocks `e2e:2origin`" failure already recorded with +the export responsive redesign, and nothing in S1 is in that path: the reader +never enters a React tree, and the provider is +`common/components/SearchSessionContext.tsx` — S2a/S3's file, untouched here. +(Note for those slices: four paths in §Corrections above are wrong, checked +2026-09-12 — `SearchSessionContext.tsx` and `SearchDataContext.tsx` are under +`common/components/`, not `export/app/lib/`; `useAskChat.ts` is +`export/app/ask/useAskChat.ts`; `siteRegistry.ts` is `common/components/`, not +`common/lib/`; and the deliberate worker copy is +`common/components/searchIndex.worker.ts`, not `export/app/lib/`. The files all +exist; only the directories in the plan are wrong.) + +Worth noting what it DOES prove: the site-mode `next build` passed here and the +HUB-mode one compiled and type-checked before the prerender — so the bundle +resolves `lib/archive/*` in both modes and never pulls `reader-fs.ts` into a +client chunk. The prerender is the only step that fails, and it fails at base. + +`plans/tools/jeralyzer-corpus-2026-09-12.json` was not re-fetched; the live diff +is S3's gate, after the search pipeline lands. diff --git a/plans/tools/compose-fixture-one-youtube-channel/README.md b/plans/tools/compose-fixture-one-youtube-channel/README.md @@ -0,0 +1,82 @@ +# A composed site, small enough to diff by hand + +This is the output of `compose-site.ts` over the repo's smallest real corpus +fixture, captured at the tip of **one-core phase 2 slice S1** +(`one-core/phase-2-s1`, branched from `c7f7b90`). It exists so the parallel +slices — S2a, S2b, S2c, S3 and S0-pause — can prove "the contract did not move" +against a committed artifact instead of each rebuilding a baseline and hoping +the two recipes matched. + +`plans/FACTS.md` § "How to compose a fixture site offline" is the recipe; this +directory is one run of it plus the two things the recipe leaves to the reader: +the hand-written `site.json`, and a diff that knows what is allowed to change. + +## What is here + +| Path | What | +|---|---| +| `site.json` | The hand-written `sites/testsite/site.json`. The recipe's only authored input. | +| `public/` | 16 files — what a deploy serves. | +| `index/` | 7 files — the intermediate index the compose step copies from. | +| `compose-diff.sh` | `./compose-diff.sh <expected> <actual>` — a normalising diff. | + +## How it was produced + +The corpus is `editor/e2e/fixtures/test-transcripts/one-youtube-channel-with-data` +(one channel, one video, one VTT). Nothing here reads or writes the real +`transcripts/`. + +```sh +F=$(mktemp -d) # the corpus root +O=$(mktemp -d); mkdir -p $O/public $O/index + +cp -r editor/e2e/fixtures/test-transcripts/one-youtube-channel-with-data/channels $F/ +mkdir -p $F/sites/testsite +cp plans/tools/compose-fixture-one-youtube-channel/site.json $F/sites/testsite/site.json +echo '{}' > $F/settings.json + +cd common +env TRANSCRIPTS_DIR=$F SITES_DIR=$F/sites \ + EXPORT_PUBLIC_DIR=$O/public EXPORT_INDEX_DIR=$O/index \ + SETTINGS_FILE=$F/settings.json SITE_ID=testsite \ + pnpm exec tsx bin/build-index.ts +env TRANSCRIPTS_DIR=$F SITES_DIR=$F/sites \ + EXPORT_PUBLIC_DIR=$O/public EXPORT_INDEX_DIR=$O/index \ + SETTINGS_FILE=$F/settings.json SITE_ID=testsite \ + pnpm exec tsx bin/compose-site.ts +cd .. + +plans/tools/compose-fixture-one-youtube-channel/compose-diff.sh \ + plans/tools/compose-fixture-one-youtube-channel/public $O/public +plans/tools/compose-fixture-one-youtube-channel/compose-diff.sh \ + plans/tools/compose-fixture-one-youtube-channel/index $O/index +``` + +Both should print `IDENTICAL modulo the build clock`. Anything else is a real +change to the wire, and belongs in a commit that says so. + +**`build-index.ts` writes `$F/index.mdb` and caches a schema version in it.** +Compose the fixture into a fresh `$F` each time, or the second run reports +`0 built, 1 up to date` and copies nothing. + +## What the diff normalises, and why only this + +- **`"generatedAt": "<iso>"`**, in both the compact and the pretty-printed + spellings. It appears in `corpus.json`, `site.json`, the five manifests, and + the archive zip's `channel.json`. +- **The archive zip's own bytes.** A zip's entry mtimes are the build clock, so + two runs of the same input differ by a byte or two in the compressed stream — + and `archives/manifest.json` records that size. Zips are therefore compared by + their unpacked, `generatedAt`-normalised **entries**, which do diff clean, and + the `bytes` field is normalised with them. + +Nothing else is excused. In particular the URL shapes in `corpus.json` — the +absolute-with-`siteUrl` form this fixture exercises — are the thing phase 2's +`lib/archive/contract.ts` must reproduce, so a diff there is the failure this +directory exists to catch. + +## The site.json is deliberately minimal but NOT trivial + +It sets `siteUrl`, which is what makes every manifest pointer in `corpus.json` +absolute. A fixture without one would pass a reader that had quietly dropped the +origin. diff --git a/plans/tools/compose-fixture-one-youtube-channel/compose-diff.sh b/plans/tools/compose-fixture-one-youtube-channel/compose-diff.sh @@ -0,0 +1,38 @@ +#!/bin/bash +# Diff two composed site outputs modulo the build clock. +# +# Normalised away, and only these: +# - "generatedAt": "<iso>" in every JSON (compact and pretty forms) +# - the archive zip's own bytes: entry mtimes are the build clock, so a zip +# built twice differs by a byte or two even when its ENTRIES are identical. +# Zips are therefore compared by unpacked, generatedAt-normalised contents, +# and archives/manifest.json's `bytes` (the zip's size) with them. +A=$1; B=$2 +norm() { sed -E -e 's/"generatedAt": *"[^"]*"/"generatedAt":"X"/g' "$1"; } +normarch() { sed -E -e 's/"generatedAt": *"[^"]*"/"generatedAt":"X"/g' -e 's/"bytes": *[0-9]+/"bytes":N/g' "$1"; } +rc=0 +( cd "$A" && find . -type f | sort ) > /tmp/fa.$$ +( cd "$B" && find . -type f | sort ) > /tmp/fb.$$ +if ! diff -q /tmp/fa.$$ /tmp/fb.$$ >/dev/null; then echo "FILE LIST DIFFERS"; diff /tmp/fa.$$ /tmp/fb.$$; rc=1; fi +while read -r f; do + if [[ "$f" == *.zip ]]; then + ta=$(mktemp -d); tb=$(mktemp -d) + unzip -qq -o "$A/$f" -d "$ta"; unzip -qq -o "$B/$f" -d "$tb" + while read -r e; do + if ! diff <(norm "$ta/$e") <(norm "$tb/$e") >/dev/null; then + echo "ZIP ENTRY DIFFERS: $f!$e"; rc=1 + fi + done < <(cd "$ta" && find . -type f | sort) + if ! diff <(cd "$ta" && find . -type f|sort) <(cd "$tb" && find . -type f|sort) >/dev/null; then + echo "ZIP ENTRY LIST DIFFERS: $f"; rc=1; fi + rm -rf "$ta" "$tb" + elif [[ "$f" == ./archives/manifest.json ]]; then + if ! diff <(normarch "$A/$f") <(normarch "$B/$f") >/dev/null; then + echo "DIFFERS: $f"; diff <(normarch "$A/$f") <(normarch "$B/$f") | head -8; rc=1; fi + elif ! diff <(norm "$A/$f") <(norm "$B/$f") >/dev/null; then + echo "DIFFERS: $f"; diff <(norm "$A/$f") <(norm "$B/$f") | head -8; rc=1 + fi +done < /tmp/fa.$$ +rm -f /tmp/fa.$$ /tmp/fb.$$ +[ $rc -eq 0 ] && echo "IDENTICAL modulo the build clock ($(cd "$A" && find . -type f | wc -l) files)" +exit $rc diff --git a/plans/tools/compose-fixture-one-youtube-channel/index/shared/transcripts/test-youtube/manifest.json b/plans/tools/compose-fixture-one-youtube-channel/index/shared/transcripts/test-youtube/manifest.json @@ -0,0 +1 @@ +{"version":1,"channelSlug":"test-youtube","pageCount":1,"maxPageBytes":8388608,"generatedAt":"2026-09-12T06:01:16.087Z","slugToPage":{"20240101_test1234567":0}} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/index/shared/transcripts/test-youtube/page-0000.json b/plans/tools/compose-fixture-one-youtube-channel/index/shared/transcripts/test-youtube/page-0000.json @@ -0,0 +1 @@ +[{"slug":"test-youtube/20240101_test1234567","id":"20240101_test1234567","channelSlug":"test-youtube","title":"Synthetic Test Video","uploadDate":"20240101","duration":120,"channel":"Test YouTube Channel","description":"A synthetic test video for e2e fixtures.","tags":[],"isLivestream":false,"ageRestricted":false,"platform":"youtube","webpageUrl":"https://www.youtube.com/watch?v=20240101_test1234567","cues":[]}] +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/index/sites/testsite/digests/manifest.json b/plans/tools/compose-fixture-one-youtube-channel/index/sites/testsite/digests/manifest.json @@ -0,0 +1 @@ +{"version":1,"channels":[],"totalCount":0,"generatedAt":"2026-09-12T06:01:16.116Z","siteId":"testsite"} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/index/sites/testsite/posts/manifest.json b/plans/tools/compose-fixture-one-youtube-channel/index/sites/testsite/posts/manifest.json @@ -0,0 +1 @@ +{"version":1,"channels":[],"totalCount":0,"generatedAt":"2026-09-12T06:01:16.115Z","siteId":"testsite"} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/index/sites/testsite/subs/manifest.json b/plans/tools/compose-fixture-one-youtube-channel/index/sites/testsite/subs/manifest.json @@ -0,0 +1 @@ +{"version":4,"channels":[],"totalCount":0,"liveChatTotalCount":0,"generatedAt":"2026-09-12T06:01:16.115Z","groups":[{"id":"default","name":"All channels","selectedByDefault":true,"inline":true}],"defaultGroupId":"default","siteId":"testsite"} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/index/sites/testsite/summaries/manifest.json b/plans/tools/compose-fixture-one-youtube-channel/index/sites/testsite/summaries/manifest.json @@ -0,0 +1 @@ +{"version":3,"totalCount":1,"pageSize":1000,"pageCount":1,"generatedAt":"2026-09-12T06:01:16.115Z","channels":[{"slug":"test-youtube","name":"Test YouTube Channel","count":1,"groupId":"default"}],"groups":[{"id":"default","name":"All channels","selectedByDefault":true,"inline":true}],"defaultGroupId":"default","siteId":"testsite"} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/index/sites/testsite/summaries/page-0000.json b/plans/tools/compose-fixture-one-youtube-channel/index/sites/testsite/summaries/page-0000.json @@ -0,0 +1 @@ +[{"slug":"test-youtube/20240101_test1234567","id":"20240101_test1234567","channelSlug":"test-youtube","title":"Synthetic Test Video","uploadDate":"20240101","date":"2024-01-01","duration":"2:00","channel":"Test YouTube Channel","isLivestream":false,"ageRestricted":false,"isDeleted":false,"isUnlisted":false,"platform":"youtube","webpageUrl":"https://www.youtube.com/watch?v=20240101_test1234567"}] +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/_headers b/plans/tools/compose-fixture-one-youtube-channel/public/_headers @@ -0,0 +1,25 @@ +# Generated by compose-site.ts — do not edit by hand. +/site.json + Access-Control-Allow-Origin: * +/search-aliases.json + Access-Control-Allow-Origin: * +/summaries/* + Access-Control-Allow-Origin: * +/subs/* + Access-Control-Allow-Origin: * +/transcripts/* + Access-Control-Allow-Origin: * +/posts/* + Access-Control-Allow-Origin: * +/stats/* + Access-Control-Allow-Origin: * +/archives/* + Access-Control-Allow-Origin: * +/corpus.json + Access-Control-Allow-Origin: * +/llms.txt + Access-Control-Allow-Origin: * +/robots.txt + Access-Control-Allow-Origin: * +/sitemap.xml + Access-Control-Allow-Origin: * diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/archives/manifest.json b/plans/tools/compose-fixture-one-youtube-channel/public/archives/manifest.json @@ -0,0 +1,13 @@ +{ + "version": 1, + "thresholdBytes": 26214400, + "entries": [ + { + "kind": "transcripts", + "scope": "test-youtube", + "filename": "test-youtube.zip", + "bytes": 1010, + "videoCount": 1 + } + ] +} diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/archives/test-youtube.zip b/plans/tools/compose-fixture-one-youtube-channel/public/archives/test-youtube.zip Binary files differ. diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/corpus.json b/plans/tools/compose-fixture-one-youtube-channel/public/corpus.json @@ -0,0 +1 @@ +{"spec":3,"kind":"site","generatedAt":"2026-09-12T06:01:16.115Z","generator":"Archilyzer (https://archilyzer.pages.dev)","site":{"id":"testsite","title":"Test Site","description":"A fixture archive composed from one YouTube channel.","url":"https://testsite.example"},"totals":{"channels":1,"videos":1},"channels":[{"slug":"test-youtube","name":"Test YouTube Channel","videoCount":1,"groupId":"default","manifests":{"transcripts":"https://testsite.example/transcripts/test-youtube/manifest.json","subs":"https://testsite.example/subs/test-youtube/manifest.json"}}],"shardScheme":{"description":"Transcripts are served as paginated JSON shards — there is no per-video file. To read one video's transcript: (1) GET the channel's transcripts manifest; (2) look up the video id in its `slugToPage` map to get a page number N; (3) GET page-<NNNN>.json (N zero-padded to 4 digits) and take the record whose `id` matches.","transcriptsManifest":"<channel.manifests.transcripts> -> { pageCount, slugToPage: { <videoId>: <pageNumber> } }","transcriptPage":"/transcripts/<slug>/page-<NNNN>.json -> array of { id, title, uploadDate, duration, channel, description, tags, webpageUrl, platform, cues: [{ start, end, text }] }","subsManifest":"<channel.manifests.subs> -> lighter list-view records under the same slugToPage scheme","summariesIndex":"/summaries/manifest.json + /summaries/page-<NNNN>.json -> cross-channel browse index","pageNumberFormat":"zero-padded to 4 digits, e.g. page 0 -> page-0000.json"},"useWithAi":"https://testsite.example/use-with-ai","bulkArchives":{"manifest":"https://testsite.example/archives/manifest.json","note":"Whole-channel transcript and live-chat archives for offline bulk ingestion."}} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/digests/manifest.json b/plans/tools/compose-fixture-one-youtube-channel/public/digests/manifest.json @@ -0,0 +1 @@ +{"version":1,"channels":[],"totalCount":0,"generatedAt":"2026-09-12T06:01:16.116Z","siteId":"testsite"} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/llms.txt b/plans/tools/compose-fixture-one-youtube-channel/public/llms.txt @@ -0,0 +1,15 @@ +# Test Site + +> A fixture archive composed from one YouTube channel. A machine-navigable archive of 1 transcripts across 1 channel(s). See corpus.json for the shard API. + +## Ask AI +- [Use with AI](https://testsite.example/use-with-ai): in-browser chat (bring your own API key) and MCP-server setup for Claude Code, Cursor, and other tools. + +## Corpus +- [corpus.json](https://testsite.example/corpus.json): machine-readable index — channels and how to fetch any transcript from the paginated JSON shards. +- [Bulk archives](https://testsite.example/downloads): whole-channel transcript and live-chat zips for offline ingestion. + +## Channels +- Test YouTube Channel — 1 transcripts: https://testsite.example/transcripts/test-youtube/manifest.json + +Generated by Archilyzer (https://archilyzer.pages.dev) diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/posts/manifest.json b/plans/tools/compose-fixture-one-youtube-channel/public/posts/manifest.json @@ -0,0 +1 @@ +{"version":1,"channels":[],"totalCount":0,"generatedAt":"2026-09-12T06:01:16.115Z","siteId":"testsite"} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/robots.txt b/plans/tools/compose-fixture-one-youtube-channel/public/robots.txt @@ -0,0 +1,5 @@ +User-agent: * +Allow: / + +# AI / LLM corpus index: /llms.txt and /corpus.json +Sitemap: https://testsite.example/sitemap.xml diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/search-aliases.json b/plans/tools/compose-fixture-one-youtube-channel/public/search-aliases.json @@ -0,0 +1 @@ +{"aliases":[{"id":"loli","label":"loli","triggers":["loli","lolly","loly","lolli"],"suggestion":"\\blol(i|ly)","useRegex":true,"note":"AI transcription often expands this to “lolly”/“loly”."},{"id":"youtube","label":"YouTube","triggers":["youtube","yt"],"suggestion":"\\byou ?tube\\b|\\byt\\b","useRegex":true,"note":"Catches “you tube” and the “yt” abbreviation."}]} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/site.json b/plans/tools/compose-fixture-one-youtube-channel/public/site.json @@ -0,0 +1 @@ +{"contract":1,"siteId":"testsite","siteTitle":"Test Site","siteDescription":"A fixture archive composed from one YouTube channel.","headerTitle":"Test Site","homeTagline":"Fixture corpus","siteUrl":"https://testsite.example","pwa":false,"socialLinks":[],"groups":[{"id":"default","name":"All channels","selectedByDefault":true,"inline":true}],"defaultGroupId":"default","channels":[{"slug":"test-youtube","name":"Test YouTube Channel","count":1,"groupId":"default"}],"generatedAt":"2026-09-12T06:01:16.115Z","summariesVersion":3} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/sitemap.xml b/plans/tools/compose-fixture-one-youtube-channel/public/sitemap.xml @@ -0,0 +1,7 @@ +<?xml version="1.0" encoding="UTF-8"?> +<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"> + <url><loc>https://testsite.example/</loc></url> + <url><loc>https://testsite.example/use-with-ai</loc></url> + <url><loc>https://testsite.example/changelog</loc></url> + <url><loc>https://testsite.example/downloads</loc></url> +</urlset> diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/subs/manifest.json b/plans/tools/compose-fixture-one-youtube-channel/public/subs/manifest.json @@ -0,0 +1 @@ +{"version":4,"channels":[],"totalCount":0,"liveChatTotalCount":0,"generatedAt":"2026-09-12T06:01:16.115Z","groups":[{"id":"default","name":"All channels","selectedByDefault":true,"inline":true}],"defaultGroupId":"default","siteId":"testsite"} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/summaries/manifest.json b/plans/tools/compose-fixture-one-youtube-channel/public/summaries/manifest.json @@ -0,0 +1 @@ +{"version":3,"totalCount":1,"pageSize":1000,"pageCount":1,"generatedAt":"2026-09-12T06:01:16.115Z","channels":[{"slug":"test-youtube","name":"Test YouTube Channel","count":1,"groupId":"default"}],"groups":[{"id":"default","name":"All channels","selectedByDefault":true,"inline":true}],"defaultGroupId":"default","siteId":"testsite"} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/summaries/page-0000.json b/plans/tools/compose-fixture-one-youtube-channel/public/summaries/page-0000.json @@ -0,0 +1 @@ +[{"slug":"test-youtube/20240101_test1234567","id":"20240101_test1234567","channelSlug":"test-youtube","title":"Synthetic Test Video","uploadDate":"20240101","date":"2024-01-01","duration":"2:00","channel":"Test YouTube Channel","isLivestream":false,"ageRestricted":false,"isDeleted":false,"isUnlisted":false,"platform":"youtube","webpageUrl":"https://www.youtube.com/watch?v=20240101_test1234567"}] +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/transcripts/test-youtube/manifest.json b/plans/tools/compose-fixture-one-youtube-channel/public/transcripts/test-youtube/manifest.json @@ -0,0 +1 @@ +{"version":1,"channelSlug":"test-youtube","pageCount":1,"maxPageBytes":8388608,"generatedAt":"2026-09-12T06:01:16.087Z","slugToPage":{"20240101_test1234567":0}} +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/public/transcripts/test-youtube/page-0000.json b/plans/tools/compose-fixture-one-youtube-channel/public/transcripts/test-youtube/page-0000.json @@ -0,0 +1 @@ +[{"slug":"test-youtube/20240101_test1234567","id":"20240101_test1234567","channelSlug":"test-youtube","title":"Synthetic Test Video","uploadDate":"20240101","duration":120,"channel":"Test YouTube Channel","description":"A synthetic test video for e2e fixtures.","tags":[],"isLivestream":false,"ageRestricted":false,"platform":"youtube","webpageUrl":"https://www.youtube.com/watch?v=20240101_test1234567","cues":[]}] +\ No newline at end of file diff --git a/plans/tools/compose-fixture-one-youtube-channel/site.json b/plans/tools/compose-fixture-one-youtube-channel/site.json @@ -0,0 +1,11 @@ +{ + "siteId": "testsite", + "siteTitle": "Test Site", + "siteDescription": "A fixture archive composed from one YouTube channel.", + "headerTitle": "Test Site", + "homeTagline": "Fixture corpus", + "groups": [{ "id": "main", "label": "Main" }], + "defaultGroupId": "main", + "channels": [{ "slug": "test-youtube", "groupId": "main" }], + "siteUrl": "https://testsite.example" +}