Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 8a49d9059dc15d98b21d33903bc16644f4f01da8
parent d0e83fead98d60c2b03de5051bbd7c6c1272f658
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 12 Sep 2026 12:23:59 -0400

merge: one-core/phase-2-s2a — the viewer's archive walks become one reader per origin

S2a of Phase 2, reviewed and the offline-cache defect fixed: the eight
component caches and the viewer's contexts fetch through one RemoteSource per
origin (common/lib/archive/readers.ts, with a browser page-cache budget);
the reader gained raw throwing document reads underneath its tolerant
methods because react-query retries a rejection and never a null; the
offline copy downloads every layer the contract has, split into a
per-channel list and a once-per-origin site list that both service workers
can evict; both workers' shard regexes and eviction prefixes are pinned to
CONTRACT.layers by a test. Gates on the slice tip: tsc clean, common 1101,
mcp 205, scripts 71/1, export build clean, export e2e 172/172, hub 5/5.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/components/SearchDataContext.tsx | 27++++++++-------------------
Mcommon/components/aliasesCache.ts | 22++++++++++++++++++----
Acommon/components/archiveCaches.test.ts | 303+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/components/digestCache.ts | 58++++++++++++++++------------------------------------------
Mcommon/components/duplicatesCache.ts | 22+++++++++++++++-------
Mcommon/components/postsCache.ts | 63++++++++++++++++++++++++++-------------------------------------
Mcommon/components/siteRegistry.ts | 5+++--
Mcommon/components/statsCache.ts | 17+++--------------
Mcommon/components/subsCache.ts | 55++++++++++++++++---------------------------------------
Mcommon/components/summariesCache.ts | 17+++--------------
Mcommon/components/transcriptCache.ts | 64+++++++++++-----------------------------------------------------
Mcommon/lib/archive/contract.test.ts | 92+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/archive/contract.ts | 27++++++++++++++++++++++++++-
Acommon/lib/archive/offlineUrls.test.ts | 178+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archive/offlineUrls.ts | 102+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/archive/reader.ts | 171++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------------------
Acommon/lib/archive/readers.test.ts | 181+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archive/readers.ts | 73+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/app/duplicates/DuplicatesClient.tsx | 12++++--------
Mexport/app/lib/offlineCache.ts | 81+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------------------
Mexport/service-worker/site-sw.js | 82++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------
Mexport/service-worker/sw-hub.js | 70+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-----
Mplans/one-core-phase-2.md | 230+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
23 files changed, 1634 insertions(+), 318 deletions(-)

diff --git a/common/components/SearchDataContext.tsx b/common/components/SearchDataContext.tsx @@ -20,11 +20,11 @@ import { useSubsManifest } from "./subsCache"; import { usePostsManifest } from "./postsCache"; import { useSearchAliases, fetchAliases } from "./aliasesCache"; import { mergeAliases, type SearchAlias } from "../lib/searchAliases"; -import { idBaseUrl, makeId } from "./originId"; +import { makeId } from "./originId"; import type { DisplaySummary } from "../lib/transcripts"; import type { Manifest, SubsManifest } from "../lib/manifest"; import type { PostsManifest } from "../lib/posts"; -import { pageFileName } from "../lib/manifest"; +import { readerFor } from "../lib/archive/readers"; import { DEFAULT_GROUP_FALLBACK_ID, FALLBACK_GROUP, @@ -158,12 +158,6 @@ export function SingleSiteDataProvider({ children }: { children: ReactNode }) { ); } -async function fetchJson<T>(url: string): Promise<T> { - const r = await fetch(url); - if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); - return (await r.json()) as T; -} - // A federated origin the hub reads from. `origin` is "" for the hub's own // same-origin pool, else a full origin ("https://x.com"). `siteTitle` labels // its channel group; `accent` is carried for provenance in the UI. @@ -194,8 +188,7 @@ export function MultiSiteDataProvider({ const manifestQueries = useQueries({ queries: sites.map((s) => ({ queryKey: ["manifest", s.origin], - queryFn: () => - fetchJson<Manifest>(`${idBaseUrl(s.origin)}/summaries/manifest.json`), + queryFn: () => readerFor(s.origin).readSummariesManifest(), })), }); const manifestsSettled = manifestQueries.every( @@ -218,10 +211,7 @@ export function MultiSiteDataProvider({ const pageQueries = useQueries({ queries: pageDescriptors.map((d) => ({ queryKey: ["summaries-page", d.origin, d.index], - queryFn: () => - fetchJson<DisplaySummary[]>( - `${idBaseUrl(d.origin)}/summaries/${pageFileName(d.index)}`, - ), + queryFn: () => readerFor(d.origin).readSummariesPage(d.index), })), }); @@ -300,8 +290,7 @@ export function MultiSiteDataProvider({ const subsQueries = useQueries({ queries: sites.map((s) => ({ queryKey: ["subs-manifest", s.origin], - queryFn: () => - fetchJson<SubsManifest>(`${idBaseUrl(s.origin)}/subs/manifest.json`), + queryFn: () => readerFor(s.origin).readSubsSiteManifest(), })), }); const subsSettled = subsQueries.every((q) => q.isSuccess || q.isError); @@ -339,9 +328,9 @@ export function MultiSiteDataProvider({ queries: sites.map((s) => ({ queryKey: ["posts-manifest", s.origin], queryFn: () => - fetchJson<PostsManifest>( - `${idBaseUrl(s.origin)}/posts/manifest.json`, - ).catch( + readerFor(s.origin) + .readPostsSiteManifest() + .catch( (): PostsManifest => ({ version: 0, channels: [], diff --git a/common/components/aliasesCache.ts b/common/components/aliasesCache.ts @@ -7,15 +7,29 @@ // purely additive, so their absence must never break search. import { useQuery } from "@tanstack/react-query"; -import { idBaseUrl } from "./originId"; +import { ArchiveHttpError } from "../lib/archive/reader"; +import { readerFor } from "../lib/archive/readers"; import { coerceAliasConfig, type SearchAlias } from "../lib/searchAliases"; const EMPTY: SearchAlias[] = []; +// ANY status the archive itself returned resolves empty — a 404 because the +// site ships no dictionary, but a 500 or a 403 the same way. That is the exact +// behaviour of the `if (!r.ok) return []` this replaced, kept deliberately: +// alias suggestions are additive, so a server that answers badly should cost +// the chip, not the search. +// +// A TRANSPORT failure is different and is rethrown, because it is not an answer +// at all. react-query then retries it rather than caching "this site has no +// aliases" for the session — the query below has staleTime: Infinity, so a +// swallowed blip would be permanent. export async function fetchAliases(origin = ""): Promise<SearchAlias[]> { - const r = await fetch(`${idBaseUrl(origin)}/search-aliases.json`); - if (!r.ok) return []; - return coerceAliasConfig(await r.json()).aliases; + try { + return coerceAliasConfig(await readerFor(origin).readAliasConfig()).aliases; + } catch (err) { + if (err instanceof ArchiveHttpError) return []; + throw err; + } } export function useSearchAliases(origin = ""): SearchAlias[] { diff --git a/common/components/archiveCaches.test.ts b/common/components/archiveCaches.test.ts @@ -0,0 +1,303 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { resetReaders } from "../lib/archive/readers"; +import { fetchTranscript } from "./transcriptCache"; +import { fetchSubs } from "./subsCache"; +import { fetchPost, fetchThread, peekPost } from "./postsCache"; +import { fetchDigest, hasDigest } from "./digestCache"; +import { fetchDuplicates, fetchDuplicateReport } from "./duplicatesCache"; +import { fetchAliases } from "./aliasesCache"; + +// ─── the eight caches as WRAPPERS ─── +// +// Each of these used to hand-roll the manifest -> slugToPage -> page walk; each +// now calls the shared ArchiveReader. What must survive that move is the memo +// behaviour, because it is what makes the viewer usable: a hit returns the same +// promise (so N components asking for the same video do not each re-render on a +// different tick) and a miss costs ONE read of each document, not one per +// caller and not one per record. +// +// So every test below counts fetches. The caches hold module-level state with +// no reset hook — that is deliberate, it is a page-lifetime cache — so each +// test uses its own channel slug and its own origin rather than trying to wipe +// them. + +type Archive = { fetches: string[]; body: Map<string, unknown> }; + +function installFetch(entries: Record<string, unknown>): { + archive: Archive; + restore: () => void; +} { + const archive: Archive = { + fetches: [], + body: new Map(Object.entries(entries)), + }; + const real = globalThis.fetch; + globalThis.fetch = (async (input: RequestInfo | URL) => { + const url = String(input); + archive.fetches.push(url); + if (!archive.body.has(url)) { + return new Response("not found", { status: 404, statusText: "Not Found" }); + } + return new Response(JSON.stringify(archive.body.get(url)), { status: 200 }); + }) as typeof fetch; + return { + archive, + restore: () => { + globalThis.fetch = real; + resetReaders(); + }, + }; +} + +function count(archive: Archive, url: string): number { + return archive.fetches.filter((u) => u === url).length; +} + +test("transcriptCache: one manifest + one page, and the whole page is warmed", async () => { + const { archive, restore } = installFetch({ + "/transcripts/tc/manifest.json": { + version: 1, + slugToPage: { a: 0, b: 0 }, + pageCount: 1, + generatedAt: "", + }, + "/transcripts/tc/page-0000.json": [ + { slug: "tc/a", id: "a", title: "A", cues: [] }, + { slug: "tc/b", id: "b", title: "B", cues: [] }, + ], + }); + try { + // Two concurrent callers for the same id share one promise — not two reads + // that happen to agree. + const p1 = fetchTranscript("tc/a"); + const p2 = fetchTranscript("tc/a"); + assert.equal(p1, p2); + assert.equal((await p1).title, "A"); + + // A hit resolves without touching the network at all. + const after = archive.fetches.length; + assert.equal((await fetchTranscript("tc/a")).title, "A"); + assert.equal(archive.fetches.length, after); + + // And the OTHER record of the page we already paid for is a hit too: the + // opportunistic warming is what makes a search over one channel cheap. + assert.equal((await fetchTranscript("tc/b")).title, "B"); + assert.equal(count(archive, "/transcripts/tc/manifest.json"), 1); + assert.equal(count(archive, "/transcripts/tc/page-0000.json"), 1); + assert.equal(archive.fetches.length, 2); + } finally { + restore(); + } +}); + +test("transcriptCache: a video the manifest does not list never costs a page read", async () => { + const { archive, restore } = installFetch({ + "/transcripts/tcmiss/manifest.json": { + version: 1, + slugToPage: {}, + pageCount: 0, + generatedAt: "", + }, + }); + try { + await assert.rejects(() => fetchTranscript("tcmiss/nope"), /Unknown transcript slug/); + assert.deepEqual(archive.fetches, ["/transcripts/tcmiss/manifest.json"]); + } finally { + restore(); + } +}); + +test("subsCache: same walk, same sharing, over the subs tree", async () => { + const { archive, restore } = installFetch({ + "/subs/sc/manifest.json": { + version: 1, + slugToPage: { a: 0, b: 0 }, + pageCount: 1, + generatedAt: "", + }, + "/subs/sc/page-0000.json": [ + { slug: "sc/a", id: "a", tracks: {} }, + { slug: "sc/b", id: "b", tracks: {} }, + ], + }); + try { + const p1 = fetchSubs("sc/a"); + assert.equal(fetchSubs("sc/a"), p1); + assert.equal((await p1).id, "a"); + assert.equal((await fetchSubs("sc/b")).id, "b"); + assert.deepEqual(archive.fetches, [ + "/subs/sc/manifest.json", + "/subs/sc/page-0000.json", + ]); + } finally { + restore(); + } +}); + +test("subsCache: an unreachable channel manifest is NOT memoised as absent", async () => { + const { archive, restore } = installFetch({}); + try { + await assert.rejects(() => fetchSubs("scfail/a")); + await assert.rejects(() => fetchSubs("scfail/a")); + // Twice, because a failure is a failure and not an answer. The reader's own + // tolerant subsManifest() would have cached `null` here and the panel would + // stay empty for the session. + assert.equal(count(archive, "/subs/scfail/manifest.json"), 2); + } finally { + restore(); + } +}); + +test("postsCache: a thread walks every page of the channel exactly once", async () => { + const { archive, restore } = installFetch({ + "/posts/pc/manifest.json": { + version: 1, + slugToPage: { p1: 0, p2: 1 }, + pageCount: 2, + generatedAt: "", + }, + "/posts/pc/page-0000.json": [ + { slug: "pc/p1", id: "p1", threadId: "p1", createdAt: "2026-01-01" }, + ], + "/posts/pc/page-0001.json": [ + { slug: "pc/p2", id: "p2", threadId: "p1", createdAt: "2026-01-02" }, + ], + }); + try { + assert.equal((await fetchPost("pc/p1")).id, "p1"); + // Fetching the post warmed the page it came from, so peekPost is sync. + assert.equal(peekPost("pc/p1")?.id, "p1"); + + const thread = await fetchThread("pc/p1"); + assert.deepEqual( + thread.map((t) => t.id), + ["p1", "p2"], + ); + // A second thread read is free: the page memo is what stops fetchThread + // re-downloading the whole channel per post opened. + await fetchThread("pc/p1"); + assert.equal(count(archive, "/posts/pc/manifest.json"), 1); + assert.equal(count(archive, "/posts/pc/page-0000.json"), 1); + assert.equal(count(archive, "/posts/pc/page-0001.json"), 1); + } finally { + restore(); + } +}); + +test("digestCache: a channel with no digests costs ONE 404, ever", async () => { + const { archive, restore } = installFetch({}); + try { + assert.equal(await hasDigest("dcnone/a"), false); + assert.equal(await fetchDigest("dcnone/a"), null); + assert.equal(await fetchDigest("dcnone/b"), null); + // The sparse layer's whole cost model: absence is an answer and is cached, + // so opening video after video in an undigested channel is free. + assert.equal(count(archive, "/digests/dcnone/manifest.json"), 1); + } finally { + restore(); + } +}); + +test("digestCache: present, page-warmed, and not-in-slugToPage means null not throw", async () => { + const { archive, restore } = installFetch({ + "/digests/dc/manifest.json": { + version: 1, + slugToPage: { a: 0, b: 0 }, + pageCount: 1, + pageHashes: ["h0"], + generatedAt: "", + }, + "/digests/dc/page-0000.json": [ + { slug: "dc/a", id: "a", chapters: [] }, + { slug: "dc/b", id: "b", chapters: [] }, + ], + }); + try { + assert.equal(await hasDigest("dc/a"), true); + assert.equal((await fetchDigest("dc/a"))?.id, "a"); + assert.equal((await fetchDigest("dc/b"))?.id, "b"); + // An undigested video in a digested channel: normal, and not an error. + assert.equal(await hasDigest("dc/c"), false); + assert.equal(await fetchDigest("dc/c"), null); + assert.equal(count(archive, "/digests/dc/manifest.json"), 1); + assert.equal(count(archive, "/digests/dc/page-0000.json"), 1); + } finally { + restore(); + } +}); + +test("duplicatesCache: absent is empty, present is a slug-keyed lookup", async () => { + const absent = installFetch({}); + try { + assert.equal(await fetchDuplicateReport("https://dup-absent.example"), null); + assert.equal((await fetchDuplicates("https://dup-absent.example")).size, 0); + } finally { + absent.restore(); + } + + const O = "https://dup-present.example"; + const present = installFetch({ + [`${O}/duplicates.json`]: { + clusters: [ + { + clusterId: "c1", + videoRefs: [{ slug: "x/1" }, { slug: "y/2" }], + }, + ], + }, + }); + try { + const lookup = await fetchDuplicates(O); + assert.equal(lookup.size, 2); + assert.equal(lookup.get("x/1")?.clusterId, "c1"); + assert.equal(lookup.get("y/2")?.clusterId, "c1"); + assert.equal(count(present.archive, `${O}/duplicates.json`), 1); + } finally { + present.restore(); + } +}); + +test("aliasesCache: a 404 is an empty dictionary; a transport failure is rethrown", async () => { + const missing = installFetch({}); + try { + assert.deepEqual(await fetchAliases("https://alias-404.example"), []); + } finally { + missing.restore(); + } + + const present = installFetch({ + "https://alias-ok.example/search-aliases.json": { + aliases: [ + { + id: "kcup", + label: "k cups", + triggers: ["k cups"], + suggestion: "(k|cake)[ -]?cup", + useRegex: true, + }, + ], + }, + }); + try { + const aliases = await fetchAliases("https://alias-ok.example"); + assert.equal(aliases.length, 1); + assert.equal(aliases[0].id, "kcup"); + } finally { + present.restore(); + } + + // The distinction that matters: the query holding this has staleTime + // Infinity, so swallowing a blip would mean "this site has no aliases" for + // the rest of the session. + const real = globalThis.fetch; + globalThis.fetch = (async () => { + throw new TypeError("network down"); + }) as typeof fetch; + try { + await assert.rejects(() => fetchAliases("https://alias-down.example"), /network down/); + } finally { + globalThis.fetch = real; + resetReaders(); + } +}); diff --git a/common/components/digestCache.ts b/common/components/digestCache.ts @@ -1,9 +1,11 @@ "use client"; import type { ChannelDigestsManifest, VideoDigest } from "../lib/digests"; -import { digestPageFileName, manifestHasDigest } from "../lib/digests"; +import { manifestHasDigest } from "../lib/digests"; +import { PromiseMap } from "../lib/archive/reader"; +import { channelRef, readerFor } from "../lib/archive/readers"; import { idbGet, idbPut } from "./digestStore"; -import { makeId, splitId, idBaseUrl } from "./originId"; +import { makeId, splitId } from "./originId"; // Fetch layer for the derived corpus, mirroring transcriptCache.ts: a memory // map, in-flight dedupe, manifest -> page resolution and opportunistic warming @@ -23,11 +25,12 @@ import { makeId, splitId, idBaseUrl } from "./originId"; const resolved = new Map<string, VideoDigest | null>(); const inFlight = new Map<string, Promise<VideoDigest | null>>(); -const channelManifests = new Map< - string, - Promise<ChannelDigestsManifest | null> ->(); -const pagePromises = new Map<string, Promise<VideoDigest[]>>(); +// PromiseMap memoises a RESOLVED value (including `null` — the cached negative +// this layer depends on) and drops a REJECTION, which is exactly the policy +// this file hand-rolled: absence is an answer worth keeping, a transport +// failure is not proof of absence. +const channelManifests = new PromiseMap<ChannelDigestsManifest | null>(); +const pagePromises = new PromiseMap<VideoDigest[]>(); // `id` is an OriginId: a bare "channelSlug/videoId" for same-origin content, or // "origin\tchannelSlug/videoId" for a federated cross-origin video. @@ -87,26 +90,9 @@ function fetchChannelManifest( channelSlug: string, origin: string, ): Promise<ChannelDigestsManifest | null> { - const key = makeId(origin, channelSlug); - let p = channelManifests.get(key); - if (!p) { - p = fetch(`${idBaseUrl(origin)}/digests/${channelSlug}/manifest.json`) - .then((r) => { - if (r.status === 404) return null; - if (!r.ok) { - throw new Error(`Failed to fetch digests manifest for ${channelSlug}`); - } - return r.json() as Promise<ChannelDigestsManifest>; - }) - .catch((err) => { - // A transport failure is not proof of absence, so it must not be - // cached as one — drop the entry and let the next caller retry. - channelManifests.delete(key); - throw err; - }); - channelManifests.set(key, p); - } - return p; + return channelManifests.take(makeId(origin, channelSlug), () => + readerFor(origin).readChannelDigestsManifest(channelSlug), + ); } function fetchPage( @@ -114,21 +100,9 @@ function fetchPage( pageIndex: number, origin: string, ): Promise<VideoDigest[]> { - const key = `${makeId(origin, channelSlug)}:${pageIndex}`; - let p = pagePromises.get(key); - if (!p) { - p = fetch( - `${idBaseUrl(origin)}/digests/${channelSlug}/${digestPageFileName(pageIndex)}`, - ).then((r) => { - if (!r.ok) { - throw new Error(`Failed to fetch digest page ${channelSlug}/${pageIndex}`); - } - return r.json() as Promise<VideoDigest[]>; - }); - p.catch(() => pagePromises.delete(key)); - pagePromises.set(key, p); - } - return p; + return pagePromises.take(`${makeId(origin, channelSlug)}:${pageIndex}`, () => + readerFor(origin).digestPage(channelRef(channelSlug, origin), pageIndex), + ); } async function load(id: string): Promise<VideoDigest | null> { diff --git a/common/components/duplicatesCache.ts b/common/components/duplicatesCache.ts @@ -16,7 +16,7 @@ // per member is `aligned` — see duplicateSiblings below. import { useQuery } from "@tanstack/react-query"; -import { idBaseUrl } from "./originId"; +import { readerFor } from "../lib/archive/readers"; import type { DuplicateCluster, DuplicateReport, @@ -27,15 +27,23 @@ export type DuplicateLookup = ReadonlyMap<string, DuplicateCluster>; const EMPTY: DuplicateLookup = new Map(); -export async function fetchDuplicates(origin = ""): Promise<DuplicateLookup> { - let report: DuplicateReport | null = null; +// The raw shipped report, or null when the site ships none (the common case — +// compose-site writes the file only for a site with at least one publishable +// cluster). Absent, unreachable and unparseable all fold to null: every +// duplicate affordance is additive, so none of them may break a page. +export async function fetchDuplicateReport( + origin = "", +): Promise<DuplicateReport | null> { try { - const r = await fetch(`${idBaseUrl(origin)}/duplicates.json`); - if (!r.ok) return EMPTY; - report = (await r.json()) as DuplicateReport; + return await readerFor(origin).readDuplicates(); } catch { - return EMPTY; + return null; } +} + +export async function fetchDuplicates(origin = ""): Promise<DuplicateLookup> { + const report = await fetchDuplicateReport(origin); + if (!report) return EMPTY; const out = new Map<string, DuplicateCluster>(); for (const cluster of report?.clusters ?? []) { for (const ref of cluster?.videoRefs ?? []) { diff --git a/common/components/postsCache.ts b/common/components/postsCache.ts @@ -5,24 +5,21 @@ // federating hub can hold several origins' posts at once without collisions. import { useQueries, useQuery } from "@tanstack/react-query"; -import { - postsPageFileName, - type ChannelPostsManifest, - type Post, - type PostsManifest, -} from "../lib/posts"; -import { makeId, splitId, idBaseUrl } from "./originId"; - -async function fetchJson<T>(url: string): Promise<T> { - const r = await fetch(url); - if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); - return (await r.json()) as T; -} +import type { ChannelPostsManifest, Post, PostsManifest } from "../lib/posts"; +import { PromiseMap } from "../lib/archive/reader"; +import { channelRef, readerFor } from "../lib/archive/readers"; +import { makeId, splitId } from "./originId"; const resolved = new Map<string, Post>(); const inFlight = new Map<string, Promise<Post>>(); -const channelManifests = new Map<string, Promise<ChannelPostsManifest>>(); -const pagePromises = new Map<string, Promise<Post[]>>(); +// Two memos the reader does NOT provide, for two different reasons. The +// manifest one wants a THROW where the reader's tolerant `postsManifest` wants +// a null (see subsCache). The page one exists because the reader caches subs +// and transcript pages but not posts pages, and fetchThread below walks EVERY +// page of a channel — without a memo, opening two posts in a thread would +// re-download the whole channel. +const channelManifests = new PromiseMap<ChannelPostsManifest>(); +const pagePromises = new PromiseMap<Post[]>(); export type PostsManifestRef = { channelSlug: string; origin?: string }; @@ -53,16 +50,9 @@ export function fetchChannelPostsManifest( channelSlug: string, origin = "", ): Promise<ChannelPostsManifest> { - const key = makeId(origin, channelSlug); - let p = channelManifests.get(key); - if (!p) { - p = fetchJson<ChannelPostsManifest>( - `${idBaseUrl(origin)}/posts/${channelSlug}/manifest.json`, - ); - p.catch(() => channelManifests.delete(key)); - channelManifests.set(key, p); - } - return p; + return channelManifests.take(makeId(origin, channelSlug), () => + readerFor(origin).readChannelPostsManifest(channelSlug), + ); } export function fetchPostsPage( @@ -70,23 +60,21 @@ export function fetchPostsPage( pageIndex: number, origin = "", ): Promise<Post[]> { - const key = `${makeId(origin, channelSlug)}:${pageIndex}`; - let p = pagePromises.get(key); - if (!p) { - p = fetchJson<Post[]>( - `${idBaseUrl(origin)}/posts/${channelSlug}/${postsPageFileName(pageIndex)}`, - ).then((page) => { + return pagePromises.take( + `${makeId(origin, channelSlug)}:${pageIndex}`, + async () => { + const page = await readerFor(origin).postsPage( + channelRef(channelSlug, origin), + pageIndex, + ); // Warm the by-id cache so a later fetchPost() for any post on this page // is synchronous. for (const entry of page) { resolved.set(makeId(origin, entry.slug), entry); } return page; - }); - p.catch(() => pagePromises.delete(key)); - pagePromises.set(key, p); - } - return p; + }, + ); } async function load(id: string): Promise<Post> { @@ -136,7 +124,8 @@ export function usePostsManifest(origin = "") { return useQuery<PostsManifest>({ queryKey: ["posts-manifest", origin], queryFn: () => - fetchJson<PostsManifest>(`${idBaseUrl(origin)}/posts/manifest.json`) + readerFor(origin) + .readPostsSiteManifest() // A site with no social channels ships no posts manifest; treat that as // an empty corpus rather than an error, so the UI degrades quietly. .catch( diff --git a/common/components/siteRegistry.ts b/common/components/siteRegistry.ts @@ -19,6 +19,7 @@ // and stay in sync when a site is added or removed. import { useEffect, useSyncExternalStore } from "react"; +import { manifestUrl, rootFileUrl } from "../lib/archive/contract"; import { SITE_DESCRIPTOR_VERSION } from "../lib/siteDescriptor"; import type { PublicSiteDescriptor } from "../lib/siteDescriptor"; @@ -259,7 +260,7 @@ export async function validateSite(input: string): Promise<ValidateResult> { let descriptor: PublicSiteDescriptor; try { - const res = await fetch(`${origin}/site.json`); + const res = await fetch(rootFileUrl("site.json", origin)); if (!res.ok) return { ok: false, reason: UNREADABLE }; descriptor = (await res.json()) as PublicSiteDescriptor; } catch { @@ -284,7 +285,7 @@ export async function validateSite(input: string): Promise<ValidateResult> { // whose descriptor is fine but whose data isn't reachable/valid would fail // silently later, so reject it now with a clear message. try { - const res = await fetch(`${origin}/summaries/manifest.json`); + const res = await fetch(manifestUrl("summaries", undefined, origin)); if (!res.ok) return { ok: false, reason: UNREADABLE }; const manifest = (await res.json()) as { version?: unknown; channels?: unknown }; if ( diff --git a/common/components/statsCache.ts b/common/components/statsCache.ts @@ -3,14 +3,7 @@ import { useMemo } from "react"; import { useQueries, useQuery } from "@tanstack/react-query"; import type { StatsManifest, VideoStat } from "../lib/stats"; -import { statsPageFileName } from "../lib/stats"; -import { idBaseUrl } from "./originId"; - -async function fetchJson<T>(url: string): Promise<T> { - const r = await fetch(url); - if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); - return (await r.json()) as T; -} +import { readerFor } from "../lib/archive/readers"; export type StatsState = { manifest: StatsManifest | null; @@ -24,8 +17,7 @@ export type StatsState = { export function useStatsManifest(origin = "") { return useQuery<StatsManifest>({ queryKey: ["stats-manifest", origin], - queryFn: () => - fetchJson<StatsManifest>(`${idBaseUrl(origin)}/stats/manifest.json`), + queryFn: () => readerFor(origin).readStatsManifest(), }); } @@ -39,10 +31,7 @@ export function useStats(origin = ""): StatsState { const pageQueries = useQueries({ queries: Array.from({ length: pageCount }, (_, i) => ({ queryKey: ["stats-page", origin, i], - queryFn: () => - fetchJson<VideoStat[]>( - `${idBaseUrl(origin)}/stats/${statsPageFileName(i)}`, - ), + queryFn: () => readerFor(origin).readStatsPage(i), enabled: pageCount > 0, })), }); diff --git a/common/components/subsCache.ts b/common/components/subsCache.ts @@ -3,19 +3,17 @@ import { useQueries, useQuery } from "@tanstack/react-query"; import type { SubsDetail } from "../lib/subs"; import type { ChannelSubsManifest, SubsManifest } from "../lib/manifest"; -import { pageFileName } from "../lib/manifest"; -import { makeId, splitId, idBaseUrl } from "./originId"; - -async function fetchJson<T>(url: string): Promise<T> { - const r = await fetch(url); - if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); - return (await r.json()) as T; -} +import { PromiseMap } from "../lib/archive/reader"; +import { channelRef, readerFor } from "../lib/archive/readers"; +import { makeId, splitId } from "./originId"; const resolved = new Map<string, SubsDetail>(); const inFlight = new Map<string, Promise<SubsDetail>>(); -const channelManifests = new Map<string, Promise<ChannelSubsManifest>>(); -const pagePromises = new Map<string, Promise<SubsDetail[]>>(); +// The reader has a per-channel subs-manifest cache too, but its answer for an +// unreachable manifest is `null` (the right answer for a scanner probing thirty +// channels, the wrong one for a UI that wants react-query to retry). So the +// memo with the THROWING policy stays here, over the reader's raw read. +const channelManifests = new PromiseMap<ChannelSubsManifest>(); // A per-channel subs reference: channel slug + the origin it lives on // ("" = same-origin). Used by the multi-origin manifest hook. @@ -42,33 +40,9 @@ function fetchChannelSubsManifest( channelSlug: string, origin = "", ): Promise<ChannelSubsManifest> { - const key = makeId(origin, channelSlug); - let p = channelManifests.get(key); - if (!p) { - p = fetchJson<ChannelSubsManifest>( - `${idBaseUrl(origin)}/subs/${channelSlug}/manifest.json`, - ); - p.catch(() => channelManifests.delete(key)); - channelManifests.set(key, p); - } - return p; -} - -function fetchPage( - channelSlug: string, - pageIndex: number, - origin = "", -): Promise<SubsDetail[]> { - const key = `${makeId(origin, channelSlug)}:${pageIndex}`; - let p = pagePromises.get(key); - if (!p) { - p = fetchJson<SubsDetail[]>( - `${idBaseUrl(origin)}/subs/${channelSlug}/${pageFileName(pageIndex)}`, - ); - p.catch(() => pagePromises.delete(key)); - pagePromises.set(key, p); - } - return p; + return channelManifests.take(makeId(origin, channelSlug), () => + readerFor(origin).readChannelSubsManifest(channelSlug), + ); } async function load(id: string): Promise<SubsDetail> { @@ -81,7 +55,10 @@ async function load(id: string): Promise<SubsDetail> { const pageIndex = manifest.slugToPage[videoId]; if (pageIndex === undefined) throw new Error(`Unknown subs slug: ${slug}`); - const page = await fetchPage(channelSlug, pageIndex, origin); + const page = await readerFor(origin).subsPage( + channelRef(channelSlug, origin), + pageIndex, + ); let found: SubsDetail | undefined; for (const entry of page) { const entryId = makeId(origin, entry.slug); @@ -95,7 +72,7 @@ async function load(id: string): Promise<SubsDetail> { export function useSubsManifest(origin = "") { return useQuery<SubsManifest>({ queryKey: ["subs-manifest", origin], - queryFn: () => fetchJson<SubsManifest>(`${idBaseUrl(origin)}/subs/manifest.json`), + queryFn: () => readerFor(origin).readSubsSiteManifest(), }); } diff --git a/common/components/summariesCache.ts b/common/components/summariesCache.ts @@ -4,14 +4,7 @@ import { useMemo } from "react"; import { useQueries, useQuery } from "@tanstack/react-query"; import type { DisplaySummary } from "../lib/transcripts"; import type { Manifest } from "../lib/manifest"; -import { pageFileName } from "../lib/manifest"; -import { idBaseUrl } from "./originId"; - -async function fetchJson<T>(url: string): Promise<T> { - const r = await fetch(url); - if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); - return (await r.json()) as T; -} +import { readerFor } from "../lib/archive/readers"; export type SummariesState = { manifest: Manifest | null; @@ -25,8 +18,7 @@ export type SummariesState = { export function useManifest(origin = "") { return useQuery<Manifest>({ queryKey: ["manifest", origin], - queryFn: () => - fetchJson<Manifest>(`${idBaseUrl(origin)}/summaries/manifest.json`), + queryFn: () => readerFor(origin).readSummariesManifest(), }); } @@ -38,10 +30,7 @@ export function useSummaries(origin = ""): SummariesState { const pageQueries = useQueries({ queries: Array.from({ length: pageCount }, (_, i) => ({ queryKey: ["summaries-page", origin, i], - queryFn: () => - fetchJson<DisplaySummary[]>( - `${idBaseUrl(origin)}/summaries/${pageFileName(i)}`, - ), + queryFn: () => readerFor(origin).readSummariesPage(i), enabled: pageCount > 0, })), }); diff --git a/common/components/transcriptCache.ts b/common/components/transcriptCache.ts @@ -1,18 +1,17 @@ "use client"; import type { TranscriptDetail } from "../lib/transcripts"; -import type { ChannelTranscriptsManifest } from "../lib/manifest"; -import { pageFileName } from "../lib/manifest"; +import { channelRef, readerFor } from "../lib/archive/readers"; import { idbGet, idbPutBatch } from "./transcriptStore"; -import { makeId, splitId, idBaseUrl } from "./originId"; +import { makeId, splitId } from "./originId"; +// The manifest -> slugToPage -> page walk is the ArchiveReader's +// (lib/archive/reader.ts), including the per-channel manifest PromiseMap and +// the byte-budgeted page LRU. What stays here is what is genuinely the +// viewer's: the per-id memo, the in-flight dedupe, and warming IndexedDB with +// every record of a page we already paid for. const resolved = new Map<string, TranscriptDetail>(); const inFlight = new Map<string, Promise<TranscriptDetail>>(); -const channelManifests = new Map< - string, - Promise<ChannelTranscriptsManifest> ->(); -const pagePromises = new Map<string, Promise<TranscriptDetail[]>>(); // `id` is an OriginId: a bare "channelSlug/videoId" for same-origin content, or // "origin\tchannelSlug/videoId" for a federated cross-origin video. Maps and @@ -35,49 +34,6 @@ export function fetchTranscript(id: string): Promise<TranscriptDetail> { return p; } -function fetchChannelManifest( - channelSlug: string, - origin: string, -): Promise<ChannelTranscriptsManifest> { - const key = makeId(origin, channelSlug); - let p = channelManifests.get(key); - if (!p) { - p = fetch(`${idBaseUrl(origin)}/transcripts/${channelSlug}/manifest.json`).then((r) => { - if (!r.ok) - throw new Error( - `Failed to fetch transcripts manifest for ${channelSlug}`, - ); - return r.json() as Promise<ChannelTranscriptsManifest>; - }); - p.catch(() => channelManifests.delete(key)); - channelManifests.set(key, p); - } - return p; -} - -function fetchPage( - channelSlug: string, - pageIndex: number, - origin: string, -): Promise<TranscriptDetail[]> { - const key = `${makeId(origin, channelSlug)}:${pageIndex}`; - let p = pagePromises.get(key); - if (!p) { - p = fetch( - `${idBaseUrl(origin)}/transcripts/${channelSlug}/${pageFileName(pageIndex)}`, - ).then((r) => { - if (!r.ok) - throw new Error( - `Failed to fetch transcript page ${channelSlug}/${pageIndex}`, - ); - return r.json() as Promise<TranscriptDetail[]>; - }); - p.catch(() => pagePromises.delete(key)); - pagePromises.set(key, p); - } - return p; -} - async function load(id: string): Promise<TranscriptDetail> { const stored = await idbGet(id); if (stored) return stored; @@ -86,11 +42,13 @@ async function load(id: string): Promise<TranscriptDetail> { if (slashIdx < 0) throw new Error(`Malformed transcript slug: ${slug}`); const channelSlug = slug.slice(0, slashIdx); const videoId = slug.slice(slashIdx + 1); - const manifest = await fetchChannelManifest(channelSlug, origin); + const reader = readerFor(origin); + const ch = channelRef(channelSlug, origin); + const manifest = await reader.transcriptsManifest(ch); const pageIndex = manifest.slugToPage[videoId]; if (pageIndex === undefined) throw new Error(`Unknown transcript slug: ${slug}`); - const page = await fetchPage(channelSlug, pageIndex, origin); + const page = await reader.transcriptPage(ch, pageIndex); let found: TranscriptDetail | undefined; for (const entry of page) { // Re-key warmed entries by their OriginId so a cross-origin diff --git a/common/lib/archive/contract.test.ts b/common/lib/archive/contract.test.ts @@ -1,7 +1,12 @@ import { test } from "node:test"; import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; import { + ARCHIVE_TREES, CONTRACT, + PER_CHANNEL_TREES, ROOT_FILES, archiveUrl, corpusUrl, @@ -15,6 +20,8 @@ import { import { buildSiteCorpus } from "../corpus"; import type { PublicSiteDescriptor } from "../siteDescriptor"; +const HERE = path.dirname(fileURLToPath(import.meta.url)); + // A minimal descriptor with two channels, one of which ships posts and digests, // so every conditional manifest pointer buildSiteCorpus can emit is exercised. function descriptor(siteUrl?: string): PublicSiteDescriptor { @@ -166,3 +173,88 @@ test("CONTRACT is frozen where it is published", () => { ["transcripts", "subs", "posts", "digests", "summaries"], ); }); + +// ─── The three hand-written copies of the contract, pinned ─── +// +// Three files legitimately cannot import this module and re-spell part of it by +// hand: the two service workers (a service worker has no module graph to import +// through — it is fetched and evaluated standalone) and the search index worker +// (a classic Worker bundle). Each copy is silent when it drifts: a layer the SW +// does not match is simply never cached, and a page file name without the +// padding 404s against a real site while passing every unit test that made up +// its own fixture names. +// +// So the copies are read off disk and compared with the contract. This is the +// guard that lets CONTRACT.layers grow: add a layer, and these fail by name. + +const REPO = path.resolve(HERE, "..", "..", ".."); + +function readSource(rel: string): string { + return readFileSync(path.join(REPO, rel), "utf8"); +} + +// Pull `const <NAME> = /…/;` out of a service worker and return the alternatives +// of its first (…) group, un-escaped. Throws rather than returning [] when the +// constant is missing, so a rename cannot make this test vacuously pass. +function alternatives(src: string, name: string): string[] { + const decl = new RegExp(`const ${name} = /(.+)/;`).exec(src); + assert.ok(decl, `${name} not found — did it move or get renamed?`); + const group = /\(([^)]+)\)/.exec(decl[1]); + assert.ok(group, `${name} has no alternation group`); + return group[1].split("|").map((a) => a.replace(/\\(.)/g, "$1")); +} + +// The `/<tree>/${slug}/` entries of an eviction prefix list. +function evictedTrees(src: string): string[] { + const block = /const prefixes = \[([\s\S]*?)\]/.exec(src); + assert.ok(block, "eviction prefix list not found"); + return [...block[1].matchAll(/`\/([^/]+)\/\$\{slug\}\/`/g)].map((m) => m[1]); +} + +for (const sw of ["export/service-worker/site-sw.js", "export/service-worker/sw-hub.js"]) { + test(`${sw} matches exactly the contract's trees and root files`, () => { + const src = readSource(sw); + + // Per-channel and flat, split the way the URL shapes are split — and + // together exactly ARCHIVE_TREES, so a layer cannot be quietly dropped from + // one family and "found" in the other. + assert.deepEqual( + alternatives(src, "SHARD_RE").sort(), + [...PER_CHANNEL_TREES].sort(), + ); + assert.deepEqual( + alternatives(src, "FLAT_RE").sort(), + ARCHIVE_TREES.filter(isFlatTree).sort(), + ); + assert.deepEqual( + [...alternatives(src, "SHARD_RE"), ...alternatives(src, "FLAT_RE")].sort(), + [...ARCHIVE_TREES].sort(), + ); + + // The root documents, by name. ROOT_RE is a full-path match, so these are + // the file names with no extra path. + assert.deepEqual(alternatives(src, "ROOT_RE").sort(), [...ROOT_FILES].sort()); + + // Eviction is per channel, so it sweeps the per-channel trees and nothing + // else. A flat tree here would be a prefix that can never match. + assert.deepEqual(evictedTrees(src).sort(), [...PER_CHANNEL_TREES].sort()); + }); +} + +test("the search index worker's private pageFileName matches CONTRACT.pagePad", () => { + // components/searchIndex.worker.ts keeps its own copy on purpose — a worker + // bundle cannot import this module. Extract the literal padding it uses and + // run the copy, so both the constant AND the produced name are pinned. + const src = readSource("common/components/searchIndex.worker.ts"); + const fn = /function pageFileName\(index: number\): string \{\s*return `page-\$\{String\(index\)\.padStart\((\d+), "0"\)\}\.json`;\s*\}/.exec( + src, + ); + assert.ok(fn, "searchIndex.worker.ts's pageFileName is not the shape this test pins"); + assert.equal(Number(fn[1]), CONTRACT.pagePad); + + // And the URLs it builds are the contract's, for the one tree it walks. + assert.ok(src.includes("`/transcripts/${slug}/manifest.json`")); + assert.equal(manifestUrl("transcripts", "alpha"), "/transcripts/alpha/manifest.json"); + assert.ok(src.includes("`/transcripts/${slug}/${pageFileName(p)}`")); + assert.equal(pageUrl("transcripts", "alpha", 12), "/transcripts/alpha/page-0012.json"); +}); diff --git a/common/lib/archive/contract.ts b/common/lib/archive/contract.ts @@ -58,6 +58,14 @@ export type ContractLayer = (typeof CONTRACT.layers)[number]; // builders take the wider ArchiveTree and CONTRACT.layers stays frozen. export type ArchiveTree = ContractLayer | "stats"; +// Every served tree, contract layers plus the undocumented `stats`. This is the +// list a cache walker enumerates (the offline cache, the two service workers); +// CONTRACT.layers stays the frozen PUBLISHED list. +export const ARCHIVE_TREES: readonly ArchiveTree[] = [ + ...CONTRACT.layers, + "stats", +]; + // The trees with no per-channel level: one manifest at the tree root and pages // beside it. Everything else is /<tree>/<slug>/…. const FLAT_TREES: ReadonlySet<string> = new Set(["summaries", "stats"]); @@ -66,6 +74,23 @@ export function isFlatTree(tree: ArchiveTree): boolean { return FLAT_TREES.has(tree); } +// The per-channel trees: /<tree>/<slug>/…. What a per-channel cache eviction +// has to sweep, and the half of ARCHIVE_TREES that takes a slug. +export const PER_CHANNEL_TREES: readonly ArchiveTree[] = ARCHIVE_TREES.filter( + (t) => !isFlatTree(t), +); + +// A tree's ROOT manifest, /<tree>/manifest.json. +// +// Every tree ships one, and for a per-channel tree it is a DIFFERENT document +// from the per-channel manifest: /subs/manifest.json is the site-level index of +// which channels ship live chat (SubsManifest), while /subs/<slug>/manifest.json +// is that channel's slugToPage. manifestUrl() below builds the second; this +// builds the first, and for a flat tree the two are the same file. +export function treeManifestUrl(tree: ArchiveTree, base?: string): string { + return archiveUrl(base, `/${tree}/manifest.json`); +} + // THE page-shard file name, for every layer. `page-0.json` is a 404 on every // published archive, so a copy that lost the padding would 404 silently against // a real site and pass every unit test. lib/manifest.ts re-exports this (and @@ -117,7 +142,7 @@ export function manifestUrl( slug?: string, base?: string, ): string { - if (isFlatTree(tree)) return archiveUrl(base, `/${tree}/manifest.json`); + if (isFlatTree(tree)) return treeManifestUrl(tree, base); if (!slug) throw new Error(`manifestUrl(${tree}) needs a channel slug`); return archiveUrl(base, `/${tree}/${slug}/manifest.json`); } diff --git a/common/lib/archive/offlineUrls.test.ts b/common/lib/archive/offlineUrls.test.ts @@ -0,0 +1,178 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { ARCHIVE_TREES, PER_CHANNEL_TREES, ROOT_FILES, isFlatTree } from "./contract"; +import { + channelArchiveUrls, + siteArchiveUrls, + type ManifestReader, +} from "./offlineUrls"; + +// THE SPLIT IS THE ASSERTION. An offline copy is two lists, and if a site-wide +// document ever leaks back into the per-channel one it is not a style problem: +// the per-channel list is paid per channel pinned, and the per-channel EVICT +// sweep (a /<tree>/<slug>/ prefix match in both service workers) cannot remove +// anything that is not under a channel prefix. So a flat-tree URL in the +// channel list is bytes that are downloaded N times and never removable. +// +// Measured on jeralyzer when that was the shipped behaviour: summaries ~14 MB + +// stats ~21.6 MB + duplicates 5.9 MB = ~41.5 MB per channel, re-fetched for +// real (the worker bulk-caches with `cache: "reload"`). + +// A stub archive: manifests present at these URLs with these page counts, +// everything else absent. Records the reads so "absent costs one probe" is +// assertable. +function reader(pageCounts: Record<string, number>): { + read: ManifestReader; + reads: string[]; +} { + const reads: string[] = []; + const read: ManifestReader = async (url) => { + reads.push(url); + return url in pageCounts ? { pageCount: pageCounts[url] } : null; + }; + return { read, reads }; +} + +const FULL = { + "/transcripts/alpha/manifest.json": 2, + "/subs/alpha/manifest.json": 1, + "/posts/alpha/manifest.json": 1, + "/digests/alpha/manifest.json": 1, + "/summaries/manifest.json": 2, + "/stats/manifest.json": 1, +}; + +test("the channel list is the PER-CHANNEL trees and nothing else", async () => { + const { read } = reader(FULL); + const urls = await channelArchiveUrls(read, "", "alpha"); + + assert.deepEqual(urls, [ + "/transcripts/alpha/manifest.json", + "/transcripts/alpha/page-0000.json", + "/transcripts/alpha/page-0001.json", + "/subs/alpha/manifest.json", + "/subs/alpha/page-0000.json", + "/posts/alpha/manifest.json", + "/posts/alpha/page-0000.json", + "/digests/alpha/manifest.json", + "/digests/alpha/page-0000.json", + ]); + + // Stated structurally as well as literally, so adding a layer to the contract + // fails here rather than silently shipping a channel that is half offline. + const trees = new Set(urls.map((u) => u.split("/")[1])); + assert.deepEqual([...trees].sort(), [...PER_CHANNEL_TREES].sort()); + + // Not one site-wide byte. This is the regression the review caught. + for (const u of urls) { + const tree = u.split("/")[1]; + assert.ok( + !ARCHIVE_TREES.some((t) => t === tree && isFlatTree(t)), + `flat tree ${tree} must not be in a per-channel download: ${u}`, + ); + } + for (const file of ROOT_FILES) { + assert.ok(!urls.includes(`/${file}`), `${file} must not be per channel`); + } + // Every URL is under a /<tree>/<slug>/ prefix — the only shape the service + // workers' per-channel evict sweep can ever remove. + for (const u of urls) { + assert.match(u, /^\/[^/]+\/alpha\//, `${u} is not evictable per channel`); + } +}); + +test("the site list is the FLAT trees plus the root files, and nothing per-channel", async () => { + const { read } = reader(FULL); + const urls = await siteArchiveUrls(read, ""); + + assert.deepEqual(urls, [ + "/summaries/manifest.json", + "/summaries/page-0000.json", + "/summaries/page-0001.json", + "/stats/manifest.json", + "/stats/page-0000.json", + "/corpus.json", + "/site.json", + "/search-aliases.json", + "/duplicates.json", + ]); + + const flat = ARCHIVE_TREES.filter(isFlatTree); + for (const tree of flat) { + assert.ok( + urls.some((u) => u.startsWith(`/${tree}/`)), + `${tree} missing from the site list`, + ); + } + for (const file of ROOT_FILES) assert.ok(urls.includes(`/${file}`)); + // No channel slug anywhere: nothing here is paid per channel. + for (const u of urls) assert.doesNotMatch(u, /alpha/); +}); + +test("together the two lists cover every tree the contract defines, once", async () => { + const { read } = reader(FULL); + const all = [ + ...(await channelArchiveUrls(read, "", "alpha")), + ...(await siteArchiveUrls(read, "")), + ]; + assert.equal(new Set(all).size, all.length, "no URL is in both lists"); + + const trees = new Set( + all.filter((u) => u.split("/").length > 2).map((u) => u.split("/")[1]), + ); + assert.deepEqual([...trees].sort(), [...ARCHIVE_TREES].sort()); +}); + +test("a tree the site does not ship costs one probe and contributes nothing", async () => { + // Transcripts only — the shape of a site with no chat, no posts, no digests + // and no stats. + const { read, reads } = reader({ "/transcripts/alpha/manifest.json": 1 }); + + assert.deepEqual(await channelArchiveUrls(read, "", "alpha"), [ + "/transcripts/alpha/manifest.json", + "/transcripts/alpha/page-0000.json", + ]); + // One probe per absent tree, not a retry storm. + assert.deepEqual(reads, [ + "/transcripts/alpha/manifest.json", + "/subs/alpha/manifest.json", + "/posts/alpha/manifest.json", + "/digests/alpha/manifest.json", + ]); + + // The root files are listed unprobed — absent ones are skipped by the worker. + assert.deepEqual(await siteArchiveUrls(read, ""), [ + "/corpus.json", + "/site.json", + "/search-aliases.json", + "/duplicates.json", + ]); +}); + +test("no transcripts manifest means no download at all", async () => { + const { read } = reader({ "/subs/alpha/manifest.json": 1 }); + // A channel whose transcripts are unreachable cannot be opened offline, so + // caching its live chat would be caching a dead end. + assert.deepEqual(await channelArchiveUrls(read, "", "alpha"), []); +}); + +test("a federated origin prefixes both lists and nothing else changes", async () => { + const O = "https://member.example"; + const { read } = reader({ + [`${O}/transcripts/alpha/manifest.json`]: 1, + [`${O}/summaries/manifest.json`]: 1, + }); + + assert.deepEqual(await channelArchiveUrls(read, O, "alpha"), [ + `${O}/transcripts/alpha/manifest.json`, + `${O}/transcripts/alpha/page-0000.json`, + ]); + assert.deepEqual(await siteArchiveUrls(read, O), [ + `${O}/summaries/manifest.json`, + `${O}/summaries/page-0000.json`, + `${O}/corpus.json`, + `${O}/site.json`, + `${O}/search-aliases.json`, + `${O}/duplicates.json`, + ]); +}); diff --git a/common/lib/archive/offlineUrls.ts b/common/lib/archive/offlineUrls.ts @@ -0,0 +1,102 @@ +// WHAT AN OFFLINE COPY OF AN ARCHIVE CONSISTS OF, split the way it is paid for. +// +// Two lists, and the split is the whole point of this module: +// +// channelArchiveUrls — the PER-CHANNEL trees of one channel. Pinning a second +// channel genuinely costs this again, because none of it +// is shared. +// siteArchiveUrls — the SITE-WIDE documents of one origin: the flat trees +// (summaries, stats) and the root files. Identical for +// every channel of that origin, so it is downloaded once +// and evicted when the last pinned channel goes. +// +// The first cut of this shipped as ONE list and it was a real cost, not a +// tidiness point. Measured on jeralyzer: summaries ~14 MB, stats ~21.6 MB (its +// page-0000 alone is 20.97 MB) and /duplicates.json 5.9 MB — ~41.5 MB of +// byte-identical site data re-fetched per channel pinned (the service worker +// bulk-fetches with `cache: "reload"`, so they really do go over the wire), and +// the per-channel evict path sweeps only the per-channel prefixes, so removing +// a channel left every byte of it behind forever. +// +// The manifest READER IS INJECTED. The caller owns the transport, which matters +// here: the viewer reads these with `cache: "no-store"` precisely because a +// normal fetch would be answered from the service-worker cache this list exists +// to refill, and would then enumerate the stale copy. Injection also makes the +// lists directly assertable without a network. + +import { + ARCHIVE_TREES, + PER_CHANNEL_TREES, + ROOT_FILES, + isFlatTree, + manifestUrl, + pageUrl, + rootFileUrl, +} from "./contract"; + +// Every manifest shape these walks need, reduced to the one field a URL list is +// built from. A tree a site does not ship answers 404 and reads as absent. +export type PagedManifest = { pageCount?: number }; + +// Reads one manifest by URL. Resolves null for "this site does not ship it", +// which for these lists is normal rather than an error. +export type ManifestReader = (url: string) => Promise<PagedManifest | null>; + +// The manifest plus every page of one tree, or [] when the tree is absent. +// `slug` is undefined for a flat tree. +async function treeUrls( + read: ManifestReader, + base: string, + tree: (typeof ARCHIVE_TREES)[number], + slug: string | undefined, +): Promise<string[]> { + const manifest = manifestUrl(tree, slug, base); + const m = await read(manifest); + if (!m) return []; + const urls = [manifest]; + for (let p = 0; p < (m.pageCount ?? 0); p++) { + urls.push(pageUrl(tree, slug, p, base)); + } + return urls; +} + +// One channel's shards, across every per-channel tree the contract defines. +// +// Returns [] when the channel has no TRANSCRIPTS manifest — that is not a tree +// a readable channel can be missing, so the caller refuses the download rather +// than caching a channel that cannot be opened. Every other tree is optional +// and costs one 404 when absent. +export async function channelArchiveUrls( + read: ManifestReader, + base: string, + slug: string, +): Promise<string[]> { + const transcripts = await treeUrls(read, base, "transcripts", slug); + if (transcripts.length === 0) return []; + const urls = [...transcripts]; + for (const tree of PER_CHANNEL_TREES) { + if (tree === "transcripts") continue; + urls.push(...(await treeUrls(read, base, tree, slug))); + } + return urls; +} + +// One origin's site-wide documents: the flat trees and the root files. +// +// The root files are listed unconditionally rather than probed. /duplicates.json +// and /search-aliases.json are legitimately absent on many sites, the caller +// hands the list to a worker that skips what it cannot fetch, and probing them +// first would double the request count to learn something the fetch already +// tells us. +export async function siteArchiveUrls( + read: ManifestReader, + base: string, +): Promise<string[]> { + const urls: string[] = []; + for (const tree of ARCHIVE_TREES) { + if (!isFlatTree(tree)) continue; + urls.push(...(await treeUrls(read, base, tree, undefined))); + } + for (const file of ROOT_FILES) urls.push(rootFileUrl(file, base)); + return urls; +} diff --git a/common/lib/archive/reader.ts b/common/lib/archive/reader.ts @@ -19,12 +19,13 @@ import { type ChannelTranscriptsManifest, type ChannelSubsManifest, + type SubsManifest, type Manifest, } from "../manifest"; import type { TranscriptDetail, DisplaySummary } from "../transcripts"; import { summaryState, type VideoState } from "../availability"; import type { SubsDetail } from "../subs"; -import type { ChannelPostsManifest, Post } from "../posts"; +import type { ChannelPostsManifest, Post, PostsManifest } from "../posts"; import { coerceAliasConfig, type SearchAlias } from "../searchAliases"; import { resolveCanonicalSlug, @@ -39,7 +40,13 @@ import { DEFAULT_GROUP_FALLBACK_ID, type ChannelGroup, } from "../channelGroups"; -import { manifestUrl, pageUrl, rootFileUrl, type HubSite } from "./contract"; +import { + manifestUrl, + pageUrl, + rootFileUrl, + treeManifestUrl, + type HubSite, +} from "./contract"; import { recordRead } from "./io-stats"; // A channel the source can serve. `siteId`/`siteUrl` are only populated in hub @@ -373,6 +380,23 @@ export interface ArchiveReader { digestPage?(ch: ChannelRef, page: number): Promise<VideoDigest[]>; } +// An HTTP STATUS the archive itself returned, as opposed to a transport +// failure. A caller that treats absence as data — a site ships no +// /search-aliases.json, a channel ships no digests — needs to tell the two +// apart: a 404 is an answer and is worth caching, a dropped connection is +// neither. `message` is byte-identical to the plain Error this replaced, so +// anything that only reads the message is unaffected. +export class ArchiveHttpError extends Error { + constructor( + readonly status: number, + readonly url: string, + statusText: string, + ) { + super(`GET ${url} -> ${status} ${statusText}`); + this.name = "ArchiveHttpError"; + } +} + // Used when a source states no preference. Deliberately modest: each in-flight // page costs its raw bytes plus ~2.7× that once parsed, and this box is shared. export const DEFAULT_PAGE_CONCURRENCY = 4; @@ -590,8 +614,7 @@ export class RemoteSource implements ArchiveReader { subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { return this.subsManifests.take(ch.slug, async () => { try { - const res = await fetch(manifestUrl("subs", ch.slug, this.base)); - return res.ok ? ((await res.json()) as ChannelSubsManifest) : null; + return await this.readChannelSubsManifest(ch.slug); } catch { return null; } @@ -609,8 +632,7 @@ export class RemoteSource implements ArchiveReader { postsManifest(ch: ChannelRef): Promise<ChannelPostsManifest | null> { return this.postsManifests.take(ch.slug, async () => { try { - const res = await fetch(manifestUrl("posts", ch.slug, this.base)); - return res.ok ? ((await res.json()) as ChannelPostsManifest) : null; + return await this.readChannelPostsManifest(ch.slug); } catch { return null; } @@ -622,15 +644,12 @@ export class RemoteSource implements ArchiveReader { } videoIndex(): Promise<VideoIndex> { + // buildVideoIndex already treats a throw from either reader as absence + // (empty index / skipped page), so the raw reads' rejections land exactly + // where the old inline `res.ok ? … : null` did. this.index ??= buildVideoIndex( - async () => { - const res = await fetch(manifestUrl("summaries", undefined, this.base)); - return res.ok ? ((await res.json()) as Manifest) : null; - }, - async (page) => { - const res = await fetch(pageUrl("summaries", undefined, page, this.base)); - return res.ok ? ((await res.json()) as DisplaySummary[]) : null; - }, + () => this.readSummariesManifest(), + (page) => this.readSummariesPage(page), ); return this.index; } @@ -642,8 +661,7 @@ export class RemoteSource implements ArchiveReader { digestsManifest(ch: ChannelRef): Promise<ChannelDigestsManifest | null> { return this.digestManifests.take(ch.slug, async () => { try { - const res = await fetch(manifestUrl("digests", ch.slug, this.base)); - return res.ok ? ((await res.json()) as ChannelDigestsManifest) : null; + return await this.readChannelDigestsManifest(ch.slug); } catch { return null; } @@ -657,10 +675,7 @@ export class RemoteSource implements ArchiveReader { duplicateIndex(): Promise<DuplicateIndex> { this.duplicates ??= (async () => { try { - const res = await fetch(rootFileUrl(DUPLICATES_FILENAME, this.base)); - return buildDuplicateIndex( - res.ok ? ((await res.json()) as DuplicateReport) : null, - ); + return buildDuplicateIndex(await this.readDuplicates()); } catch { return buildDuplicateIndex(null); } @@ -670,14 +685,8 @@ export class RemoteSource implements ArchiveReader { statsIndex(): Promise<ReadonlyMap<string, VideoStat>> { this.stats ??= buildStatsIndex( - async () => { - const res = await fetch(manifestUrl("stats", undefined, this.base)); - return res.ok ? ((await res.json()) as StatsManifest) : null; - }, - async (page) => { - const res = await fetch(pageUrl("stats", undefined, page, this.base)); - return res.ok ? ((await res.json()) as VideoStat[]) : null; - }, + () => this.readStatsManifest(), + (page) => this.readStatsPage(page), ); return this.stats; } @@ -685,10 +694,7 @@ export class RemoteSource implements ArchiveReader { async loadAliases(): Promise<SearchAlias[]> { if (this.aliases) return this.aliases; try { - const res = await fetch(rootFileUrl("search-aliases.json", this.base)); - this.aliases = res.ok - ? coerceAliasConfig(await res.json()).aliases - : []; + this.aliases = coerceAliasConfig(await this.readAliasConfig()).aliases; } catch { this.aliases = []; } @@ -698,10 +704,7 @@ export class RemoteSource implements ArchiveReader { async loadGroups(): Promise<ChannelGroups> { if (this.groups) return this.groups; try { - const res = await fetch(manifestUrl("summaries", undefined, this.base)); - this.groups = res.ok - ? parseGroupsManifest(await res.json()) - : EMPTY_GROUPS; + this.groups = parseGroupsManifest(await this.readSummariesManifest()); } catch { this.groups = EMPTY_GROUPS; } @@ -713,23 +716,106 @@ export class RemoteSource implements ArchiveReader { // // `p` is a ROOT-RELATIVE path from the contract builders; the base is joined // here so the thrown error names the full URL, exactly as before. + // + // `kind` is the io-stats bucket. It is OPTIONAL, and the reads that pass none + // are the ones that never recorded a read before this file owned them (the + // summaries/stats/duplicates/alias folds all used a bare `fetch`). Adding + // them to the ledger would move the bench's structural counters for a reason + // that has nothing to do with the walk, so the blind spots are preserved + // deliberately rather than quietly closed. private async getJsonSized<T>( p: string, - kind: string, + kind?: string, ): Promise<{ value: T; bytes: number }> { const res = await fetch(`${this.base}${p}`); if (!res.ok) { - throw new Error(`GET ${this.base}${p} -> ${res.status} ${res.statusText}`); + throw new ArchiveHttpError(res.status, `${this.base}${p}`, res.statusText); } const raw = await res.text(); - recordRead(kind, raw.length); + if (kind !== undefined) recordRead(kind, raw.length); return { value: JSON.parse(raw) as T, bytes: raw.length }; } - private async getJson<T>(p: string, kind = "json"): Promise<T> { + private async getJson<T>(p: string, kind?: string): Promise<T> { return (await this.getJsonSized<T>(p, kind)).value; } + // ─── Raw document reads ─── + // + // One URL, one fetch, one parse, and a THROW on anything but a 200. Every + // tolerance policy is built on top of these — this class's own + // empty-when-absent folds below, and the viewer's component caches — so each + // published document's URL is written once and each caller picks its own + // answer to "absent". + // + // That split is not cosmetic: a tool scanning 30 channels wants `null` for a + // missing posts tree, while the viewer wants a rejection, because react-query + // retries a rejected query and will never retry a resolved `null`. Folding + // both into one tolerant method is how a transient network blip becomes a + // permanently empty panel. + + readCorpus(): Promise<SiteCorpusJson> { + return this.getJson<SiteCorpusJson>(rootFileUrl("corpus.json"), "corpus"); + } + + readAliasConfig(): Promise<unknown> { + return this.getJson<unknown>(rootFileUrl("search-aliases.json")); + } + + readDuplicates(): Promise<DuplicateReport> { + return this.getJson<DuplicateReport>(rootFileUrl(DUPLICATES_FILENAME)); + } + + readSummariesManifest(): Promise<Manifest> { + return this.getJson<Manifest>(manifestUrl("summaries")); + } + + readSummariesPage(page: number): Promise<DisplaySummary[]> { + return this.getJson<DisplaySummary[]>(pageUrl("summaries", undefined, page)); + } + + readStatsManifest(): Promise<StatsManifest> { + return this.getJson<StatsManifest>(manifestUrl("stats")); + } + + readStatsPage(page: number): Promise<VideoStat[]> { + return this.getJson<VideoStat[]>(pageUrl("stats", undefined, page)); + } + + // The SITE-level index of which channels ship a per-channel tree — a + // different document from a channel's own manifest (see treeManifestUrl). + readSubsSiteManifest(): Promise<SubsManifest> { + return this.getJson<SubsManifest>(treeManifestUrl("subs")); + } + + readPostsSiteManifest(): Promise<PostsManifest> { + return this.getJson<PostsManifest>(treeManifestUrl("posts")); + } + + readChannelSubsManifest(slug: string): Promise<ChannelSubsManifest> { + return this.getJson<ChannelSubsManifest>(manifestUrl("subs", slug)); + } + + readChannelPostsManifest(slug: string): Promise<ChannelPostsManifest> { + return this.getJson<ChannelPostsManifest>(manifestUrl("posts", slug)); + } + + // 404 → null, everything else → throw. The digest corpus is SPARSE BY DESIGN + // (a channel with no digests ships no manifest at all), so the 404 is the + // document's own answer and belongs in the raw read; a 500 or a dropped + // connection is not proof of absence and must stay distinguishable. + async readChannelDigestsManifest( + slug: string, + ): Promise<ChannelDigestsManifest | null> { + const p = manifestUrl("digests", slug); + const res = await fetch(`${this.base}${p}`); + if (res.status === 404) return null; + if (!res.ok) { + throw new ArchiveHttpError(res.status, `${this.base}${p}`, res.statusText); + } + return (await res.json()) as ChannelDigestsManifest; + } + private channelList?: Promise<ChannelRef[]>; listChannels(opts: { refresh?: boolean } = {}): Promise<ChannelRef[]> { @@ -745,10 +831,7 @@ export class RemoteSource implements ArchiveReader { } private async readChannels(): Promise<ChannelRef[]> { - const corpus = await this.getJson<SiteCorpusJson>( - rootFileUrl("corpus.json"), - "corpus", - ); + const corpus = await this.readCorpus(); return (corpus.channels ?? []).map((c) => ({ key: c.slug, slug: c.slug, diff --git a/common/lib/archive/readers.test.ts b/common/lib/archive/readers.test.ts @@ -0,0 +1,181 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { ArchiveHttpError, RemoteSource } from "./reader"; +import { channelRef, readerFor, resetReaders } from "./readers"; + +// The same in-memory-archive trick as reader.test.ts: a Map from URL to JSON +// body installed as `fetch`, so every assertion below is about the URLs the +// reader actually asks for. The viewer's caches are thin wrappers over these +// reads, so pinning the reads pins the wire. + +type Archive = { fetches: string[]; body: Map<string, unknown> }; + +function installFetch(entries: Record<string, unknown>): { + archive: Archive; + restore: () => void; +} { + const archive: Archive = { + fetches: [], + body: new Map(Object.entries(entries)), + }; + const real = globalThis.fetch; + globalThis.fetch = (async (input: RequestInfo | URL) => { + const url = String(input); + archive.fetches.push(url); + if (!archive.body.has(url)) { + return new Response("not found", { status: 404, statusText: "Not Found" }); + } + return new Response(JSON.stringify(archive.body.get(url)), { status: 200 }); + }) as typeof fetch; + return { + archive, + restore: () => { + globalThis.fetch = real; + resetReaders(); + }, + }; +} + +test("readerFor: one reader per origin, kept for the life of the page", () => { + resetReaders(); + const same = readerFor(""); + assert.equal(readerFor(""), same); + assert.equal(readerFor(), same, "the default argument is the same-origin key"); + const other = readerFor("https://b.example"); + assert.notEqual(other, same); + assert.equal(readerFor("https://b.example"), other); + assert.ok(same instanceof RemoteSource); + resetReaders(); + assert.notEqual(readerFor(""), same, "resetReaders drops the caches with it"); +}); + +test("the same-origin reader emits ROOT-RELATIVE urls, byte-identical to the old walk", async () => { + const { archive, restore } = installFetch({ + "/summaries/manifest.json": { version: 3, pageCount: 2, channels: [] }, + "/summaries/page-0001.json": [], + "/stats/manifest.json": { version: 1, pageCount: 1, channels: [] }, + "/stats/page-0000.json": [], + "/subs/manifest.json": { version: 4, channels: [] }, + "/posts/manifest.json": { version: 1, channels: [] }, + "/search-aliases.json": { aliases: [] }, + "/duplicates.json": { clusters: [] }, + "/transcripts/alpha/manifest.json": { slugToPage: {}, pageCount: 1 }, + "/subs/alpha/manifest.json": { slugToPage: {}, pageCount: 1 }, + "/posts/alpha/manifest.json": { slugToPage: {}, pageCount: 1 }, + }); + try { + const r = readerFor(""); + await r.readSummariesManifest(); + await r.readSummariesPage(1); + await r.readStatsManifest(); + await r.readStatsPage(0); + await r.readSubsSiteManifest(); + await r.readPostsSiteManifest(); + await r.readAliasConfig(); + await r.readDuplicates(); + await r.readChannelSubsManifest("alpha"); + await r.readChannelPostsManifest("alpha"); + assert.deepEqual(archive.fetches, [ + "/summaries/manifest.json", + "/summaries/page-0001.json", + "/stats/manifest.json", + "/stats/page-0000.json", + "/subs/manifest.json", + "/posts/manifest.json", + "/search-aliases.json", + "/duplicates.json", + "/subs/alpha/manifest.json", + "/posts/alpha/manifest.json", + ]); + } finally { + restore(); + } +}); + +test("a federated origin prefixes every one of those, and nothing else changes", async () => { + const O = "https://member.example"; + const { archive, restore } = installFetch({ + [`${O}/summaries/manifest.json`]: { version: 3, pageCount: 0, channels: [] }, + [`${O}/duplicates.json`]: { clusters: [] }, + }); + try { + const r = readerFor(O); + await r.readSummariesManifest(); + await r.readDuplicates(); + assert.deepEqual(archive.fetches, [ + `${O}/summaries/manifest.json`, + `${O}/duplicates.json`, + ]); + } finally { + restore(); + } +}); + +test("a raw read rejects with the status; the tolerant fold over it resolves empty", async () => { + const { restore } = installFetch({}); + try { + const r = readerFor(""); + await assert.rejects( + () => r.readDuplicates(), + (err: unknown) => + err instanceof ArchiveHttpError && + err.status === 404 && + /GET \/duplicates\.json -> 404/.test((err as Error).message), + ); + // Same document, same 404, through the fold the MCP uses. + assert.equal((await r.duplicateIndex()).size, 0); + // And through the summaries fold, which a partial/absent index must not + // turn into an error — the page planner reads empty as "scan anyway". + assert.equal((await r.videoIndex()).size, 0); + assert.deepEqual(await r.loadAliases(), []); + } finally { + restore(); + } +}); + +test("a digests manifest 404 is the layer's own answer; any other status is not", async () => { + const { archive, restore } = installFetch({ + "/digests/beta/manifest.json": { slugToPage: { v1: 0 }, pageCount: 1 }, + }); + try { + const r = readerFor(""); + assert.equal(await r.readChannelDigestsManifest("alpha"), null); + assert.ok(await r.readChannelDigestsManifest("beta")); + assert.deepEqual(archive.fetches, [ + "/digests/alpha/manifest.json", + "/digests/beta/manifest.json", + ]); + } finally { + restore(); + } + + const bad = installFetch({}); + const realFetch = globalThis.fetch; + globalThis.fetch = (async (input: RequestInfo | URL) => { + void input; + return new Response("boom", { status: 500, statusText: "Server Error" }); + }) as typeof fetch; + try { + await assert.rejects( + () => readerFor("").readChannelDigestsManifest("alpha"), + /-> 500/, + ); + } finally { + globalThis.fetch = realFetch; + bad.restore(); + } +}); + +test("channelRef: slug-addressed, and carries the origin only when there is one", () => { + assert.deepEqual(channelRef("alpha"), { + key: "alpha", + slug: "alpha", + name: "alpha", + }); + assert.deepEqual(channelRef("alpha", "https://b.example"), { + key: "alpha", + slug: "alpha", + name: "alpha", + siteUrl: "https://b.example", + }); +}); diff --git a/common/lib/archive/readers.ts b/common/lib/archive/readers.ts @@ -0,0 +1,73 @@ +// One ArchiveReader per origin, for the code that reads an archive from a +// BROWSER. +// +// The viewer's component caches are keyed by origin ("" = same-origin, a full +// origin like "https://x.example" for a federated hub member — see +// components/originId.ts). Each one used to carry its own hand-written walk of +// the shard scheme; now each one calls a RemoteSource from here, so the URL +// shape is the contract's and the promise-coalescing/LRU behaviour is the +// reader's rather than eight near-copies of it. +// +// WHY THIS LIVES IN lib/archive/ AND NOT components/: it is the reader +// registry, not a React concern — no hooks, no JSX, no node imports — and a new +// components/*.ts file would need its own `exports` entry in common's +// package.json. lib/* is wildcarded, so this file costs nothing to reach from +// editor/ or export/. +// +// RemoteSource with an EMPTY base is the same-origin case: archiveUrl() leaves a +// path root-relative when there is no base, which is byte-for-byte the +// `${idBaseUrl(origin)}/transcripts/…` the caches built before. So the +// single-site export app's URLs do not move at all. + +import { RemoteSource, type ChannelRef } from "./reader"; + +// A BROWSER'S page-cache budget, which is not a server's. +// +// RemoteSource defaults to 48 MB (pageCacheBudgetBytes) and holds TWO page +// caches — transcripts and subs — so an origin costs 96 MB. That is a sane +// ceiling for one long-lived MCP process reading one corpus. It is not one for +// a tab, and it is emphatically not one for a hub page, which holds a reader +// per member site with no shared ceiling and no way for a browser to turn the +// env knob down. +// +// 8 MB, so an origin is 16 MB and five federated members are 80 MB rather than +// 480 MB. The reason it can be this small without costing reads: the viewer's +// caches memoise every RECORD they have ever seen (transcriptCache.resolved, +// plus IndexedDB), so an evicted page is only re-fetched for a video nobody has +// opened yet — the case that was going to be a fetch anyway. PageCache's +// MIN_CACHED_PAGES floor still keeps two entries even when one page of a +// VOD-sized corpus exceeds the whole budget on its own. +// +// The MCP path does NOT come through here: mcp/src/source.ts constructs +// RemoteSource directly and keeps the 48 MB default, so no bench counter moves. +const BROWSER_PAGE_CACHE_BYTES = 8 * 1024 * 1024; + +const byOrigin = new Map<string, RemoteSource>(); + +// The reader for an origin, created on first use and kept for the life of the +// page — the readers hold the manifest/page caches, so a fresh one per call +// would be a fresh cache per call. +export function readerFor(origin = ""): RemoteSource { + let reader = byOrigin.get(origin); + if (reader === undefined) { + reader = new RemoteSource(origin, BROWSER_PAGE_CACHE_BYTES); + byOrigin.set(origin, reader); + } + return reader; +} + +// A minimal ChannelRef for the per-channel reader calls. The viewer addresses a +// channel by slug alone; `key`/`name` exist for the MCP's channel list and are +// not read on this path. Reader caches key off `slug`, so a fresh object per +// call is free. +export function channelRef(slug: string, origin = ""): ChannelRef { + return origin + ? { key: slug, slug, name: slug, siteUrl: origin } + : { key: slug, slug, name: slug }; +} + +// Drop every reader and its caches. For tests, and for any future "the archive +// was rebuilt under us" reset — not called on a normal page. +export function resetReaders(): void { + byOrigin.clear(); +} diff --git a/export/app/duplicates/DuplicatesClient.tsx b/export/app/duplicates/DuplicatesClient.tsx @@ -10,6 +10,7 @@ import { useMediaQuery } from "yt-dlp-transcript-common/lib/useMediaQuery"; import { Checkbox } from "yt-dlp-transcript-common/components/ui/checkbox"; import { VirtualRow } from "yt-dlp-transcript-common/components/VirtualRow"; import { fetchTranscript } from "yt-dlp-transcript-common/components/transcriptCache"; +import { fetchDuplicateReport } from "yt-dlp-transcript-common/components/duplicatesCache"; import { formatDate } from "yt-dlp-transcript-common/lib/format"; import type { DuplicateCluster, @@ -102,14 +103,9 @@ export function DuplicatesClient() { // state rather than an error. useEffect(() => { let alive = true; - fetch("/duplicates.json") - .then((r) => (r.ok ? r.json() : null)) - .then((report: DuplicateReport | null) => { - if (alive) setState({ status: "ready", report }); - }) - .catch(() => { - if (alive) setState({ status: "ready", report: null }); - }); + fetchDuplicateReport().then((report) => { + if (alive) setState({ status: "ready", report }); + }); return () => { alive = false; }; diff --git a/export/app/lib/offlineCache.ts b/export/app/lib/offlineCache.ts @@ -12,10 +12,11 @@ // per (origin, slug); the site SW ignores it (same-origin only). import { - pageFileName, - type ChannelTranscriptsManifest, -} from "yt-dlp-transcript-common/lib/manifest"; -import { idBaseUrl, makeId } from "yt-dlp-transcript-common/components/originId"; + channelArchiveUrls, + siteArchiveUrls, + type PagedManifest, +} from "yt-dlp-transcript-common/lib/archive/offlineUrls"; +import { idBaseUrl, makeId, splitId } from "yt-dlp-transcript-common/components/originId"; const PINNED_KEY = "ytdlp-tb:offline-channels"; @@ -56,16 +57,17 @@ function sendToSw( }); } -async function fetchManifest( - origin: string, - slug: string, -): Promise<ChannelTranscriptsManifest | null> { +// Deliberately NOT the shared ArchiveReader: the reader is a normal `fetch`, so +// a page walking it would be answered from the very service-worker cache this +// function exists to REFILL, and would compute its download list from the stale +// copy. `cache: "no-store"` is the whole point of this read. What is shared is +// the thing that actually drifted — the URL shape, and the two lists in +// common/lib/archive/offlineUrls.ts. +async function readManifest(url: string): Promise<PagedManifest | null> { try { - const res = await fetch(`${idBaseUrl(origin)}/transcripts/${slug}/manifest.json`, { - cache: "no-store", - }); + const res = await fetch(url, { cache: "no-store" }); if (!res.ok) return null; - return (await res.json()) as ChannelTranscriptsManifest; + return (await res.json()) as PagedManifest; } catch { return null; } @@ -73,9 +75,34 @@ async function fetchManifest( export type DownloadProgress = { done: number; total: number }; -// Download every shard of a channel into the SW cache, reporting progress. The -// URL list is the channel manifest plus each page-NNNN.json, prefixed with the -// channel's origin so a hub can cache cross-origin channels. +// Is any channel of this origin already pinned? That is the test for whether +// the origin's SITE DATA (summaries, stats, the root files) is already in the +// cache, and it is deliberately answered from the pinned registry rather than +// by asking the worker: the two lists are written and evicted together, so one +// pinned channel means one downloaded copy of the site data. +// +// The failure mode this accepts is the one the registry already has — a viewer +// who clears site data while keeping localStorage gets a pin with no bytes +// behind it, which is exactly why channelCachedPages() exists. The honest fix +// is a status round-trip per origin; it is not worth a second SW message for a +// case a re-pin already repairs. +function originHasPin(origin: string): boolean { + return pinnedChannels().some((id) => splitId(id).origin === origin); +} + +// Download a channel for offline use, reporting progress. +// +// EVERY PER-CHANNEL LAYER THAT HAS A MANIFEST, not just /transcripts — that is +// why an offline channel now has its live chat, its social posts and its AI +// digests — plus, ONCE PER ORIGIN, the site-wide documents the viewer needs to +// render any of it: the summaries index, stats, and the root files. The +// duplicates report is one of those root files, which is why the duplicates page +// used to be dead with no connection. +// +// The site data is downloaded with the FIRST channel pinned on an origin and +// skipped for every channel after it. Pinning it per channel is ~41.5 MB of +// identical bytes each time on a corpus the size of jeralyzer, re-fetched for +// real because the worker bulk-caches with `cache: "reload"`. export async function downloadChannelOffline( origin: string, slug: string, @@ -83,13 +110,19 @@ export async function downloadChannelOffline( ): Promise<boolean> { const sw = await controller(); if (!sw) return false; - const manifest = await fetchManifest(origin, slug); - if (!manifest) return false; const base = idBaseUrl(origin); - const urls = [`${base}/transcripts/${slug}/manifest.json`]; - for (let p = 0; p < manifest.pageCount; p++) { - urls.push(`${base}/transcripts/${slug}/${pageFileName(p)}`); + + const urls = await channelArchiveUrls(readManifest, base, slug); + // No transcripts manifest: the channel is not readable, so there is nothing + // to take offline. Refused before anything is written, exactly as before. + if (urls.length === 0) return false; + + // Checked BEFORE this channel is pinned, so the first channel of an origin + // brings the site data with it and later ones do not. + if (!originHasPin(origin)) { + urls.push(...(await siteArchiveUrls(readManifest, base))); } + await sendToSw(sw, { type: "CACHE_URLS", origin, slug, urls }, (data) => { if (data.type === "progress") { onProgress?.({ done: data.done, total: data.total }); @@ -99,6 +132,11 @@ export async function downloadChannelOffline( return true; } +// Remove a channel's offline copy — and, when it was the LAST pinned channel of +// its origin, that origin's site data with it. Without the second half the +// summaries/stats/root bytes survive every removal and there is no way to get +// them back out of the cache at all; they are downloaded once, so they have to +// be evicted once. export async function evictChannelOffline( origin: string, slug: string, @@ -108,6 +146,9 @@ export async function evictChannelOffline( await sendToSw(sw, { type: "EVICT_CHANNEL", origin, slug }).catch(() => {}); } unpin(origin, slug); + if (sw && !originHasPin(origin)) { + await sendToSw(sw, { type: "EVICT_SITE", origin }).catch(() => {}); + } } // How many page shards of a channel are currently cached (0 = not offline). diff --git a/export/service-worker/site-sw.js b/export/service-worker/site-sw.js @@ -8,12 +8,18 @@ * Caches: * SHELL — app shell: hashed /_next/static/* (immutable, cache-first) + HTML * navigations (network-first, cache fallback) + icons/manifest. - * PAGES — transcript/subs/summaries JSON shards (manifest.json + page-NNNN.json). - * Pages are cache-first; manifests are network-first so a rebuild is - * seen. Invalidation is keyed off manifest.generatedAt: when a channel's - * manifest generatedAt changes, that channel's cached page shards are - * evicted (the shards have stable, non-content-hashed URLs, so this is - * the only reliable staleness signal). + * PAGES — every published JSON document: the per-channel shard trees + * (manifest.json + page-NNNN.json), the site-wide flat trees + * (summaries, stats) and the root documents (corpus.json, site.json, + * search-aliases.json, duplicates.json). Per-channel pages are + * cache-first; per-channel manifests are network-first so a rebuild is + * seen, with invalidation keyed off manifest.generatedAt — when a + * channel's manifest generatedAt changes, that channel's cached page + * shards are evicted (the shards have stable, non-content-hashed URLs, + * so this is the only reliable staleness signal). The site-wide + * documents have no per-channel generatedAt to evict against, so they + * are network-first with a cache fallback: fresh online, present + * offline, never stale-forever. * META — tiny synthetic Responses storing each channel's last-seen generatedAt * (avoids IndexedDB inside the SW). */ @@ -23,8 +29,18 @@ const SHELL = `shell-${VERSION}`; const PAGES = `pages-${VERSION}`; const META = `meta-${VERSION}`; -// Matches /transcripts/<slug>/... and the parallel /subs, /summaries trees. -const SHARD_RE = /^\/(transcripts|subs|posts|digests|summaries)\/([^/]+)\/(.+)$/; +// THE THREE URL FAMILIES OF THE PUBLISHED CONTRACT. A service worker cannot +// import, so these lists are hand-written — and `common/lib/archive/contract. +// test.ts` reads THIS FILE and fails if they drift from CONTRACT.layers, +// PER_CHANNEL_TREES and ROOT_FILES. Add a layer to the contract and this file +// is what the test sends you to. + +// Per-channel trees: /<tree>/<slug>/(manifest.json|page-NNNN.json). +const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/; +// Flat trees: one manifest and its pages at the tree root, no channel level. +const FLAT_RE = /^\/(summaries|stats)\/(.+)$/; +// The root documents a reader fetches by name. +const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/; self.addEventListener("install", () => { // Activate immediately — no precache list (corpus is too large to bundle). @@ -60,6 +76,16 @@ self.addEventListener("fetch", (event) => { return; } + // Site-wide archive documents. Not cache-first: there is no per-channel + // generatedAt to evict them against, so a cache-first copy would be stale + // until the SW version changed. Network-first keeps them fresh online and + // present offline — which is what made the duplicates page work with no + // connection. + if (FLAT_RE.test(url.pathname) || ROOT_RE.test(url.pathname)) { + event.respondWith(networkFirst(req, PAGES)); + return; + } + // App shell. if (url.pathname.startsWith("/_next/static/") || url.pathname.startsWith("/icons/")) { event.respondWith(cacheFirst(req, SHELL)); @@ -86,6 +112,20 @@ async function cacheFirst(req, cacheName) { } } +// Network-first: serve the fresh copy and remember it; fall back to whatever is +// cached when the network is gone. +async function networkFirst(req, cacheName) { + const cache = await caches.open(cacheName); + try { + const res = await fetch(req); + if (res.ok) cache.put(req, res.clone()); + return res; + } catch (err) { + const hit = await cache.match(req); + return hit || Response.error(); + } +} + // Navigations: network-first (fresh HTML after deploys), fall back to cache, then // to any cached document so the installed app still opens offline. async function networkFirstDoc(req) { @@ -135,10 +175,14 @@ async function handleManifest(req, slug) { async function evictChannelPages(slug) { const pages = await caches.open(PAGES); const keys = await pages.keys(); + // The PER-CHANNEL trees, and only those: /summaries/<slug>/ never existed + // (summaries is flat) and /posts, /digests were missing, so a rebuilt channel + // kept serving its old social posts and AI digests from cache. const prefixes = [ `/transcripts/${slug}/`, `/subs/${slug}/`, - `/summaries/${slug}/`, + `/posts/${slug}/`, + `/digests/${slug}/`, ]; await Promise.all( keys.map((k) => { @@ -151,6 +195,22 @@ async function evictChannelPages(slug) { ); } +// The site-wide documents: the flat trees and the root files, i.e. exactly what +// the page downloads ONCE per origin. Evicted when its last pinned channel goes, +// because nothing else ever removes them — the per-channel sweep above only +// knows about /<tree>/<slug>/ prefixes. The same two regexes drive the fetch +// handler, so this can never fall out of step with what was cached. +async function evictSiteData() { + const pages = await caches.open(PAGES); + const keys = await pages.keys(); + await Promise.all( + keys.map((k) => { + const p = new URL(k.url).pathname; + return FLAT_RE.test(p) || ROOT_RE.test(p) ? pages.delete(k) : null; + }), + ); +} + // Per-channel generatedAt stored as a synthetic Response in the META cache. async function readMeta(slug) { const meta = await caches.open(META); @@ -172,6 +232,10 @@ self.addEventListener("message", (event) => { event.waitUntil( evictChannelPages(data.slug).then(() => port && port.postMessage({ ok: true })), ); + } else if (data.type === "EVICT_SITE") { + event.waitUntil( + evictSiteData().then(() => port && port.postMessage({ ok: true })), + ); } else if (data.type === "CHANNEL_STATUS") { event.waitUntil(channelStatus(data.slug, port)); } diff --git a/export/service-worker/sw-hub.js b/export/service-worker/sw-hub.js @@ -2,7 +2,7 @@ * * Same offline model as the site SW (browse-cache + opt-in per-channel * download), but CROSS-ORIGIN: the hub federates content from many origins, so - * this worker caches transcript/subs/summaries shards from ANY origin — not + * this worker caches every published archive document from ANY origin — not * just its own. The only origins it ever reaches are ones the user added, whose * JSON is CORS-open (Access-Control-Allow-Origin: *); a cross-origin fetch is * therefore a readable `cors` response that can be cached (never `no-cors` — @@ -21,9 +21,19 @@ const SHELL = `shell-${VERSION}`; const PAGES = `pages-${VERSION}`; const META = `meta-${VERSION}`; -// Matches /transcripts/<slug>/... and the parallel /subs, /summaries trees, on -// any origin (we test url.pathname, so it's origin-independent). -const SHARD_RE = /^\/(transcripts|subs|posts|summaries)\/([^/]+)\/(.+)$/; +// THE THREE URL FAMILIES OF THE PUBLISHED CONTRACT, matched on url.pathname so +// they are origin-independent. A service worker cannot import, so these lists +// are hand-written — and `common/lib/archive/contract.test.ts` reads THIS FILE +// (and site-sw.js) and fails if they drift from CONTRACT.layers, +// PER_CHANNEL_TREES and ROOT_FILES. `digests` was missing here entirely: the +// hub federated AI digests it could never cache. + +// Per-channel trees: /<tree>/<slug>/(manifest.json|page-NNNN.json). +const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/; +// Flat trees: one manifest and its pages at the tree root, no channel level. +const FLAT_RE = /^\/(summaries|stats)\/(.+)$/; +// The root documents a reader fetches by name. +const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/; self.addEventListener("install", () => { self.skipWaiting(); @@ -57,6 +67,15 @@ self.addEventListener("fetch", (event) => { return; } + // Site-wide archive documents, also from ANY origin — a member site's + // summaries index and its /duplicates.json are as federated as its shards. + // Network-first: no per-channel generatedAt to evict them against, so fresh + // online and present offline rather than stale-forever. + if (FLAT_RE.test(url.pathname) || ROOT_RE.test(url.pathname)) { + event.respondWith(networkFirst(req, PAGES)); + return; + } + // App shell is same-origin only (the hub's own bundle). if (url.origin !== self.location.origin) return; if (url.pathname.startsWith("/_next/static/") || url.pathname.startsWith("/icons/")) { @@ -83,6 +102,20 @@ async function cacheFirst(req, cacheName) { } } +// Network-first: serve the fresh copy and remember it; fall back to whatever is +// cached when the network is gone. +async function networkFirst(req, cacheName) { + const cache = await caches.open(cacheName); + try { + const res = await fetch(req); + if (res.ok) cache.put(req, res.clone()); + return res; + } catch (err) { + const hit = await cache.match(req); + return hit || Response.error(); + } +} + async function networkFirstDoc(req) { const cache = await caches.open(SHELL); try { @@ -133,10 +166,13 @@ function belongsToChannel(entryUrl, origin, slug) { const u = new URL(entryUrl); const wantOrigin = origin || self.location.origin; if (u.origin !== wantOrigin) return false; + // The PER-CHANNEL trees, and only those: /summaries/<slug>/ never existed + // (summaries is flat) and /posts, /digests were missing. const prefixes = [ `/transcripts/${slug}/`, `/subs/${slug}/`, - `/summaries/${slug}/`, + `/posts/${slug}/`, + `/digests/${slug}/`, ]; return prefixes.some( (pre) => u.pathname.startsWith(pre) && !u.pathname.endsWith("manifest.json"), @@ -151,6 +187,26 @@ async function evictChannelPages(origin, slug) { ); } +// One member origin's site-wide documents — the flat trees and the root files, +// exactly what the page downloads ONCE per origin. Scoped by origin like +// everything else here, so evicting one member never touches another's. The +// per-channel sweep above only knows /<tree>/<slug>/ prefixes, so without this +// these bytes would survive every removal. +async function evictSiteData(origin) { + const pages = await caches.open(PAGES); + const wantOrigin = origin || self.location.origin; + const keys = await pages.keys(); + await Promise.all( + keys.map((k) => { + const u = new URL(k.url); + if (u.origin !== wantOrigin) return null; + return FLAT_RE.test(u.pathname) || ROOT_RE.test(u.pathname) + ? pages.delete(k) + : null; + }), + ); +} + // Per-(origin, channel) generatedAt stored as a synthetic Response in META. function metaKey(origin, slug) { return `/__gen__/${encodeURIComponent(origin || "")}/${slug}`; @@ -178,6 +234,10 @@ self.addEventListener("message", (event) => { () => port && port.postMessage({ ok: true }), ), ); + } else if (data.type === "EVICT_SITE") { + event.waitUntil( + evictSiteData(origin).then(() => port && port.postMessage({ ok: true })), + ); } else if (data.type === "CHANNEL_STATUS") { event.waitUntil(channelStatus(origin, data.slug, port)); } diff --git a/plans/one-core-phase-2.md b/plans/one-core-phase-2.md @@ -780,6 +780,114 @@ orchestration, `scanPlan.test.ts` sits beside it, and it calls the single `passesFilters` from `lib/search/evalTree` — so the pruner and the scanner still cannot disagree about what matches, which was the one property that mattered. +### S2a — shipped 2026-09-12 + +Branch `one-core/phase-2-s2a`, off `7f86aef` (the +`integrate/2026-09-storage-priority` tip, S1 merged). Four commits, +three code commits `dc3aef1` → `2eb9c3b` plus this note, unmerged. **No URL shape moved, no `corpus.json` byte +moved, no CONTRACT version moved, no architecture allow-list entry added or +burned, and no `common/package.json` exports line added.** + +| commit | what | +|---|---| +| `dc3aef1` | the eight caches — plus `SearchDataContext`, `siteRegistry` and `DuplicatesClient` — onto one `RemoteSource` per origin; `lib/archive/readers.ts`; the reader's raw document reads | +| `8ca7022` | `offlineCache` enumerates `ARCHIVE_TREES` + `ROOT_FILES`; both service workers widened; the SW and search-worker guard cases in `contract.test.ts` | +| `2eb9c3b` | `archive/readers.test.ts` + `components/archiveCaches.test.ts` | +| (this commit) | this note | + +#### Seven divergences from the plan above, each because the code said so + +**1. The caches do NOT adopt `ArchiveReader`'s tolerant methods. The reader +gained a block of RAW DOCUMENT READS underneath them instead.** The slice was +specified as `memo(originId, () => reader.X(...))`, and for transcripts that is +exactly what it is. For everything else `ArchiveReader` is the wrong surface, +twice over. + +It has no method at all for four of the documents the viewer reads: +`/subs/manifest.json` and `/posts/manifest.json` (the SITE-level index of which +channels ship a tree — a different document from a channel's own manifest), and +the raw summaries/stats manifest-plus-pages. What it exposes for those is +`videoIndex()` / `statsIndex()`, folded slug-keyed maps, which are the wrong +shape for a UI that renders arrays and loads page by page under react-query. + +And where a method does exist, its answer to absence is the opposite of the +viewer's. `subsManifest`/`postsManifest`/`digestsManifest` resolve `null` for +anything that is not a 200 — right for a tool probing thirty channels, wrong +for a browser, because **react-query retries a rejected query and will never +retry a resolved `null`**. Folding both policies into one method is exactly how +a transient blip becomes a panel that is empty for the rest of the session. + +So `RemoteSource` now has twelve one-to-three-line reads that do URL + fetch + +parse and THROW, and every tolerance policy sits on top and picks its own +answer. **The MCP's behaviour is unchanged by construction**: each tolerant +method is rebased on the raw read it duplicated, and `buildVideoIndex` / +`buildStatsIndex` already treat a throw from their readers as absence, so the +degradation paths are the same code they were. This is not the `record(layer, +slug, id)` the plan forbids — these are the DOCUMENTS of the walk, not records +inside a shard, and no read count moves. + +`ArchiveHttpError` came with them, for the one caller that has to tell a 404 +from a dropped connection (the alias dictionary, whose query has `staleTime: +Infinity`). Its `message` is byte-identical to the `Error` it replaced. + +**2. Two memos stay in `components/`, and neither is a leftover.** The +subs/posts CHANNEL-manifest memos keep the throwing policy of (1). The posts +and digest PAGE memos exist because the reader caches subs and transcript pages +but not those two — and `fetchThread` walks every page of a channel, so without +a memo opening two posts in one thread re-downloads the whole channel. Adding +those caches to `RemoteSource` instead would have moved the bench's read counts, +which is not a thing to do in a slice that claims the walk is all that changed. + +**3. `io-stats` blind spots are preserved deliberately.** The raw reads take an +OPTIONAL `kind`, and the ones that pass none are exactly the reads that +recorded nothing before (the summaries/stats/duplicates/alias folds used a bare +`fetch`). Closing them would move the bench's structural counters for a reason +unrelated to the walk. That is also why the bench was not re-run for this slice +— nothing in the reader's semantics or read counts moved. + +**4. `contract.ts` needed `treeManifestUrl`, and `ARCHIVE_TREES` / +`PER_CHANNEL_TREES` beside it.** `manifestUrl("subs")` throws with no slug, and +correctly so — but `/subs/manifest.json` is a real published document. A +per-channel tree has BOTH a root manifest and a per-channel one; `manifestUrl` +builds the second, `treeManifestUrl` the first, and for a flat tree they are the +same file. The two lists are what the offline cache and the SW guard test +enumerate; `CONTRACT.layers` stays the frozen published list. + +**5. The service workers needed more than a wider `SHARD_RE`, and the plan's +stated goal is why.** `SHARD_RE` matches three path segments, so +`/duplicates.json` and `/summaries/*` never matched it and the fetch handler let +them through to the network — downloading a root file into the cache and never +serving it from there closes nothing. Both workers gain `FLAT_RE` and `ROOT_RE` +answered NETWORK-FIRST (those documents have no per-channel `generatedAt` to +evict against, so cache-first would be stale until the SW version moved). + +Three drifts surfaced while writing the lists down. `summaries` was in +site-sw's `SHARD_RE` and in both eviction prefix lists: dead in both places, +because it is a flat tree and `/summaries/<slug>/` does not exist. `/posts/` and +`/digests/` were absent from both eviction lists, so a rebuilt channel kept +serving its old posts and digests from cache. And sw-hub had no `digests` at +all — the hub federated a layer it could never cache. + +**6. `offlineCache` keeps its own `cache: "no-store"` fetch** rather than taking +the shared reader. The reader is a plain `fetch`, so it would be answered from +the very service-worker cache that function exists to refill, and would compute +its download list from the stale copy. What is shared is the thing that actually +drifted: the URL shape. + +**7. One new `components/*.ts`, against the invariant, and why it is not the +thing the invariant protects.** `components/archiveCaches.test.ts` is a test: it +is imported by nothing, the `"./components/*"` catch-all maps to `.tsx`, and it +needs no `package.json` exports line — which is the cost the invariant names. It +cannot live under `lib/` because S3 adds `"components"` to `FORBIDDEN.lib` in +the architecture test, and a test for the caches has to import them. + +Three walk sites beyond the eight caches moved too, all of them covered by the +brief's "only the fetch walk moves": `SearchDataContext.tsx` (four URLs, state +untouched), `siteRegistry.ts` (two), and `export/app/duplicates/ +DuplicatesClient.tsx`'s bare `fetch("/duplicates.json")`, which became a new +`fetchDuplicateReport()` export on `duplicatesCache`. A repo-wide sweep for +hand-written archive paths now returns only `searchIndex.worker.ts` — the +deliberate copy, and it is pinned. #### Gates @@ -903,3 +1011,125 @@ settings, social links or the footer. load-bearing prose: they are the only thing stopping the two evaluations of the same algebra from drifting, because no test can compare a streaming slug-set walk against a per-record boolean. +- `pnpm --filter yt-dlp-transcript-common test` — **1095 passed / 0 failed** + (baseline 1077 at `7f86aef`, + 3 `contract.test.ts`, + 6 `readers.test.ts`, + + 9 `archiveCaches.test.ts`; none lost). Architecture test green, allow-list + untouched. +- `pnpm --filter yt-dlp-transcript-mcp test` — **205 passed / 0 failed**. +- `pnpm test:scripts` — **71 passed / 1 skipped**. +- `pnpm --filter export exec next build` — **compiled successfully**, 11 static + pages. The only proof `reader-fs.ts` is unreachable from a client module, and + it now has eight more `components/` modules reaching `lib/archive/` to prove + it about. (The one warning is S1's pre-existing NFT trace.) +- **compose-site byte-identity** over the committed fixture: `IDENTICAL modulo + the build clock` for both trees — 16 files under `public/`, 7 under `index/`. +- **The mcp bench was NOT run**, per the brief — the reader's semantics and read + counts do not move (see divergence 3). +- **e2e**, behind the queue lock from worktree #10 (ports 4001/4011/4010/4020): + export `e2e` **172 passed / 0 failed** (S1's baseline exactly), `e2e:hub` + **5 passed / 0 failed**. `e2e:2origin` was skipped: it is known-red on the + base for the hub `/ask` prerender, which S1 verified at `c7f7b90` and which + nothing here is in the path of. +- The editor specs `export-search` + `export-player-platform-cache`: **20 passed + / 1 failed**, and the one failure is a SUBSET-RUN ARTIFACT, not a regression. + `export-search.spec.ts:612` opens `editor/test-settings.json` with `readJson`; + that file is gitignored and is only ever written by `writeSettings()`, i.e. by + an earlier spec in the full suite. A fresh worktree running just these two + files has never had one written, so the test dies on `ENOENT` before it + touches any code. **Confirmed by re-running the same two spec files at + `7f86aef` in the same worktree: the same spec fails the same way.** Worth + fixing independently — the spec should seed the file rather than assume a + predecessor left one — but it is not this slice's. + +#### What this slice deliberately leaves for S2c + +A hub caching a member's `/duplicates.json` or `/digests/*` still depends on +those paths being in the composed `_headers` CORS set, which they are not. +S2c's `_headers` commit is what makes the widened `sw-hub.js` reach them +cross-origin; same-origin (the site SW) works today. + +The offline DOWNLOAD path has no e2e coverage and gains none here: the service +worker registers in production builds only, so `pwa.spec` can assert the control +renders but never that a download lands. The guard that the SW lists and the +download list agree with the contract is `contract.test.ts`, verified by +mutation (dropping `digests` from sw-hub's `SHARD_RE` turns it red). + +#### Review fixes — `9451be8`, `5ac97b3` + +Three items came back. One was a real defect and it is worth recording as a +lesson about the shape of the thing, not just as a bug. + +**F1 — an offline copy is TWO lists, and shipping it as one put site-wide bytes +on a per-channel bill.** `downloadChannelOffline` appended every `ARCHIVE_TREES` +entry — the flat, site-wide `summaries` and `stats` included — plus all four +`ROOT_FILES` to EACH channel's list. Measured on jeralyzer: summaries ~14 MB, +stats ~21.6 MB (its `page-0000` alone is 20.97 MB), `/duplicates.json` 5.9 MB. +**~41.5 MB of byte-identical data per channel pinned**, genuinely re-fetched +because `cacheUrls` bulk-caches with `cache: "reload"`. + +The download was the smaller half. **None of it was removable.** Both evict +paths sweep `/<tree>/<slug>/` prefixes, and not one of those URLs lives under a +channel prefix — so "remove offline copy" left every byte in `PAGES` forever, +with no path in the app that could ever reclaim it. Enumerating the contract was +the right instinct and it was applied at the wrong granularity: the contract has +two kinds of tree, and the cost model follows that split exactly. + +So the lists split, in `common/lib/archive/offlineUrls.ts`: +`channelArchiveUrls` (the per-channel trees of one channel) and +`siteArchiveUrls` (the flat trees plus the root files of one origin). Site data +rides with the FIRST channel pinned on an origin and is skipped after; a new +`EVICT_SITE` message in both workers removes it when that origin's last pinned +channel goes. Downloaded once, evicted once. **The per-channel download is +byte-for-byte what it was.** + +"Already present?" is answered from the pinned registry rather than a new +status round-trip: the two lists are written and removed together, so one pinned +channel means one copy of the site data. That inherits the registry's existing +drift — a viewer who clears site data while keeping `localStorage` — which is +the same gap `channelCachedPages()` already exists for, and which a re-pin +repairs. A per-origin `SITE_STATUS` message would make it truthful; it did not +seem worth a second SW round-trip for a case that self-heals. + +The builders live in `common/` and not `export/` for a reason beyond tidiness: +**`export/` has no test runner**, and this split is precisely the thing that +needs one. Six cases, and the load-bearing ones are structural rather than +literal — the channel list contains no flat tree and no root file; every channel +URL matches `/<tree>/<slug>/`, the only shape either worker's per-channel sweep +can remove; the two lists are disjoint and together cover `ARCHIVE_TREES` +exactly. + +**F2 — a browser reader now gets a browser's budget.** `RemoteSource` defaults +to 48 MB per page cache and holds two, so the registry was handing every origin +96 MB: reasonable for one long-lived MCP process, not for a tab, and least of +all for a hub page holding a reader per member with no shared ceiling and no way +for a browser to reach the env knob. `readers.ts` now passes 8 MB explicitly — +16 MB per origin, so five federated members are 80 MB rather than 480 MB. It can +be this small at no cost in reads because the viewer memoises every RECORD it +has seen (`transcriptCache.resolved` plus IndexedDB), so an evicted page is only +re-read for a video nobody has opened yet; and `MIN_CACHED_PAGES` still floors +it at two entries where one page exceeds the whole budget. **`mcp/src/source.ts` +constructs `RemoteSource` directly and keeps the 48 MB default**, so no bench +counter moves. + +**F4 — `aliasesCache`'s comment now says what the code does.** It claimed +"404 → data" while the catch returns `[]` for any `ArchiveHttpError`. The code +is right (it is the old `if (!r.ok) return []`, and a server answering 500 +should cost the suggestion chip rather than the search); the comment was the +part that was wrong. + +Noted for the merger, no action taken here: `contract.ts`'s `manifestUrl` flat +branch now calls `treeManifestUrl`, and S2c edits nearby. + +##### Gates after the fixes + +- `pnpm -r exec tsc --noEmit` — clean in all six packages. +- common — **1101 passed / 0 failed** (1095 + 6 `offlineUrls.test.ts`). +- mcp — **205 passed / 0 failed**. `pnpm test:scripts` — **71 passed / 1 + skipped**. +- `pnpm --filter export exec next build` — **compiled successfully**, 11 static + pages. +- **e2e** re-run behind the queue lock: export `e2e` **172 passed / 0 failed**, + `e2e:hub` **5 passed / 0 failed** — unchanged from before the fixes, which is + the point: the per-channel download and every URL shape are what they were. + The editor pair and `e2e:2origin` were not re-run; nothing in these two + commits touches the editor, and 2origin is red at base.