commit 8a49d9059dc15d98b21d33903bc16644f4f01da8
parent d0e83fead98d60c2b03de5051bbd7c6c1272f658
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 12 Sep 2026 12:23:59 -0400
merge: one-core/phase-2-s2a — the viewer's archive walks become one reader per origin
S2a of Phase 2, reviewed and the offline-cache defect fixed: the eight
component caches and the viewer's contexts fetch through one RemoteSource per
origin (common/lib/archive/readers.ts, with a browser page-cache budget);
the reader gained raw throwing document reads underneath its tolerant
methods because react-query retries a rejection and never a null; the
offline copy downloads every layer the contract has, split into a
per-channel list and a once-per-origin site list that both service workers
can evict; both workers' shard regexes and eviction prefixes are pinned to
CONTRACT.layers by a test. Gates on the slice tip: tsc clean, common 1101,
mcp 205, scripts 71/1, export build clean, export e2e 172/172, hub 5/5.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
23 files changed, 1634 insertions(+), 318 deletions(-)
diff --git a/common/components/SearchDataContext.tsx b/common/components/SearchDataContext.tsx
@@ -20,11 +20,11 @@ import { useSubsManifest } from "./subsCache";
import { usePostsManifest } from "./postsCache";
import { useSearchAliases, fetchAliases } from "./aliasesCache";
import { mergeAliases, type SearchAlias } from "../lib/searchAliases";
-import { idBaseUrl, makeId } from "./originId";
+import { makeId } from "./originId";
import type { DisplaySummary } from "../lib/transcripts";
import type { Manifest, SubsManifest } from "../lib/manifest";
import type { PostsManifest } from "../lib/posts";
-import { pageFileName } from "../lib/manifest";
+import { readerFor } from "../lib/archive/readers";
import {
DEFAULT_GROUP_FALLBACK_ID,
FALLBACK_GROUP,
@@ -158,12 +158,6 @@ export function SingleSiteDataProvider({ children }: { children: ReactNode }) {
);
}
-async function fetchJson<T>(url: string): Promise<T> {
- const r = await fetch(url);
- if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`);
- return (await r.json()) as T;
-}
-
// A federated origin the hub reads from. `origin` is "" for the hub's own
// same-origin pool, else a full origin ("https://x.com"). `siteTitle` labels
// its channel group; `accent` is carried for provenance in the UI.
@@ -194,8 +188,7 @@ export function MultiSiteDataProvider({
const manifestQueries = useQueries({
queries: sites.map((s) => ({
queryKey: ["manifest", s.origin],
- queryFn: () =>
- fetchJson<Manifest>(`${idBaseUrl(s.origin)}/summaries/manifest.json`),
+ queryFn: () => readerFor(s.origin).readSummariesManifest(),
})),
});
const manifestsSettled = manifestQueries.every(
@@ -218,10 +211,7 @@ export function MultiSiteDataProvider({
const pageQueries = useQueries({
queries: pageDescriptors.map((d) => ({
queryKey: ["summaries-page", d.origin, d.index],
- queryFn: () =>
- fetchJson<DisplaySummary[]>(
- `${idBaseUrl(d.origin)}/summaries/${pageFileName(d.index)}`,
- ),
+ queryFn: () => readerFor(d.origin).readSummariesPage(d.index),
})),
});
@@ -300,8 +290,7 @@ export function MultiSiteDataProvider({
const subsQueries = useQueries({
queries: sites.map((s) => ({
queryKey: ["subs-manifest", s.origin],
- queryFn: () =>
- fetchJson<SubsManifest>(`${idBaseUrl(s.origin)}/subs/manifest.json`),
+ queryFn: () => readerFor(s.origin).readSubsSiteManifest(),
})),
});
const subsSettled = subsQueries.every((q) => q.isSuccess || q.isError);
@@ -339,9 +328,9 @@ export function MultiSiteDataProvider({
queries: sites.map((s) => ({
queryKey: ["posts-manifest", s.origin],
queryFn: () =>
- fetchJson<PostsManifest>(
- `${idBaseUrl(s.origin)}/posts/manifest.json`,
- ).catch(
+ readerFor(s.origin)
+ .readPostsSiteManifest()
+ .catch(
(): PostsManifest => ({
version: 0,
channels: [],
diff --git a/common/components/aliasesCache.ts b/common/components/aliasesCache.ts
@@ -7,15 +7,29 @@
// purely additive, so their absence must never break search.
import { useQuery } from "@tanstack/react-query";
-import { idBaseUrl } from "./originId";
+import { ArchiveHttpError } from "../lib/archive/reader";
+import { readerFor } from "../lib/archive/readers";
import { coerceAliasConfig, type SearchAlias } from "../lib/searchAliases";
const EMPTY: SearchAlias[] = [];
+// ANY status the archive itself returned resolves empty — a 404 because the
+// site ships no dictionary, but a 500 or a 403 the same way. That is the exact
+// behaviour of the `if (!r.ok) return []` this replaced, kept deliberately:
+// alias suggestions are additive, so a server that answers badly should cost
+// the chip, not the search.
+//
+// A TRANSPORT failure is different and is rethrown, because it is not an answer
+// at all. react-query then retries it rather than caching "this site has no
+// aliases" for the session — the query below has staleTime: Infinity, so a
+// swallowed blip would be permanent.
export async function fetchAliases(origin = ""): Promise<SearchAlias[]> {
- const r = await fetch(`${idBaseUrl(origin)}/search-aliases.json`);
- if (!r.ok) return [];
- return coerceAliasConfig(await r.json()).aliases;
+ try {
+ return coerceAliasConfig(await readerFor(origin).readAliasConfig()).aliases;
+ } catch (err) {
+ if (err instanceof ArchiveHttpError) return [];
+ throw err;
+ }
}
export function useSearchAliases(origin = ""): SearchAlias[] {
diff --git a/common/components/archiveCaches.test.ts b/common/components/archiveCaches.test.ts
@@ -0,0 +1,303 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { resetReaders } from "../lib/archive/readers";
+import { fetchTranscript } from "./transcriptCache";
+import { fetchSubs } from "./subsCache";
+import { fetchPost, fetchThread, peekPost } from "./postsCache";
+import { fetchDigest, hasDigest } from "./digestCache";
+import { fetchDuplicates, fetchDuplicateReport } from "./duplicatesCache";
+import { fetchAliases } from "./aliasesCache";
+
+// ─── the eight caches as WRAPPERS ───
+//
+// Each of these used to hand-roll the manifest -> slugToPage -> page walk; each
+// now calls the shared ArchiveReader. What must survive that move is the memo
+// behaviour, because it is what makes the viewer usable: a hit returns the same
+// promise (so N components asking for the same video do not each re-render on a
+// different tick) and a miss costs ONE read of each document, not one per
+// caller and not one per record.
+//
+// So every test below counts fetches. The caches hold module-level state with
+// no reset hook — that is deliberate, it is a page-lifetime cache — so each
+// test uses its own channel slug and its own origin rather than trying to wipe
+// them.
+
+type Archive = { fetches: string[]; body: Map<string, unknown> };
+
+function installFetch(entries: Record<string, unknown>): {
+ archive: Archive;
+ restore: () => void;
+} {
+ const archive: Archive = {
+ fetches: [],
+ body: new Map(Object.entries(entries)),
+ };
+ const real = globalThis.fetch;
+ globalThis.fetch = (async (input: RequestInfo | URL) => {
+ const url = String(input);
+ archive.fetches.push(url);
+ if (!archive.body.has(url)) {
+ return new Response("not found", { status: 404, statusText: "Not Found" });
+ }
+ return new Response(JSON.stringify(archive.body.get(url)), { status: 200 });
+ }) as typeof fetch;
+ return {
+ archive,
+ restore: () => {
+ globalThis.fetch = real;
+ resetReaders();
+ },
+ };
+}
+
+function count(archive: Archive, url: string): number {
+ return archive.fetches.filter((u) => u === url).length;
+}
+
+test("transcriptCache: one manifest + one page, and the whole page is warmed", async () => {
+ const { archive, restore } = installFetch({
+ "/transcripts/tc/manifest.json": {
+ version: 1,
+ slugToPage: { a: 0, b: 0 },
+ pageCount: 1,
+ generatedAt: "",
+ },
+ "/transcripts/tc/page-0000.json": [
+ { slug: "tc/a", id: "a", title: "A", cues: [] },
+ { slug: "tc/b", id: "b", title: "B", cues: [] },
+ ],
+ });
+ try {
+ // Two concurrent callers for the same id share one promise — not two reads
+ // that happen to agree.
+ const p1 = fetchTranscript("tc/a");
+ const p2 = fetchTranscript("tc/a");
+ assert.equal(p1, p2);
+ assert.equal((await p1).title, "A");
+
+ // A hit resolves without touching the network at all.
+ const after = archive.fetches.length;
+ assert.equal((await fetchTranscript("tc/a")).title, "A");
+ assert.equal(archive.fetches.length, after);
+
+ // And the OTHER record of the page we already paid for is a hit too: the
+ // opportunistic warming is what makes a search over one channel cheap.
+ assert.equal((await fetchTranscript("tc/b")).title, "B");
+ assert.equal(count(archive, "/transcripts/tc/manifest.json"), 1);
+ assert.equal(count(archive, "/transcripts/tc/page-0000.json"), 1);
+ assert.equal(archive.fetches.length, 2);
+ } finally {
+ restore();
+ }
+});
+
+test("transcriptCache: a video the manifest does not list never costs a page read", async () => {
+ const { archive, restore } = installFetch({
+ "/transcripts/tcmiss/manifest.json": {
+ version: 1,
+ slugToPage: {},
+ pageCount: 0,
+ generatedAt: "",
+ },
+ });
+ try {
+ await assert.rejects(() => fetchTranscript("tcmiss/nope"), /Unknown transcript slug/);
+ assert.deepEqual(archive.fetches, ["/transcripts/tcmiss/manifest.json"]);
+ } finally {
+ restore();
+ }
+});
+
+test("subsCache: same walk, same sharing, over the subs tree", async () => {
+ const { archive, restore } = installFetch({
+ "/subs/sc/manifest.json": {
+ version: 1,
+ slugToPage: { a: 0, b: 0 },
+ pageCount: 1,
+ generatedAt: "",
+ },
+ "/subs/sc/page-0000.json": [
+ { slug: "sc/a", id: "a", tracks: {} },
+ { slug: "sc/b", id: "b", tracks: {} },
+ ],
+ });
+ try {
+ const p1 = fetchSubs("sc/a");
+ assert.equal(fetchSubs("sc/a"), p1);
+ assert.equal((await p1).id, "a");
+ assert.equal((await fetchSubs("sc/b")).id, "b");
+ assert.deepEqual(archive.fetches, [
+ "/subs/sc/manifest.json",
+ "/subs/sc/page-0000.json",
+ ]);
+ } finally {
+ restore();
+ }
+});
+
+test("subsCache: an unreachable channel manifest is NOT memoised as absent", async () => {
+ const { archive, restore } = installFetch({});
+ try {
+ await assert.rejects(() => fetchSubs("scfail/a"));
+ await assert.rejects(() => fetchSubs("scfail/a"));
+ // Twice, because a failure is a failure and not an answer. The reader's own
+ // tolerant subsManifest() would have cached `null` here and the panel would
+ // stay empty for the session.
+ assert.equal(count(archive, "/subs/scfail/manifest.json"), 2);
+ } finally {
+ restore();
+ }
+});
+
+test("postsCache: a thread walks every page of the channel exactly once", async () => {
+ const { archive, restore } = installFetch({
+ "/posts/pc/manifest.json": {
+ version: 1,
+ slugToPage: { p1: 0, p2: 1 },
+ pageCount: 2,
+ generatedAt: "",
+ },
+ "/posts/pc/page-0000.json": [
+ { slug: "pc/p1", id: "p1", threadId: "p1", createdAt: "2026-01-01" },
+ ],
+ "/posts/pc/page-0001.json": [
+ { slug: "pc/p2", id: "p2", threadId: "p1", createdAt: "2026-01-02" },
+ ],
+ });
+ try {
+ assert.equal((await fetchPost("pc/p1")).id, "p1");
+ // Fetching the post warmed the page it came from, so peekPost is sync.
+ assert.equal(peekPost("pc/p1")?.id, "p1");
+
+ const thread = await fetchThread("pc/p1");
+ assert.deepEqual(
+ thread.map((t) => t.id),
+ ["p1", "p2"],
+ );
+ // A second thread read is free: the page memo is what stops fetchThread
+ // re-downloading the whole channel per post opened.
+ await fetchThread("pc/p1");
+ assert.equal(count(archive, "/posts/pc/manifest.json"), 1);
+ assert.equal(count(archive, "/posts/pc/page-0000.json"), 1);
+ assert.equal(count(archive, "/posts/pc/page-0001.json"), 1);
+ } finally {
+ restore();
+ }
+});
+
+test("digestCache: a channel with no digests costs ONE 404, ever", async () => {
+ const { archive, restore } = installFetch({});
+ try {
+ assert.equal(await hasDigest("dcnone/a"), false);
+ assert.equal(await fetchDigest("dcnone/a"), null);
+ assert.equal(await fetchDigest("dcnone/b"), null);
+ // The sparse layer's whole cost model: absence is an answer and is cached,
+ // so opening video after video in an undigested channel is free.
+ assert.equal(count(archive, "/digests/dcnone/manifest.json"), 1);
+ } finally {
+ restore();
+ }
+});
+
+test("digestCache: present, page-warmed, and not-in-slugToPage means null not throw", async () => {
+ const { archive, restore } = installFetch({
+ "/digests/dc/manifest.json": {
+ version: 1,
+ slugToPage: { a: 0, b: 0 },
+ pageCount: 1,
+ pageHashes: ["h0"],
+ generatedAt: "",
+ },
+ "/digests/dc/page-0000.json": [
+ { slug: "dc/a", id: "a", chapters: [] },
+ { slug: "dc/b", id: "b", chapters: [] },
+ ],
+ });
+ try {
+ assert.equal(await hasDigest("dc/a"), true);
+ assert.equal((await fetchDigest("dc/a"))?.id, "a");
+ assert.equal((await fetchDigest("dc/b"))?.id, "b");
+ // An undigested video in a digested channel: normal, and not an error.
+ assert.equal(await hasDigest("dc/c"), false);
+ assert.equal(await fetchDigest("dc/c"), null);
+ assert.equal(count(archive, "/digests/dc/manifest.json"), 1);
+ assert.equal(count(archive, "/digests/dc/page-0000.json"), 1);
+ } finally {
+ restore();
+ }
+});
+
+test("duplicatesCache: absent is empty, present is a slug-keyed lookup", async () => {
+ const absent = installFetch({});
+ try {
+ assert.equal(await fetchDuplicateReport("https://dup-absent.example"), null);
+ assert.equal((await fetchDuplicates("https://dup-absent.example")).size, 0);
+ } finally {
+ absent.restore();
+ }
+
+ const O = "https://dup-present.example";
+ const present = installFetch({
+ [`${O}/duplicates.json`]: {
+ clusters: [
+ {
+ clusterId: "c1",
+ videoRefs: [{ slug: "x/1" }, { slug: "y/2" }],
+ },
+ ],
+ },
+ });
+ try {
+ const lookup = await fetchDuplicates(O);
+ assert.equal(lookup.size, 2);
+ assert.equal(lookup.get("x/1")?.clusterId, "c1");
+ assert.equal(lookup.get("y/2")?.clusterId, "c1");
+ assert.equal(count(present.archive, `${O}/duplicates.json`), 1);
+ } finally {
+ present.restore();
+ }
+});
+
+test("aliasesCache: a 404 is an empty dictionary; a transport failure is rethrown", async () => {
+ const missing = installFetch({});
+ try {
+ assert.deepEqual(await fetchAliases("https://alias-404.example"), []);
+ } finally {
+ missing.restore();
+ }
+
+ const present = installFetch({
+ "https://alias-ok.example/search-aliases.json": {
+ aliases: [
+ {
+ id: "kcup",
+ label: "k cups",
+ triggers: ["k cups"],
+ suggestion: "(k|cake)[ -]?cup",
+ useRegex: true,
+ },
+ ],
+ },
+ });
+ try {
+ const aliases = await fetchAliases("https://alias-ok.example");
+ assert.equal(aliases.length, 1);
+ assert.equal(aliases[0].id, "kcup");
+ } finally {
+ present.restore();
+ }
+
+ // The distinction that matters: the query holding this has staleTime
+ // Infinity, so swallowing a blip would mean "this site has no aliases" for
+ // the rest of the session.
+ const real = globalThis.fetch;
+ globalThis.fetch = (async () => {
+ throw new TypeError("network down");
+ }) as typeof fetch;
+ try {
+ await assert.rejects(() => fetchAliases("https://alias-down.example"), /network down/);
+ } finally {
+ globalThis.fetch = real;
+ resetReaders();
+ }
+});
diff --git a/common/components/digestCache.ts b/common/components/digestCache.ts
@@ -1,9 +1,11 @@
"use client";
import type { ChannelDigestsManifest, VideoDigest } from "../lib/digests";
-import { digestPageFileName, manifestHasDigest } from "../lib/digests";
+import { manifestHasDigest } from "../lib/digests";
+import { PromiseMap } from "../lib/archive/reader";
+import { channelRef, readerFor } from "../lib/archive/readers";
import { idbGet, idbPut } from "./digestStore";
-import { makeId, splitId, idBaseUrl } from "./originId";
+import { makeId, splitId } from "./originId";
// Fetch layer for the derived corpus, mirroring transcriptCache.ts: a memory
// map, in-flight dedupe, manifest -> page resolution and opportunistic warming
@@ -23,11 +25,12 @@ import { makeId, splitId, idBaseUrl } from "./originId";
const resolved = new Map<string, VideoDigest | null>();
const inFlight = new Map<string, Promise<VideoDigest | null>>();
-const channelManifests = new Map<
- string,
- Promise<ChannelDigestsManifest | null>
->();
-const pagePromises = new Map<string, Promise<VideoDigest[]>>();
+// PromiseMap memoises a RESOLVED value (including `null` — the cached negative
+// this layer depends on) and drops a REJECTION, which is exactly the policy
+// this file hand-rolled: absence is an answer worth keeping, a transport
+// failure is not proof of absence.
+const channelManifests = new PromiseMap<ChannelDigestsManifest | null>();
+const pagePromises = new PromiseMap<VideoDigest[]>();
// `id` is an OriginId: a bare "channelSlug/videoId" for same-origin content, or
// "origin\tchannelSlug/videoId" for a federated cross-origin video.
@@ -87,26 +90,9 @@ function fetchChannelManifest(
channelSlug: string,
origin: string,
): Promise<ChannelDigestsManifest | null> {
- const key = makeId(origin, channelSlug);
- let p = channelManifests.get(key);
- if (!p) {
- p = fetch(`${idBaseUrl(origin)}/digests/${channelSlug}/manifest.json`)
- .then((r) => {
- if (r.status === 404) return null;
- if (!r.ok) {
- throw new Error(`Failed to fetch digests manifest for ${channelSlug}`);
- }
- return r.json() as Promise<ChannelDigestsManifest>;
- })
- .catch((err) => {
- // A transport failure is not proof of absence, so it must not be
- // cached as one — drop the entry and let the next caller retry.
- channelManifests.delete(key);
- throw err;
- });
- channelManifests.set(key, p);
- }
- return p;
+ return channelManifests.take(makeId(origin, channelSlug), () =>
+ readerFor(origin).readChannelDigestsManifest(channelSlug),
+ );
}
function fetchPage(
@@ -114,21 +100,9 @@ function fetchPage(
pageIndex: number,
origin: string,
): Promise<VideoDigest[]> {
- const key = `${makeId(origin, channelSlug)}:${pageIndex}`;
- let p = pagePromises.get(key);
- if (!p) {
- p = fetch(
- `${idBaseUrl(origin)}/digests/${channelSlug}/${digestPageFileName(pageIndex)}`,
- ).then((r) => {
- if (!r.ok) {
- throw new Error(`Failed to fetch digest page ${channelSlug}/${pageIndex}`);
- }
- return r.json() as Promise<VideoDigest[]>;
- });
- p.catch(() => pagePromises.delete(key));
- pagePromises.set(key, p);
- }
- return p;
+ return pagePromises.take(`${makeId(origin, channelSlug)}:${pageIndex}`, () =>
+ readerFor(origin).digestPage(channelRef(channelSlug, origin), pageIndex),
+ );
}
async function load(id: string): Promise<VideoDigest | null> {
diff --git a/common/components/duplicatesCache.ts b/common/components/duplicatesCache.ts
@@ -16,7 +16,7 @@
// per member is `aligned` — see duplicateSiblings below.
import { useQuery } from "@tanstack/react-query";
-import { idBaseUrl } from "./originId";
+import { readerFor } from "../lib/archive/readers";
import type {
DuplicateCluster,
DuplicateReport,
@@ -27,15 +27,23 @@ export type DuplicateLookup = ReadonlyMap<string, DuplicateCluster>;
const EMPTY: DuplicateLookup = new Map();
-export async function fetchDuplicates(origin = ""): Promise<DuplicateLookup> {
- let report: DuplicateReport | null = null;
+// The raw shipped report, or null when the site ships none (the common case —
+// compose-site writes the file only for a site with at least one publishable
+// cluster). Absent, unreachable and unparseable all fold to null: every
+// duplicate affordance is additive, so none of them may break a page.
+export async function fetchDuplicateReport(
+ origin = "",
+): Promise<DuplicateReport | null> {
try {
- const r = await fetch(`${idBaseUrl(origin)}/duplicates.json`);
- if (!r.ok) return EMPTY;
- report = (await r.json()) as DuplicateReport;
+ return await readerFor(origin).readDuplicates();
} catch {
- return EMPTY;
+ return null;
}
+}
+
+export async function fetchDuplicates(origin = ""): Promise<DuplicateLookup> {
+ const report = await fetchDuplicateReport(origin);
+ if (!report) return EMPTY;
const out = new Map<string, DuplicateCluster>();
for (const cluster of report?.clusters ?? []) {
for (const ref of cluster?.videoRefs ?? []) {
diff --git a/common/components/postsCache.ts b/common/components/postsCache.ts
@@ -5,24 +5,21 @@
// federating hub can hold several origins' posts at once without collisions.
import { useQueries, useQuery } from "@tanstack/react-query";
-import {
- postsPageFileName,
- type ChannelPostsManifest,
- type Post,
- type PostsManifest,
-} from "../lib/posts";
-import { makeId, splitId, idBaseUrl } from "./originId";
-
-async function fetchJson<T>(url: string): Promise<T> {
- const r = await fetch(url);
- if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`);
- return (await r.json()) as T;
-}
+import type { ChannelPostsManifest, Post, PostsManifest } from "../lib/posts";
+import { PromiseMap } from "../lib/archive/reader";
+import { channelRef, readerFor } from "../lib/archive/readers";
+import { makeId, splitId } from "./originId";
const resolved = new Map<string, Post>();
const inFlight = new Map<string, Promise<Post>>();
-const channelManifests = new Map<string, Promise<ChannelPostsManifest>>();
-const pagePromises = new Map<string, Promise<Post[]>>();
+// Two memos the reader does NOT provide, for two different reasons. The
+// manifest one wants a THROW where the reader's tolerant `postsManifest` wants
+// a null (see subsCache). The page one exists because the reader caches subs
+// and transcript pages but not posts pages, and fetchThread below walks EVERY
+// page of a channel — without a memo, opening two posts in a thread would
+// re-download the whole channel.
+const channelManifests = new PromiseMap<ChannelPostsManifest>();
+const pagePromises = new PromiseMap<Post[]>();
export type PostsManifestRef = { channelSlug: string; origin?: string };
@@ -53,16 +50,9 @@ export function fetchChannelPostsManifest(
channelSlug: string,
origin = "",
): Promise<ChannelPostsManifest> {
- const key = makeId(origin, channelSlug);
- let p = channelManifests.get(key);
- if (!p) {
- p = fetchJson<ChannelPostsManifest>(
- `${idBaseUrl(origin)}/posts/${channelSlug}/manifest.json`,
- );
- p.catch(() => channelManifests.delete(key));
- channelManifests.set(key, p);
- }
- return p;
+ return channelManifests.take(makeId(origin, channelSlug), () =>
+ readerFor(origin).readChannelPostsManifest(channelSlug),
+ );
}
export function fetchPostsPage(
@@ -70,23 +60,21 @@ export function fetchPostsPage(
pageIndex: number,
origin = "",
): Promise<Post[]> {
- const key = `${makeId(origin, channelSlug)}:${pageIndex}`;
- let p = pagePromises.get(key);
- if (!p) {
- p = fetchJson<Post[]>(
- `${idBaseUrl(origin)}/posts/${channelSlug}/${postsPageFileName(pageIndex)}`,
- ).then((page) => {
+ return pagePromises.take(
+ `${makeId(origin, channelSlug)}:${pageIndex}`,
+ async () => {
+ const page = await readerFor(origin).postsPage(
+ channelRef(channelSlug, origin),
+ pageIndex,
+ );
// Warm the by-id cache so a later fetchPost() for any post on this page
// is synchronous.
for (const entry of page) {
resolved.set(makeId(origin, entry.slug), entry);
}
return page;
- });
- p.catch(() => pagePromises.delete(key));
- pagePromises.set(key, p);
- }
- return p;
+ },
+ );
}
async function load(id: string): Promise<Post> {
@@ -136,7 +124,8 @@ export function usePostsManifest(origin = "") {
return useQuery<PostsManifest>({
queryKey: ["posts-manifest", origin],
queryFn: () =>
- fetchJson<PostsManifest>(`${idBaseUrl(origin)}/posts/manifest.json`)
+ readerFor(origin)
+ .readPostsSiteManifest()
// A site with no social channels ships no posts manifest; treat that as
// an empty corpus rather than an error, so the UI degrades quietly.
.catch(
diff --git a/common/components/siteRegistry.ts b/common/components/siteRegistry.ts
@@ -19,6 +19,7 @@
// and stay in sync when a site is added or removed.
import { useEffect, useSyncExternalStore } from "react";
+import { manifestUrl, rootFileUrl } from "../lib/archive/contract";
import { SITE_DESCRIPTOR_VERSION } from "../lib/siteDescriptor";
import type { PublicSiteDescriptor } from "../lib/siteDescriptor";
@@ -259,7 +260,7 @@ export async function validateSite(input: string): Promise<ValidateResult> {
let descriptor: PublicSiteDescriptor;
try {
- const res = await fetch(`${origin}/site.json`);
+ const res = await fetch(rootFileUrl("site.json", origin));
if (!res.ok) return { ok: false, reason: UNREADABLE };
descriptor = (await res.json()) as PublicSiteDescriptor;
} catch {
@@ -284,7 +285,7 @@ export async function validateSite(input: string): Promise<ValidateResult> {
// whose descriptor is fine but whose data isn't reachable/valid would fail
// silently later, so reject it now with a clear message.
try {
- const res = await fetch(`${origin}/summaries/manifest.json`);
+ const res = await fetch(manifestUrl("summaries", undefined, origin));
if (!res.ok) return { ok: false, reason: UNREADABLE };
const manifest = (await res.json()) as { version?: unknown; channels?: unknown };
if (
diff --git a/common/components/statsCache.ts b/common/components/statsCache.ts
@@ -3,14 +3,7 @@
import { useMemo } from "react";
import { useQueries, useQuery } from "@tanstack/react-query";
import type { StatsManifest, VideoStat } from "../lib/stats";
-import { statsPageFileName } from "../lib/stats";
-import { idBaseUrl } from "./originId";
-
-async function fetchJson<T>(url: string): Promise<T> {
- const r = await fetch(url);
- if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`);
- return (await r.json()) as T;
-}
+import { readerFor } from "../lib/archive/readers";
export type StatsState = {
manifest: StatsManifest | null;
@@ -24,8 +17,7 @@ export type StatsState = {
export function useStatsManifest(origin = "") {
return useQuery<StatsManifest>({
queryKey: ["stats-manifest", origin],
- queryFn: () =>
- fetchJson<StatsManifest>(`${idBaseUrl(origin)}/stats/manifest.json`),
+ queryFn: () => readerFor(origin).readStatsManifest(),
});
}
@@ -39,10 +31,7 @@ export function useStats(origin = ""): StatsState {
const pageQueries = useQueries({
queries: Array.from({ length: pageCount }, (_, i) => ({
queryKey: ["stats-page", origin, i],
- queryFn: () =>
- fetchJson<VideoStat[]>(
- `${idBaseUrl(origin)}/stats/${statsPageFileName(i)}`,
- ),
+ queryFn: () => readerFor(origin).readStatsPage(i),
enabled: pageCount > 0,
})),
});
diff --git a/common/components/subsCache.ts b/common/components/subsCache.ts
@@ -3,19 +3,17 @@
import { useQueries, useQuery } from "@tanstack/react-query";
import type { SubsDetail } from "../lib/subs";
import type { ChannelSubsManifest, SubsManifest } from "../lib/manifest";
-import { pageFileName } from "../lib/manifest";
-import { makeId, splitId, idBaseUrl } from "./originId";
-
-async function fetchJson<T>(url: string): Promise<T> {
- const r = await fetch(url);
- if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`);
- return (await r.json()) as T;
-}
+import { PromiseMap } from "../lib/archive/reader";
+import { channelRef, readerFor } from "../lib/archive/readers";
+import { makeId, splitId } from "./originId";
const resolved = new Map<string, SubsDetail>();
const inFlight = new Map<string, Promise<SubsDetail>>();
-const channelManifests = new Map<string, Promise<ChannelSubsManifest>>();
-const pagePromises = new Map<string, Promise<SubsDetail[]>>();
+// The reader has a per-channel subs-manifest cache too, but its answer for an
+// unreachable manifest is `null` (the right answer for a scanner probing thirty
+// channels, the wrong one for a UI that wants react-query to retry). So the
+// memo with the THROWING policy stays here, over the reader's raw read.
+const channelManifests = new PromiseMap<ChannelSubsManifest>();
// A per-channel subs reference: channel slug + the origin it lives on
// ("" = same-origin). Used by the multi-origin manifest hook.
@@ -42,33 +40,9 @@ function fetchChannelSubsManifest(
channelSlug: string,
origin = "",
): Promise<ChannelSubsManifest> {
- const key = makeId(origin, channelSlug);
- let p = channelManifests.get(key);
- if (!p) {
- p = fetchJson<ChannelSubsManifest>(
- `${idBaseUrl(origin)}/subs/${channelSlug}/manifest.json`,
- );
- p.catch(() => channelManifests.delete(key));
- channelManifests.set(key, p);
- }
- return p;
-}
-
-function fetchPage(
- channelSlug: string,
- pageIndex: number,
- origin = "",
-): Promise<SubsDetail[]> {
- const key = `${makeId(origin, channelSlug)}:${pageIndex}`;
- let p = pagePromises.get(key);
- if (!p) {
- p = fetchJson<SubsDetail[]>(
- `${idBaseUrl(origin)}/subs/${channelSlug}/${pageFileName(pageIndex)}`,
- );
- p.catch(() => pagePromises.delete(key));
- pagePromises.set(key, p);
- }
- return p;
+ return channelManifests.take(makeId(origin, channelSlug), () =>
+ readerFor(origin).readChannelSubsManifest(channelSlug),
+ );
}
async function load(id: string): Promise<SubsDetail> {
@@ -81,7 +55,10 @@ async function load(id: string): Promise<SubsDetail> {
const pageIndex = manifest.slugToPage[videoId];
if (pageIndex === undefined)
throw new Error(`Unknown subs slug: ${slug}`);
- const page = await fetchPage(channelSlug, pageIndex, origin);
+ const page = await readerFor(origin).subsPage(
+ channelRef(channelSlug, origin),
+ pageIndex,
+ );
let found: SubsDetail | undefined;
for (const entry of page) {
const entryId = makeId(origin, entry.slug);
@@ -95,7 +72,7 @@ async function load(id: string): Promise<SubsDetail> {
export function useSubsManifest(origin = "") {
return useQuery<SubsManifest>({
queryKey: ["subs-manifest", origin],
- queryFn: () => fetchJson<SubsManifest>(`${idBaseUrl(origin)}/subs/manifest.json`),
+ queryFn: () => readerFor(origin).readSubsSiteManifest(),
});
}
diff --git a/common/components/summariesCache.ts b/common/components/summariesCache.ts
@@ -4,14 +4,7 @@ import { useMemo } from "react";
import { useQueries, useQuery } from "@tanstack/react-query";
import type { DisplaySummary } from "../lib/transcripts";
import type { Manifest } from "../lib/manifest";
-import { pageFileName } from "../lib/manifest";
-import { idBaseUrl } from "./originId";
-
-async function fetchJson<T>(url: string): Promise<T> {
- const r = await fetch(url);
- if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`);
- return (await r.json()) as T;
-}
+import { readerFor } from "../lib/archive/readers";
export type SummariesState = {
manifest: Manifest | null;
@@ -25,8 +18,7 @@ export type SummariesState = {
export function useManifest(origin = "") {
return useQuery<Manifest>({
queryKey: ["manifest", origin],
- queryFn: () =>
- fetchJson<Manifest>(`${idBaseUrl(origin)}/summaries/manifest.json`),
+ queryFn: () => readerFor(origin).readSummariesManifest(),
});
}
@@ -38,10 +30,7 @@ export function useSummaries(origin = ""): SummariesState {
const pageQueries = useQueries({
queries: Array.from({ length: pageCount }, (_, i) => ({
queryKey: ["summaries-page", origin, i],
- queryFn: () =>
- fetchJson<DisplaySummary[]>(
- `${idBaseUrl(origin)}/summaries/${pageFileName(i)}`,
- ),
+ queryFn: () => readerFor(origin).readSummariesPage(i),
enabled: pageCount > 0,
})),
});
diff --git a/common/components/transcriptCache.ts b/common/components/transcriptCache.ts
@@ -1,18 +1,17 @@
"use client";
import type { TranscriptDetail } from "../lib/transcripts";
-import type { ChannelTranscriptsManifest } from "../lib/manifest";
-import { pageFileName } from "../lib/manifest";
+import { channelRef, readerFor } from "../lib/archive/readers";
import { idbGet, idbPutBatch } from "./transcriptStore";
-import { makeId, splitId, idBaseUrl } from "./originId";
+import { makeId, splitId } from "./originId";
+// The manifest -> slugToPage -> page walk is the ArchiveReader's
+// (lib/archive/reader.ts), including the per-channel manifest PromiseMap and
+// the byte-budgeted page LRU. What stays here is what is genuinely the
+// viewer's: the per-id memo, the in-flight dedupe, and warming IndexedDB with
+// every record of a page we already paid for.
const resolved = new Map<string, TranscriptDetail>();
const inFlight = new Map<string, Promise<TranscriptDetail>>();
-const channelManifests = new Map<
- string,
- Promise<ChannelTranscriptsManifest>
->();
-const pagePromises = new Map<string, Promise<TranscriptDetail[]>>();
// `id` is an OriginId: a bare "channelSlug/videoId" for same-origin content, or
// "origin\tchannelSlug/videoId" for a federated cross-origin video. Maps and
@@ -35,49 +34,6 @@ export function fetchTranscript(id: string): Promise<TranscriptDetail> {
return p;
}
-function fetchChannelManifest(
- channelSlug: string,
- origin: string,
-): Promise<ChannelTranscriptsManifest> {
- const key = makeId(origin, channelSlug);
- let p = channelManifests.get(key);
- if (!p) {
- p = fetch(`${idBaseUrl(origin)}/transcripts/${channelSlug}/manifest.json`).then((r) => {
- if (!r.ok)
- throw new Error(
- `Failed to fetch transcripts manifest for ${channelSlug}`,
- );
- return r.json() as Promise<ChannelTranscriptsManifest>;
- });
- p.catch(() => channelManifests.delete(key));
- channelManifests.set(key, p);
- }
- return p;
-}
-
-function fetchPage(
- channelSlug: string,
- pageIndex: number,
- origin: string,
-): Promise<TranscriptDetail[]> {
- const key = `${makeId(origin, channelSlug)}:${pageIndex}`;
- let p = pagePromises.get(key);
- if (!p) {
- p = fetch(
- `${idBaseUrl(origin)}/transcripts/${channelSlug}/${pageFileName(pageIndex)}`,
- ).then((r) => {
- if (!r.ok)
- throw new Error(
- `Failed to fetch transcript page ${channelSlug}/${pageIndex}`,
- );
- return r.json() as Promise<TranscriptDetail[]>;
- });
- p.catch(() => pagePromises.delete(key));
- pagePromises.set(key, p);
- }
- return p;
-}
-
async function load(id: string): Promise<TranscriptDetail> {
const stored = await idbGet(id);
if (stored) return stored;
@@ -86,11 +42,13 @@ async function load(id: string): Promise<TranscriptDetail> {
if (slashIdx < 0) throw new Error(`Malformed transcript slug: ${slug}`);
const channelSlug = slug.slice(0, slashIdx);
const videoId = slug.slice(slashIdx + 1);
- const manifest = await fetchChannelManifest(channelSlug, origin);
+ const reader = readerFor(origin);
+ const ch = channelRef(channelSlug, origin);
+ const manifest = await reader.transcriptsManifest(ch);
const pageIndex = manifest.slugToPage[videoId];
if (pageIndex === undefined)
throw new Error(`Unknown transcript slug: ${slug}`);
- const page = await fetchPage(channelSlug, pageIndex, origin);
+ const page = await reader.transcriptPage(ch, pageIndex);
let found: TranscriptDetail | undefined;
for (const entry of page) {
// Re-key warmed entries by their OriginId so a cross-origin
diff --git a/common/lib/archive/contract.test.ts b/common/lib/archive/contract.test.ts
@@ -1,7 +1,12 @@
import { test } from "node:test";
import assert from "node:assert/strict";
+import { readFileSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
import {
+ ARCHIVE_TREES,
CONTRACT,
+ PER_CHANNEL_TREES,
ROOT_FILES,
archiveUrl,
corpusUrl,
@@ -15,6 +20,8 @@ import {
import { buildSiteCorpus } from "../corpus";
import type { PublicSiteDescriptor } from "../siteDescriptor";
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+
// A minimal descriptor with two channels, one of which ships posts and digests,
// so every conditional manifest pointer buildSiteCorpus can emit is exercised.
function descriptor(siteUrl?: string): PublicSiteDescriptor {
@@ -166,3 +173,88 @@ test("CONTRACT is frozen where it is published", () => {
["transcripts", "subs", "posts", "digests", "summaries"],
);
});
+
+// ─── The three hand-written copies of the contract, pinned ───
+//
+// Three files legitimately cannot import this module and re-spell part of it by
+// hand: the two service workers (a service worker has no module graph to import
+// through — it is fetched and evaluated standalone) and the search index worker
+// (a classic Worker bundle). Each copy is silent when it drifts: a layer the SW
+// does not match is simply never cached, and a page file name without the
+// padding 404s against a real site while passing every unit test that made up
+// its own fixture names.
+//
+// So the copies are read off disk and compared with the contract. This is the
+// guard that lets CONTRACT.layers grow: add a layer, and these fail by name.
+
+const REPO = path.resolve(HERE, "..", "..", "..");
+
+function readSource(rel: string): string {
+ return readFileSync(path.join(REPO, rel), "utf8");
+}
+
+// Pull `const <NAME> = /…/;` out of a service worker and return the alternatives
+// of its first (…) group, un-escaped. Throws rather than returning [] when the
+// constant is missing, so a rename cannot make this test vacuously pass.
+function alternatives(src: string, name: string): string[] {
+ const decl = new RegExp(`const ${name} = /(.+)/;`).exec(src);
+ assert.ok(decl, `${name} not found — did it move or get renamed?`);
+ const group = /\(([^)]+)\)/.exec(decl[1]);
+ assert.ok(group, `${name} has no alternation group`);
+ return group[1].split("|").map((a) => a.replace(/\\(.)/g, "$1"));
+}
+
+// The `/<tree>/${slug}/` entries of an eviction prefix list.
+function evictedTrees(src: string): string[] {
+ const block = /const prefixes = \[([\s\S]*?)\]/.exec(src);
+ assert.ok(block, "eviction prefix list not found");
+ return [...block[1].matchAll(/`\/([^/]+)\/\$\{slug\}\/`/g)].map((m) => m[1]);
+}
+
+for (const sw of ["export/service-worker/site-sw.js", "export/service-worker/sw-hub.js"]) {
+ test(`${sw} matches exactly the contract's trees and root files`, () => {
+ const src = readSource(sw);
+
+ // Per-channel and flat, split the way the URL shapes are split — and
+ // together exactly ARCHIVE_TREES, so a layer cannot be quietly dropped from
+ // one family and "found" in the other.
+ assert.deepEqual(
+ alternatives(src, "SHARD_RE").sort(),
+ [...PER_CHANNEL_TREES].sort(),
+ );
+ assert.deepEqual(
+ alternatives(src, "FLAT_RE").sort(),
+ ARCHIVE_TREES.filter(isFlatTree).sort(),
+ );
+ assert.deepEqual(
+ [...alternatives(src, "SHARD_RE"), ...alternatives(src, "FLAT_RE")].sort(),
+ [...ARCHIVE_TREES].sort(),
+ );
+
+ // The root documents, by name. ROOT_RE is a full-path match, so these are
+ // the file names with no extra path.
+ assert.deepEqual(alternatives(src, "ROOT_RE").sort(), [...ROOT_FILES].sort());
+
+ // Eviction is per channel, so it sweeps the per-channel trees and nothing
+ // else. A flat tree here would be a prefix that can never match.
+ assert.deepEqual(evictedTrees(src).sort(), [...PER_CHANNEL_TREES].sort());
+ });
+}
+
+test("the search index worker's private pageFileName matches CONTRACT.pagePad", () => {
+ // components/searchIndex.worker.ts keeps its own copy on purpose — a worker
+ // bundle cannot import this module. Extract the literal padding it uses and
+ // run the copy, so both the constant AND the produced name are pinned.
+ const src = readSource("common/components/searchIndex.worker.ts");
+ const fn = /function pageFileName\(index: number\): string \{\s*return `page-\$\{String\(index\)\.padStart\((\d+), "0"\)\}\.json`;\s*\}/.exec(
+ src,
+ );
+ assert.ok(fn, "searchIndex.worker.ts's pageFileName is not the shape this test pins");
+ assert.equal(Number(fn[1]), CONTRACT.pagePad);
+
+ // And the URLs it builds are the contract's, for the one tree it walks.
+ assert.ok(src.includes("`/transcripts/${slug}/manifest.json`"));
+ assert.equal(manifestUrl("transcripts", "alpha"), "/transcripts/alpha/manifest.json");
+ assert.ok(src.includes("`/transcripts/${slug}/${pageFileName(p)}`"));
+ assert.equal(pageUrl("transcripts", "alpha", 12), "/transcripts/alpha/page-0012.json");
+});
diff --git a/common/lib/archive/contract.ts b/common/lib/archive/contract.ts
@@ -58,6 +58,14 @@ export type ContractLayer = (typeof CONTRACT.layers)[number];
// builders take the wider ArchiveTree and CONTRACT.layers stays frozen.
export type ArchiveTree = ContractLayer | "stats";
+// Every served tree, contract layers plus the undocumented `stats`. This is the
+// list a cache walker enumerates (the offline cache, the two service workers);
+// CONTRACT.layers stays the frozen PUBLISHED list.
+export const ARCHIVE_TREES: readonly ArchiveTree[] = [
+ ...CONTRACT.layers,
+ "stats",
+];
+
// The trees with no per-channel level: one manifest at the tree root and pages
// beside it. Everything else is /<tree>/<slug>/….
const FLAT_TREES: ReadonlySet<string> = new Set(["summaries", "stats"]);
@@ -66,6 +74,23 @@ export function isFlatTree(tree: ArchiveTree): boolean {
return FLAT_TREES.has(tree);
}
+// The per-channel trees: /<tree>/<slug>/…. What a per-channel cache eviction
+// has to sweep, and the half of ARCHIVE_TREES that takes a slug.
+export const PER_CHANNEL_TREES: readonly ArchiveTree[] = ARCHIVE_TREES.filter(
+ (t) => !isFlatTree(t),
+);
+
+// A tree's ROOT manifest, /<tree>/manifest.json.
+//
+// Every tree ships one, and for a per-channel tree it is a DIFFERENT document
+// from the per-channel manifest: /subs/manifest.json is the site-level index of
+// which channels ship live chat (SubsManifest), while /subs/<slug>/manifest.json
+// is that channel's slugToPage. manifestUrl() below builds the second; this
+// builds the first, and for a flat tree the two are the same file.
+export function treeManifestUrl(tree: ArchiveTree, base?: string): string {
+ return archiveUrl(base, `/${tree}/manifest.json`);
+}
+
// THE page-shard file name, for every layer. `page-0.json` is a 404 on every
// published archive, so a copy that lost the padding would 404 silently against
// a real site and pass every unit test. lib/manifest.ts re-exports this (and
@@ -117,7 +142,7 @@ export function manifestUrl(
slug?: string,
base?: string,
): string {
- if (isFlatTree(tree)) return archiveUrl(base, `/${tree}/manifest.json`);
+ if (isFlatTree(tree)) return treeManifestUrl(tree, base);
if (!slug) throw new Error(`manifestUrl(${tree}) needs a channel slug`);
return archiveUrl(base, `/${tree}/${slug}/manifest.json`);
}
diff --git a/common/lib/archive/offlineUrls.test.ts b/common/lib/archive/offlineUrls.test.ts
@@ -0,0 +1,178 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { ARCHIVE_TREES, PER_CHANNEL_TREES, ROOT_FILES, isFlatTree } from "./contract";
+import {
+ channelArchiveUrls,
+ siteArchiveUrls,
+ type ManifestReader,
+} from "./offlineUrls";
+
+// THE SPLIT IS THE ASSERTION. An offline copy is two lists, and if a site-wide
+// document ever leaks back into the per-channel one it is not a style problem:
+// the per-channel list is paid per channel pinned, and the per-channel EVICT
+// sweep (a /<tree>/<slug>/ prefix match in both service workers) cannot remove
+// anything that is not under a channel prefix. So a flat-tree URL in the
+// channel list is bytes that are downloaded N times and never removable.
+//
+// Measured on jeralyzer when that was the shipped behaviour: summaries ~14 MB +
+// stats ~21.6 MB + duplicates 5.9 MB = ~41.5 MB per channel, re-fetched for
+// real (the worker bulk-caches with `cache: "reload"`).
+
+// A stub archive: manifests present at these URLs with these page counts,
+// everything else absent. Records the reads so "absent costs one probe" is
+// assertable.
+function reader(pageCounts: Record<string, number>): {
+ read: ManifestReader;
+ reads: string[];
+} {
+ const reads: string[] = [];
+ const read: ManifestReader = async (url) => {
+ reads.push(url);
+ return url in pageCounts ? { pageCount: pageCounts[url] } : null;
+ };
+ return { read, reads };
+}
+
+const FULL = {
+ "/transcripts/alpha/manifest.json": 2,
+ "/subs/alpha/manifest.json": 1,
+ "/posts/alpha/manifest.json": 1,
+ "/digests/alpha/manifest.json": 1,
+ "/summaries/manifest.json": 2,
+ "/stats/manifest.json": 1,
+};
+
+test("the channel list is the PER-CHANNEL trees and nothing else", async () => {
+ const { read } = reader(FULL);
+ const urls = await channelArchiveUrls(read, "", "alpha");
+
+ assert.deepEqual(urls, [
+ "/transcripts/alpha/manifest.json",
+ "/transcripts/alpha/page-0000.json",
+ "/transcripts/alpha/page-0001.json",
+ "/subs/alpha/manifest.json",
+ "/subs/alpha/page-0000.json",
+ "/posts/alpha/manifest.json",
+ "/posts/alpha/page-0000.json",
+ "/digests/alpha/manifest.json",
+ "/digests/alpha/page-0000.json",
+ ]);
+
+ // Stated structurally as well as literally, so adding a layer to the contract
+ // fails here rather than silently shipping a channel that is half offline.
+ const trees = new Set(urls.map((u) => u.split("/")[1]));
+ assert.deepEqual([...trees].sort(), [...PER_CHANNEL_TREES].sort());
+
+ // Not one site-wide byte. This is the regression the review caught.
+ for (const u of urls) {
+ const tree = u.split("/")[1];
+ assert.ok(
+ !ARCHIVE_TREES.some((t) => t === tree && isFlatTree(t)),
+ `flat tree ${tree} must not be in a per-channel download: ${u}`,
+ );
+ }
+ for (const file of ROOT_FILES) {
+ assert.ok(!urls.includes(`/${file}`), `${file} must not be per channel`);
+ }
+ // Every URL is under a /<tree>/<slug>/ prefix — the only shape the service
+ // workers' per-channel evict sweep can ever remove.
+ for (const u of urls) {
+ assert.match(u, /^\/[^/]+\/alpha\//, `${u} is not evictable per channel`);
+ }
+});
+
+test("the site list is the FLAT trees plus the root files, and nothing per-channel", async () => {
+ const { read } = reader(FULL);
+ const urls = await siteArchiveUrls(read, "");
+
+ assert.deepEqual(urls, [
+ "/summaries/manifest.json",
+ "/summaries/page-0000.json",
+ "/summaries/page-0001.json",
+ "/stats/manifest.json",
+ "/stats/page-0000.json",
+ "/corpus.json",
+ "/site.json",
+ "/search-aliases.json",
+ "/duplicates.json",
+ ]);
+
+ const flat = ARCHIVE_TREES.filter(isFlatTree);
+ for (const tree of flat) {
+ assert.ok(
+ urls.some((u) => u.startsWith(`/${tree}/`)),
+ `${tree} missing from the site list`,
+ );
+ }
+ for (const file of ROOT_FILES) assert.ok(urls.includes(`/${file}`));
+ // No channel slug anywhere: nothing here is paid per channel.
+ for (const u of urls) assert.doesNotMatch(u, /alpha/);
+});
+
+test("together the two lists cover every tree the contract defines, once", async () => {
+ const { read } = reader(FULL);
+ const all = [
+ ...(await channelArchiveUrls(read, "", "alpha")),
+ ...(await siteArchiveUrls(read, "")),
+ ];
+ assert.equal(new Set(all).size, all.length, "no URL is in both lists");
+
+ const trees = new Set(
+ all.filter((u) => u.split("/").length > 2).map((u) => u.split("/")[1]),
+ );
+ assert.deepEqual([...trees].sort(), [...ARCHIVE_TREES].sort());
+});
+
+test("a tree the site does not ship costs one probe and contributes nothing", async () => {
+ // Transcripts only — the shape of a site with no chat, no posts, no digests
+ // and no stats.
+ const { read, reads } = reader({ "/transcripts/alpha/manifest.json": 1 });
+
+ assert.deepEqual(await channelArchiveUrls(read, "", "alpha"), [
+ "/transcripts/alpha/manifest.json",
+ "/transcripts/alpha/page-0000.json",
+ ]);
+ // One probe per absent tree, not a retry storm.
+ assert.deepEqual(reads, [
+ "/transcripts/alpha/manifest.json",
+ "/subs/alpha/manifest.json",
+ "/posts/alpha/manifest.json",
+ "/digests/alpha/manifest.json",
+ ]);
+
+ // The root files are listed unprobed — absent ones are skipped by the worker.
+ assert.deepEqual(await siteArchiveUrls(read, ""), [
+ "/corpus.json",
+ "/site.json",
+ "/search-aliases.json",
+ "/duplicates.json",
+ ]);
+});
+
+test("no transcripts manifest means no download at all", async () => {
+ const { read } = reader({ "/subs/alpha/manifest.json": 1 });
+ // A channel whose transcripts are unreachable cannot be opened offline, so
+ // caching its live chat would be caching a dead end.
+ assert.deepEqual(await channelArchiveUrls(read, "", "alpha"), []);
+});
+
+test("a federated origin prefixes both lists and nothing else changes", async () => {
+ const O = "https://member.example";
+ const { read } = reader({
+ [`${O}/transcripts/alpha/manifest.json`]: 1,
+ [`${O}/summaries/manifest.json`]: 1,
+ });
+
+ assert.deepEqual(await channelArchiveUrls(read, O, "alpha"), [
+ `${O}/transcripts/alpha/manifest.json`,
+ `${O}/transcripts/alpha/page-0000.json`,
+ ]);
+ assert.deepEqual(await siteArchiveUrls(read, O), [
+ `${O}/summaries/manifest.json`,
+ `${O}/summaries/page-0000.json`,
+ `${O}/corpus.json`,
+ `${O}/site.json`,
+ `${O}/search-aliases.json`,
+ `${O}/duplicates.json`,
+ ]);
+});
diff --git a/common/lib/archive/offlineUrls.ts b/common/lib/archive/offlineUrls.ts
@@ -0,0 +1,102 @@
+// WHAT AN OFFLINE COPY OF AN ARCHIVE CONSISTS OF, split the way it is paid for.
+//
+// Two lists, and the split is the whole point of this module:
+//
+// channelArchiveUrls — the PER-CHANNEL trees of one channel. Pinning a second
+// channel genuinely costs this again, because none of it
+// is shared.
+// siteArchiveUrls — the SITE-WIDE documents of one origin: the flat trees
+// (summaries, stats) and the root files. Identical for
+// every channel of that origin, so it is downloaded once
+// and evicted when the last pinned channel goes.
+//
+// The first cut of this shipped as ONE list and it was a real cost, not a
+// tidiness point. Measured on jeralyzer: summaries ~14 MB, stats ~21.6 MB (its
+// page-0000 alone is 20.97 MB) and /duplicates.json 5.9 MB — ~41.5 MB of
+// byte-identical site data re-fetched per channel pinned (the service worker
+// bulk-fetches with `cache: "reload"`, so they really do go over the wire), and
+// the per-channel evict path sweeps only the per-channel prefixes, so removing
+// a channel left every byte of it behind forever.
+//
+// The manifest READER IS INJECTED. The caller owns the transport, which matters
+// here: the viewer reads these with `cache: "no-store"` precisely because a
+// normal fetch would be answered from the service-worker cache this list exists
+// to refill, and would then enumerate the stale copy. Injection also makes the
+// lists directly assertable without a network.
+
+import {
+ ARCHIVE_TREES,
+ PER_CHANNEL_TREES,
+ ROOT_FILES,
+ isFlatTree,
+ manifestUrl,
+ pageUrl,
+ rootFileUrl,
+} from "./contract";
+
+// Every manifest shape these walks need, reduced to the one field a URL list is
+// built from. A tree a site does not ship answers 404 and reads as absent.
+export type PagedManifest = { pageCount?: number };
+
+// Reads one manifest by URL. Resolves null for "this site does not ship it",
+// which for these lists is normal rather than an error.
+export type ManifestReader = (url: string) => Promise<PagedManifest | null>;
+
+// The manifest plus every page of one tree, or [] when the tree is absent.
+// `slug` is undefined for a flat tree.
+async function treeUrls(
+ read: ManifestReader,
+ base: string,
+ tree: (typeof ARCHIVE_TREES)[number],
+ slug: string | undefined,
+): Promise<string[]> {
+ const manifest = manifestUrl(tree, slug, base);
+ const m = await read(manifest);
+ if (!m) return [];
+ const urls = [manifest];
+ for (let p = 0; p < (m.pageCount ?? 0); p++) {
+ urls.push(pageUrl(tree, slug, p, base));
+ }
+ return urls;
+}
+
+// One channel's shards, across every per-channel tree the contract defines.
+//
+// Returns [] when the channel has no TRANSCRIPTS manifest — that is not a tree
+// a readable channel can be missing, so the caller refuses the download rather
+// than caching a channel that cannot be opened. Every other tree is optional
+// and costs one 404 when absent.
+export async function channelArchiveUrls(
+ read: ManifestReader,
+ base: string,
+ slug: string,
+): Promise<string[]> {
+ const transcripts = await treeUrls(read, base, "transcripts", slug);
+ if (transcripts.length === 0) return [];
+ const urls = [...transcripts];
+ for (const tree of PER_CHANNEL_TREES) {
+ if (tree === "transcripts") continue;
+ urls.push(...(await treeUrls(read, base, tree, slug)));
+ }
+ return urls;
+}
+
+// One origin's site-wide documents: the flat trees and the root files.
+//
+// The root files are listed unconditionally rather than probed. /duplicates.json
+// and /search-aliases.json are legitimately absent on many sites, the caller
+// hands the list to a worker that skips what it cannot fetch, and probing them
+// first would double the request count to learn something the fetch already
+// tells us.
+export async function siteArchiveUrls(
+ read: ManifestReader,
+ base: string,
+): Promise<string[]> {
+ const urls: string[] = [];
+ for (const tree of ARCHIVE_TREES) {
+ if (!isFlatTree(tree)) continue;
+ urls.push(...(await treeUrls(read, base, tree, undefined)));
+ }
+ for (const file of ROOT_FILES) urls.push(rootFileUrl(file, base));
+ return urls;
+}
diff --git a/common/lib/archive/reader.ts b/common/lib/archive/reader.ts
@@ -19,12 +19,13 @@
import {
type ChannelTranscriptsManifest,
type ChannelSubsManifest,
+ type SubsManifest,
type Manifest,
} from "../manifest";
import type { TranscriptDetail, DisplaySummary } from "../transcripts";
import { summaryState, type VideoState } from "../availability";
import type { SubsDetail } from "../subs";
-import type { ChannelPostsManifest, Post } from "../posts";
+import type { ChannelPostsManifest, Post, PostsManifest } from "../posts";
import { coerceAliasConfig, type SearchAlias } from "../searchAliases";
import {
resolveCanonicalSlug,
@@ -39,7 +40,13 @@ import {
DEFAULT_GROUP_FALLBACK_ID,
type ChannelGroup,
} from "../channelGroups";
-import { manifestUrl, pageUrl, rootFileUrl, type HubSite } from "./contract";
+import {
+ manifestUrl,
+ pageUrl,
+ rootFileUrl,
+ treeManifestUrl,
+ type HubSite,
+} from "./contract";
import { recordRead } from "./io-stats";
// A channel the source can serve. `siteId`/`siteUrl` are only populated in hub
@@ -373,6 +380,23 @@ export interface ArchiveReader {
digestPage?(ch: ChannelRef, page: number): Promise<VideoDigest[]>;
}
+// An HTTP STATUS the archive itself returned, as opposed to a transport
+// failure. A caller that treats absence as data — a site ships no
+// /search-aliases.json, a channel ships no digests — needs to tell the two
+// apart: a 404 is an answer and is worth caching, a dropped connection is
+// neither. `message` is byte-identical to the plain Error this replaced, so
+// anything that only reads the message is unaffected.
+export class ArchiveHttpError extends Error {
+ constructor(
+ readonly status: number,
+ readonly url: string,
+ statusText: string,
+ ) {
+ super(`GET ${url} -> ${status} ${statusText}`);
+ this.name = "ArchiveHttpError";
+ }
+}
+
// Used when a source states no preference. Deliberately modest: each in-flight
// page costs its raw bytes plus ~2.7× that once parsed, and this box is shared.
export const DEFAULT_PAGE_CONCURRENCY = 4;
@@ -590,8 +614,7 @@ export class RemoteSource implements ArchiveReader {
subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
return this.subsManifests.take(ch.slug, async () => {
try {
- const res = await fetch(manifestUrl("subs", ch.slug, this.base));
- return res.ok ? ((await res.json()) as ChannelSubsManifest) : null;
+ return await this.readChannelSubsManifest(ch.slug);
} catch {
return null;
}
@@ -609,8 +632,7 @@ export class RemoteSource implements ArchiveReader {
postsManifest(ch: ChannelRef): Promise<ChannelPostsManifest | null> {
return this.postsManifests.take(ch.slug, async () => {
try {
- const res = await fetch(manifestUrl("posts", ch.slug, this.base));
- return res.ok ? ((await res.json()) as ChannelPostsManifest) : null;
+ return await this.readChannelPostsManifest(ch.slug);
} catch {
return null;
}
@@ -622,15 +644,12 @@ export class RemoteSource implements ArchiveReader {
}
videoIndex(): Promise<VideoIndex> {
+ // buildVideoIndex already treats a throw from either reader as absence
+ // (empty index / skipped page), so the raw reads' rejections land exactly
+ // where the old inline `res.ok ? … : null` did.
this.index ??= buildVideoIndex(
- async () => {
- const res = await fetch(manifestUrl("summaries", undefined, this.base));
- return res.ok ? ((await res.json()) as Manifest) : null;
- },
- async (page) => {
- const res = await fetch(pageUrl("summaries", undefined, page, this.base));
- return res.ok ? ((await res.json()) as DisplaySummary[]) : null;
- },
+ () => this.readSummariesManifest(),
+ (page) => this.readSummariesPage(page),
);
return this.index;
}
@@ -642,8 +661,7 @@ export class RemoteSource implements ArchiveReader {
digestsManifest(ch: ChannelRef): Promise<ChannelDigestsManifest | null> {
return this.digestManifests.take(ch.slug, async () => {
try {
- const res = await fetch(manifestUrl("digests", ch.slug, this.base));
- return res.ok ? ((await res.json()) as ChannelDigestsManifest) : null;
+ return await this.readChannelDigestsManifest(ch.slug);
} catch {
return null;
}
@@ -657,10 +675,7 @@ export class RemoteSource implements ArchiveReader {
duplicateIndex(): Promise<DuplicateIndex> {
this.duplicates ??= (async () => {
try {
- const res = await fetch(rootFileUrl(DUPLICATES_FILENAME, this.base));
- return buildDuplicateIndex(
- res.ok ? ((await res.json()) as DuplicateReport) : null,
- );
+ return buildDuplicateIndex(await this.readDuplicates());
} catch {
return buildDuplicateIndex(null);
}
@@ -670,14 +685,8 @@ export class RemoteSource implements ArchiveReader {
statsIndex(): Promise<ReadonlyMap<string, VideoStat>> {
this.stats ??= buildStatsIndex(
- async () => {
- const res = await fetch(manifestUrl("stats", undefined, this.base));
- return res.ok ? ((await res.json()) as StatsManifest) : null;
- },
- async (page) => {
- const res = await fetch(pageUrl("stats", undefined, page, this.base));
- return res.ok ? ((await res.json()) as VideoStat[]) : null;
- },
+ () => this.readStatsManifest(),
+ (page) => this.readStatsPage(page),
);
return this.stats;
}
@@ -685,10 +694,7 @@ export class RemoteSource implements ArchiveReader {
async loadAliases(): Promise<SearchAlias[]> {
if (this.aliases) return this.aliases;
try {
- const res = await fetch(rootFileUrl("search-aliases.json", this.base));
- this.aliases = res.ok
- ? coerceAliasConfig(await res.json()).aliases
- : [];
+ this.aliases = coerceAliasConfig(await this.readAliasConfig()).aliases;
} catch {
this.aliases = [];
}
@@ -698,10 +704,7 @@ export class RemoteSource implements ArchiveReader {
async loadGroups(): Promise<ChannelGroups> {
if (this.groups) return this.groups;
try {
- const res = await fetch(manifestUrl("summaries", undefined, this.base));
- this.groups = res.ok
- ? parseGroupsManifest(await res.json())
- : EMPTY_GROUPS;
+ this.groups = parseGroupsManifest(await this.readSummariesManifest());
} catch {
this.groups = EMPTY_GROUPS;
}
@@ -713,23 +716,106 @@ export class RemoteSource implements ArchiveReader {
//
// `p` is a ROOT-RELATIVE path from the contract builders; the base is joined
// here so the thrown error names the full URL, exactly as before.
+ //
+ // `kind` is the io-stats bucket. It is OPTIONAL, and the reads that pass none
+ // are the ones that never recorded a read before this file owned them (the
+ // summaries/stats/duplicates/alias folds all used a bare `fetch`). Adding
+ // them to the ledger would move the bench's structural counters for a reason
+ // that has nothing to do with the walk, so the blind spots are preserved
+ // deliberately rather than quietly closed.
private async getJsonSized<T>(
p: string,
- kind: string,
+ kind?: string,
): Promise<{ value: T; bytes: number }> {
const res = await fetch(`${this.base}${p}`);
if (!res.ok) {
- throw new Error(`GET ${this.base}${p} -> ${res.status} ${res.statusText}`);
+ throw new ArchiveHttpError(res.status, `${this.base}${p}`, res.statusText);
}
const raw = await res.text();
- recordRead(kind, raw.length);
+ if (kind !== undefined) recordRead(kind, raw.length);
return { value: JSON.parse(raw) as T, bytes: raw.length };
}
- private async getJson<T>(p: string, kind = "json"): Promise<T> {
+ private async getJson<T>(p: string, kind?: string): Promise<T> {
return (await this.getJsonSized<T>(p, kind)).value;
}
+ // ─── Raw document reads ───
+ //
+ // One URL, one fetch, one parse, and a THROW on anything but a 200. Every
+ // tolerance policy is built on top of these — this class's own
+ // empty-when-absent folds below, and the viewer's component caches — so each
+ // published document's URL is written once and each caller picks its own
+ // answer to "absent".
+ //
+ // That split is not cosmetic: a tool scanning 30 channels wants `null` for a
+ // missing posts tree, while the viewer wants a rejection, because react-query
+ // retries a rejected query and will never retry a resolved `null`. Folding
+ // both into one tolerant method is how a transient network blip becomes a
+ // permanently empty panel.
+
+ readCorpus(): Promise<SiteCorpusJson> {
+ return this.getJson<SiteCorpusJson>(rootFileUrl("corpus.json"), "corpus");
+ }
+
+ readAliasConfig(): Promise<unknown> {
+ return this.getJson<unknown>(rootFileUrl("search-aliases.json"));
+ }
+
+ readDuplicates(): Promise<DuplicateReport> {
+ return this.getJson<DuplicateReport>(rootFileUrl(DUPLICATES_FILENAME));
+ }
+
+ readSummariesManifest(): Promise<Manifest> {
+ return this.getJson<Manifest>(manifestUrl("summaries"));
+ }
+
+ readSummariesPage(page: number): Promise<DisplaySummary[]> {
+ return this.getJson<DisplaySummary[]>(pageUrl("summaries", undefined, page));
+ }
+
+ readStatsManifest(): Promise<StatsManifest> {
+ return this.getJson<StatsManifest>(manifestUrl("stats"));
+ }
+
+ readStatsPage(page: number): Promise<VideoStat[]> {
+ return this.getJson<VideoStat[]>(pageUrl("stats", undefined, page));
+ }
+
+ // The SITE-level index of which channels ship a per-channel tree — a
+ // different document from a channel's own manifest (see treeManifestUrl).
+ readSubsSiteManifest(): Promise<SubsManifest> {
+ return this.getJson<SubsManifest>(treeManifestUrl("subs"));
+ }
+
+ readPostsSiteManifest(): Promise<PostsManifest> {
+ return this.getJson<PostsManifest>(treeManifestUrl("posts"));
+ }
+
+ readChannelSubsManifest(slug: string): Promise<ChannelSubsManifest> {
+ return this.getJson<ChannelSubsManifest>(manifestUrl("subs", slug));
+ }
+
+ readChannelPostsManifest(slug: string): Promise<ChannelPostsManifest> {
+ return this.getJson<ChannelPostsManifest>(manifestUrl("posts", slug));
+ }
+
+ // 404 → null, everything else → throw. The digest corpus is SPARSE BY DESIGN
+ // (a channel with no digests ships no manifest at all), so the 404 is the
+ // document's own answer and belongs in the raw read; a 500 or a dropped
+ // connection is not proof of absence and must stay distinguishable.
+ async readChannelDigestsManifest(
+ slug: string,
+ ): Promise<ChannelDigestsManifest | null> {
+ const p = manifestUrl("digests", slug);
+ const res = await fetch(`${this.base}${p}`);
+ if (res.status === 404) return null;
+ if (!res.ok) {
+ throw new ArchiveHttpError(res.status, `${this.base}${p}`, res.statusText);
+ }
+ return (await res.json()) as ChannelDigestsManifest;
+ }
+
private channelList?: Promise<ChannelRef[]>;
listChannels(opts: { refresh?: boolean } = {}): Promise<ChannelRef[]> {
@@ -745,10 +831,7 @@ export class RemoteSource implements ArchiveReader {
}
private async readChannels(): Promise<ChannelRef[]> {
- const corpus = await this.getJson<SiteCorpusJson>(
- rootFileUrl("corpus.json"),
- "corpus",
- );
+ const corpus = await this.readCorpus();
return (corpus.channels ?? []).map((c) => ({
key: c.slug,
slug: c.slug,
diff --git a/common/lib/archive/readers.test.ts b/common/lib/archive/readers.test.ts
@@ -0,0 +1,181 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { ArchiveHttpError, RemoteSource } from "./reader";
+import { channelRef, readerFor, resetReaders } from "./readers";
+
+// The same in-memory-archive trick as reader.test.ts: a Map from URL to JSON
+// body installed as `fetch`, so every assertion below is about the URLs the
+// reader actually asks for. The viewer's caches are thin wrappers over these
+// reads, so pinning the reads pins the wire.
+
+type Archive = { fetches: string[]; body: Map<string, unknown> };
+
+function installFetch(entries: Record<string, unknown>): {
+ archive: Archive;
+ restore: () => void;
+} {
+ const archive: Archive = {
+ fetches: [],
+ body: new Map(Object.entries(entries)),
+ };
+ const real = globalThis.fetch;
+ globalThis.fetch = (async (input: RequestInfo | URL) => {
+ const url = String(input);
+ archive.fetches.push(url);
+ if (!archive.body.has(url)) {
+ return new Response("not found", { status: 404, statusText: "Not Found" });
+ }
+ return new Response(JSON.stringify(archive.body.get(url)), { status: 200 });
+ }) as typeof fetch;
+ return {
+ archive,
+ restore: () => {
+ globalThis.fetch = real;
+ resetReaders();
+ },
+ };
+}
+
+test("readerFor: one reader per origin, kept for the life of the page", () => {
+ resetReaders();
+ const same = readerFor("");
+ assert.equal(readerFor(""), same);
+ assert.equal(readerFor(), same, "the default argument is the same-origin key");
+ const other = readerFor("https://b.example");
+ assert.notEqual(other, same);
+ assert.equal(readerFor("https://b.example"), other);
+ assert.ok(same instanceof RemoteSource);
+ resetReaders();
+ assert.notEqual(readerFor(""), same, "resetReaders drops the caches with it");
+});
+
+test("the same-origin reader emits ROOT-RELATIVE urls, byte-identical to the old walk", async () => {
+ const { archive, restore } = installFetch({
+ "/summaries/manifest.json": { version: 3, pageCount: 2, channels: [] },
+ "/summaries/page-0001.json": [],
+ "/stats/manifest.json": { version: 1, pageCount: 1, channels: [] },
+ "/stats/page-0000.json": [],
+ "/subs/manifest.json": { version: 4, channels: [] },
+ "/posts/manifest.json": { version: 1, channels: [] },
+ "/search-aliases.json": { aliases: [] },
+ "/duplicates.json": { clusters: [] },
+ "/transcripts/alpha/manifest.json": { slugToPage: {}, pageCount: 1 },
+ "/subs/alpha/manifest.json": { slugToPage: {}, pageCount: 1 },
+ "/posts/alpha/manifest.json": { slugToPage: {}, pageCount: 1 },
+ });
+ try {
+ const r = readerFor("");
+ await r.readSummariesManifest();
+ await r.readSummariesPage(1);
+ await r.readStatsManifest();
+ await r.readStatsPage(0);
+ await r.readSubsSiteManifest();
+ await r.readPostsSiteManifest();
+ await r.readAliasConfig();
+ await r.readDuplicates();
+ await r.readChannelSubsManifest("alpha");
+ await r.readChannelPostsManifest("alpha");
+ assert.deepEqual(archive.fetches, [
+ "/summaries/manifest.json",
+ "/summaries/page-0001.json",
+ "/stats/manifest.json",
+ "/stats/page-0000.json",
+ "/subs/manifest.json",
+ "/posts/manifest.json",
+ "/search-aliases.json",
+ "/duplicates.json",
+ "/subs/alpha/manifest.json",
+ "/posts/alpha/manifest.json",
+ ]);
+ } finally {
+ restore();
+ }
+});
+
+test("a federated origin prefixes every one of those, and nothing else changes", async () => {
+ const O = "https://member.example";
+ const { archive, restore } = installFetch({
+ [`${O}/summaries/manifest.json`]: { version: 3, pageCount: 0, channels: [] },
+ [`${O}/duplicates.json`]: { clusters: [] },
+ });
+ try {
+ const r = readerFor(O);
+ await r.readSummariesManifest();
+ await r.readDuplicates();
+ assert.deepEqual(archive.fetches, [
+ `${O}/summaries/manifest.json`,
+ `${O}/duplicates.json`,
+ ]);
+ } finally {
+ restore();
+ }
+});
+
+test("a raw read rejects with the status; the tolerant fold over it resolves empty", async () => {
+ const { restore } = installFetch({});
+ try {
+ const r = readerFor("");
+ await assert.rejects(
+ () => r.readDuplicates(),
+ (err: unknown) =>
+ err instanceof ArchiveHttpError &&
+ err.status === 404 &&
+ /GET \/duplicates\.json -> 404/.test((err as Error).message),
+ );
+ // Same document, same 404, through the fold the MCP uses.
+ assert.equal((await r.duplicateIndex()).size, 0);
+ // And through the summaries fold, which a partial/absent index must not
+ // turn into an error — the page planner reads empty as "scan anyway".
+ assert.equal((await r.videoIndex()).size, 0);
+ assert.deepEqual(await r.loadAliases(), []);
+ } finally {
+ restore();
+ }
+});
+
+test("a digests manifest 404 is the layer's own answer; any other status is not", async () => {
+ const { archive, restore } = installFetch({
+ "/digests/beta/manifest.json": { slugToPage: { v1: 0 }, pageCount: 1 },
+ });
+ try {
+ const r = readerFor("");
+ assert.equal(await r.readChannelDigestsManifest("alpha"), null);
+ assert.ok(await r.readChannelDigestsManifest("beta"));
+ assert.deepEqual(archive.fetches, [
+ "/digests/alpha/manifest.json",
+ "/digests/beta/manifest.json",
+ ]);
+ } finally {
+ restore();
+ }
+
+ const bad = installFetch({});
+ const realFetch = globalThis.fetch;
+ globalThis.fetch = (async (input: RequestInfo | URL) => {
+ void input;
+ return new Response("boom", { status: 500, statusText: "Server Error" });
+ }) as typeof fetch;
+ try {
+ await assert.rejects(
+ () => readerFor("").readChannelDigestsManifest("alpha"),
+ /-> 500/,
+ );
+ } finally {
+ globalThis.fetch = realFetch;
+ bad.restore();
+ }
+});
+
+test("channelRef: slug-addressed, and carries the origin only when there is one", () => {
+ assert.deepEqual(channelRef("alpha"), {
+ key: "alpha",
+ slug: "alpha",
+ name: "alpha",
+ });
+ assert.deepEqual(channelRef("alpha", "https://b.example"), {
+ key: "alpha",
+ slug: "alpha",
+ name: "alpha",
+ siteUrl: "https://b.example",
+ });
+});
diff --git a/common/lib/archive/readers.ts b/common/lib/archive/readers.ts
@@ -0,0 +1,73 @@
+// One ArchiveReader per origin, for the code that reads an archive from a
+// BROWSER.
+//
+// The viewer's component caches are keyed by origin ("" = same-origin, a full
+// origin like "https://x.example" for a federated hub member — see
+// components/originId.ts). Each one used to carry its own hand-written walk of
+// the shard scheme; now each one calls a RemoteSource from here, so the URL
+// shape is the contract's and the promise-coalescing/LRU behaviour is the
+// reader's rather than eight near-copies of it.
+//
+// WHY THIS LIVES IN lib/archive/ AND NOT components/: it is the reader
+// registry, not a React concern — no hooks, no JSX, no node imports — and a new
+// components/*.ts file would need its own `exports` entry in common's
+// package.json. lib/* is wildcarded, so this file costs nothing to reach from
+// editor/ or export/.
+//
+// RemoteSource with an EMPTY base is the same-origin case: archiveUrl() leaves a
+// path root-relative when there is no base, which is byte-for-byte the
+// `${idBaseUrl(origin)}/transcripts/…` the caches built before. So the
+// single-site export app's URLs do not move at all.
+
+import { RemoteSource, type ChannelRef } from "./reader";
+
+// A BROWSER'S page-cache budget, which is not a server's.
+//
+// RemoteSource defaults to 48 MB (pageCacheBudgetBytes) and holds TWO page
+// caches — transcripts and subs — so an origin costs 96 MB. That is a sane
+// ceiling for one long-lived MCP process reading one corpus. It is not one for
+// a tab, and it is emphatically not one for a hub page, which holds a reader
+// per member site with no shared ceiling and no way for a browser to turn the
+// env knob down.
+//
+// 8 MB, so an origin is 16 MB and five federated members are 80 MB rather than
+// 480 MB. The reason it can be this small without costing reads: the viewer's
+// caches memoise every RECORD they have ever seen (transcriptCache.resolved,
+// plus IndexedDB), so an evicted page is only re-fetched for a video nobody has
+// opened yet — the case that was going to be a fetch anyway. PageCache's
+// MIN_CACHED_PAGES floor still keeps two entries even when one page of a
+// VOD-sized corpus exceeds the whole budget on its own.
+//
+// The MCP path does NOT come through here: mcp/src/source.ts constructs
+// RemoteSource directly and keeps the 48 MB default, so no bench counter moves.
+const BROWSER_PAGE_CACHE_BYTES = 8 * 1024 * 1024;
+
+const byOrigin = new Map<string, RemoteSource>();
+
+// The reader for an origin, created on first use and kept for the life of the
+// page — the readers hold the manifest/page caches, so a fresh one per call
+// would be a fresh cache per call.
+export function readerFor(origin = ""): RemoteSource {
+ let reader = byOrigin.get(origin);
+ if (reader === undefined) {
+ reader = new RemoteSource(origin, BROWSER_PAGE_CACHE_BYTES);
+ byOrigin.set(origin, reader);
+ }
+ return reader;
+}
+
+// A minimal ChannelRef for the per-channel reader calls. The viewer addresses a
+// channel by slug alone; `key`/`name` exist for the MCP's channel list and are
+// not read on this path. Reader caches key off `slug`, so a fresh object per
+// call is free.
+export function channelRef(slug: string, origin = ""): ChannelRef {
+ return origin
+ ? { key: slug, slug, name: slug, siteUrl: origin }
+ : { key: slug, slug, name: slug };
+}
+
+// Drop every reader and its caches. For tests, and for any future "the archive
+// was rebuilt under us" reset — not called on a normal page.
+export function resetReaders(): void {
+ byOrigin.clear();
+}
diff --git a/export/app/duplicates/DuplicatesClient.tsx b/export/app/duplicates/DuplicatesClient.tsx
@@ -10,6 +10,7 @@ import { useMediaQuery } from "yt-dlp-transcript-common/lib/useMediaQuery";
import { Checkbox } from "yt-dlp-transcript-common/components/ui/checkbox";
import { VirtualRow } from "yt-dlp-transcript-common/components/VirtualRow";
import { fetchTranscript } from "yt-dlp-transcript-common/components/transcriptCache";
+import { fetchDuplicateReport } from "yt-dlp-transcript-common/components/duplicatesCache";
import { formatDate } from "yt-dlp-transcript-common/lib/format";
import type {
DuplicateCluster,
@@ -102,14 +103,9 @@ export function DuplicatesClient() {
// state rather than an error.
useEffect(() => {
let alive = true;
- fetch("/duplicates.json")
- .then((r) => (r.ok ? r.json() : null))
- .then((report: DuplicateReport | null) => {
- if (alive) setState({ status: "ready", report });
- })
- .catch(() => {
- if (alive) setState({ status: "ready", report: null });
- });
+ fetchDuplicateReport().then((report) => {
+ if (alive) setState({ status: "ready", report });
+ });
return () => {
alive = false;
};
diff --git a/export/app/lib/offlineCache.ts b/export/app/lib/offlineCache.ts
@@ -12,10 +12,11 @@
// per (origin, slug); the site SW ignores it (same-origin only).
import {
- pageFileName,
- type ChannelTranscriptsManifest,
-} from "yt-dlp-transcript-common/lib/manifest";
-import { idBaseUrl, makeId } from "yt-dlp-transcript-common/components/originId";
+ channelArchiveUrls,
+ siteArchiveUrls,
+ type PagedManifest,
+} from "yt-dlp-transcript-common/lib/archive/offlineUrls";
+import { idBaseUrl, makeId, splitId } from "yt-dlp-transcript-common/components/originId";
const PINNED_KEY = "ytdlp-tb:offline-channels";
@@ -56,16 +57,17 @@ function sendToSw(
});
}
-async function fetchManifest(
- origin: string,
- slug: string,
-): Promise<ChannelTranscriptsManifest | null> {
+// Deliberately NOT the shared ArchiveReader: the reader is a normal `fetch`, so
+// a page walking it would be answered from the very service-worker cache this
+// function exists to REFILL, and would compute its download list from the stale
+// copy. `cache: "no-store"` is the whole point of this read. What is shared is
+// the thing that actually drifted — the URL shape, and the two lists in
+// common/lib/archive/offlineUrls.ts.
+async function readManifest(url: string): Promise<PagedManifest | null> {
try {
- const res = await fetch(`${idBaseUrl(origin)}/transcripts/${slug}/manifest.json`, {
- cache: "no-store",
- });
+ const res = await fetch(url, { cache: "no-store" });
if (!res.ok) return null;
- return (await res.json()) as ChannelTranscriptsManifest;
+ return (await res.json()) as PagedManifest;
} catch {
return null;
}
@@ -73,9 +75,34 @@ async function fetchManifest(
export type DownloadProgress = { done: number; total: number };
-// Download every shard of a channel into the SW cache, reporting progress. The
-// URL list is the channel manifest plus each page-NNNN.json, prefixed with the
-// channel's origin so a hub can cache cross-origin channels.
+// Is any channel of this origin already pinned? That is the test for whether
+// the origin's SITE DATA (summaries, stats, the root files) is already in the
+// cache, and it is deliberately answered from the pinned registry rather than
+// by asking the worker: the two lists are written and evicted together, so one
+// pinned channel means one downloaded copy of the site data.
+//
+// The failure mode this accepts is the one the registry already has — a viewer
+// who clears site data while keeping localStorage gets a pin with no bytes
+// behind it, which is exactly why channelCachedPages() exists. The honest fix
+// is a status round-trip per origin; it is not worth a second SW message for a
+// case a re-pin already repairs.
+function originHasPin(origin: string): boolean {
+ return pinnedChannels().some((id) => splitId(id).origin === origin);
+}
+
+// Download a channel for offline use, reporting progress.
+//
+// EVERY PER-CHANNEL LAYER THAT HAS A MANIFEST, not just /transcripts — that is
+// why an offline channel now has its live chat, its social posts and its AI
+// digests — plus, ONCE PER ORIGIN, the site-wide documents the viewer needs to
+// render any of it: the summaries index, stats, and the root files. The
+// duplicates report is one of those root files, which is why the duplicates page
+// used to be dead with no connection.
+//
+// The site data is downloaded with the FIRST channel pinned on an origin and
+// skipped for every channel after it. Pinning it per channel is ~41.5 MB of
+// identical bytes each time on a corpus the size of jeralyzer, re-fetched for
+// real because the worker bulk-caches with `cache: "reload"`.
export async function downloadChannelOffline(
origin: string,
slug: string,
@@ -83,13 +110,19 @@ export async function downloadChannelOffline(
): Promise<boolean> {
const sw = await controller();
if (!sw) return false;
- const manifest = await fetchManifest(origin, slug);
- if (!manifest) return false;
const base = idBaseUrl(origin);
- const urls = [`${base}/transcripts/${slug}/manifest.json`];
- for (let p = 0; p < manifest.pageCount; p++) {
- urls.push(`${base}/transcripts/${slug}/${pageFileName(p)}`);
+
+ const urls = await channelArchiveUrls(readManifest, base, slug);
+ // No transcripts manifest: the channel is not readable, so there is nothing
+ // to take offline. Refused before anything is written, exactly as before.
+ if (urls.length === 0) return false;
+
+ // Checked BEFORE this channel is pinned, so the first channel of an origin
+ // brings the site data with it and later ones do not.
+ if (!originHasPin(origin)) {
+ urls.push(...(await siteArchiveUrls(readManifest, base)));
}
+
await sendToSw(sw, { type: "CACHE_URLS", origin, slug, urls }, (data) => {
if (data.type === "progress") {
onProgress?.({ done: data.done, total: data.total });
@@ -99,6 +132,11 @@ export async function downloadChannelOffline(
return true;
}
+// Remove a channel's offline copy — and, when it was the LAST pinned channel of
+// its origin, that origin's site data with it. Without the second half the
+// summaries/stats/root bytes survive every removal and there is no way to get
+// them back out of the cache at all; they are downloaded once, so they have to
+// be evicted once.
export async function evictChannelOffline(
origin: string,
slug: string,
@@ -108,6 +146,9 @@ export async function evictChannelOffline(
await sendToSw(sw, { type: "EVICT_CHANNEL", origin, slug }).catch(() => {});
}
unpin(origin, slug);
+ if (sw && !originHasPin(origin)) {
+ await sendToSw(sw, { type: "EVICT_SITE", origin }).catch(() => {});
+ }
}
// How many page shards of a channel are currently cached (0 = not offline).
diff --git a/export/service-worker/site-sw.js b/export/service-worker/site-sw.js
@@ -8,12 +8,18 @@
* Caches:
* SHELL — app shell: hashed /_next/static/* (immutable, cache-first) + HTML
* navigations (network-first, cache fallback) + icons/manifest.
- * PAGES — transcript/subs/summaries JSON shards (manifest.json + page-NNNN.json).
- * Pages are cache-first; manifests are network-first so a rebuild is
- * seen. Invalidation is keyed off manifest.generatedAt: when a channel's
- * manifest generatedAt changes, that channel's cached page shards are
- * evicted (the shards have stable, non-content-hashed URLs, so this is
- * the only reliable staleness signal).
+ * PAGES — every published JSON document: the per-channel shard trees
+ * (manifest.json + page-NNNN.json), the site-wide flat trees
+ * (summaries, stats) and the root documents (corpus.json, site.json,
+ * search-aliases.json, duplicates.json). Per-channel pages are
+ * cache-first; per-channel manifests are network-first so a rebuild is
+ * seen, with invalidation keyed off manifest.generatedAt — when a
+ * channel's manifest generatedAt changes, that channel's cached page
+ * shards are evicted (the shards have stable, non-content-hashed URLs,
+ * so this is the only reliable staleness signal). The site-wide
+ * documents have no per-channel generatedAt to evict against, so they
+ * are network-first with a cache fallback: fresh online, present
+ * offline, never stale-forever.
* META — tiny synthetic Responses storing each channel's last-seen generatedAt
* (avoids IndexedDB inside the SW).
*/
@@ -23,8 +29,18 @@ const SHELL = `shell-${VERSION}`;
const PAGES = `pages-${VERSION}`;
const META = `meta-${VERSION}`;
-// Matches /transcripts/<slug>/... and the parallel /subs, /summaries trees.
-const SHARD_RE = /^\/(transcripts|subs|posts|digests|summaries)\/([^/]+)\/(.+)$/;
+// THE THREE URL FAMILIES OF THE PUBLISHED CONTRACT. A service worker cannot
+// import, so these lists are hand-written — and `common/lib/archive/contract.
+// test.ts` reads THIS FILE and fails if they drift from CONTRACT.layers,
+// PER_CHANNEL_TREES and ROOT_FILES. Add a layer to the contract and this file
+// is what the test sends you to.
+
+// Per-channel trees: /<tree>/<slug>/(manifest.json|page-NNNN.json).
+const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/;
+// Flat trees: one manifest and its pages at the tree root, no channel level.
+const FLAT_RE = /^\/(summaries|stats)\/(.+)$/;
+// The root documents a reader fetches by name.
+const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/;
self.addEventListener("install", () => {
// Activate immediately — no precache list (corpus is too large to bundle).
@@ -60,6 +76,16 @@ self.addEventListener("fetch", (event) => {
return;
}
+ // Site-wide archive documents. Not cache-first: there is no per-channel
+ // generatedAt to evict them against, so a cache-first copy would be stale
+ // until the SW version changed. Network-first keeps them fresh online and
+ // present offline — which is what made the duplicates page work with no
+ // connection.
+ if (FLAT_RE.test(url.pathname) || ROOT_RE.test(url.pathname)) {
+ event.respondWith(networkFirst(req, PAGES));
+ return;
+ }
+
// App shell.
if (url.pathname.startsWith("/_next/static/") || url.pathname.startsWith("/icons/")) {
event.respondWith(cacheFirst(req, SHELL));
@@ -86,6 +112,20 @@ async function cacheFirst(req, cacheName) {
}
}
+// Network-first: serve the fresh copy and remember it; fall back to whatever is
+// cached when the network is gone.
+async function networkFirst(req, cacheName) {
+ const cache = await caches.open(cacheName);
+ try {
+ const res = await fetch(req);
+ if (res.ok) cache.put(req, res.clone());
+ return res;
+ } catch (err) {
+ const hit = await cache.match(req);
+ return hit || Response.error();
+ }
+}
+
// Navigations: network-first (fresh HTML after deploys), fall back to cache, then
// to any cached document so the installed app still opens offline.
async function networkFirstDoc(req) {
@@ -135,10 +175,14 @@ async function handleManifest(req, slug) {
async function evictChannelPages(slug) {
const pages = await caches.open(PAGES);
const keys = await pages.keys();
+ // The PER-CHANNEL trees, and only those: /summaries/<slug>/ never existed
+ // (summaries is flat) and /posts, /digests were missing, so a rebuilt channel
+ // kept serving its old social posts and AI digests from cache.
const prefixes = [
`/transcripts/${slug}/`,
`/subs/${slug}/`,
- `/summaries/${slug}/`,
+ `/posts/${slug}/`,
+ `/digests/${slug}/`,
];
await Promise.all(
keys.map((k) => {
@@ -151,6 +195,22 @@ async function evictChannelPages(slug) {
);
}
+// The site-wide documents: the flat trees and the root files, i.e. exactly what
+// the page downloads ONCE per origin. Evicted when its last pinned channel goes,
+// because nothing else ever removes them — the per-channel sweep above only
+// knows about /<tree>/<slug>/ prefixes. The same two regexes drive the fetch
+// handler, so this can never fall out of step with what was cached.
+async function evictSiteData() {
+ const pages = await caches.open(PAGES);
+ const keys = await pages.keys();
+ await Promise.all(
+ keys.map((k) => {
+ const p = new URL(k.url).pathname;
+ return FLAT_RE.test(p) || ROOT_RE.test(p) ? pages.delete(k) : null;
+ }),
+ );
+}
+
// Per-channel generatedAt stored as a synthetic Response in the META cache.
async function readMeta(slug) {
const meta = await caches.open(META);
@@ -172,6 +232,10 @@ self.addEventListener("message", (event) => {
event.waitUntil(
evictChannelPages(data.slug).then(() => port && port.postMessage({ ok: true })),
);
+ } else if (data.type === "EVICT_SITE") {
+ event.waitUntil(
+ evictSiteData().then(() => port && port.postMessage({ ok: true })),
+ );
} else if (data.type === "CHANNEL_STATUS") {
event.waitUntil(channelStatus(data.slug, port));
}
diff --git a/export/service-worker/sw-hub.js b/export/service-worker/sw-hub.js
@@ -2,7 +2,7 @@
*
* Same offline model as the site SW (browse-cache + opt-in per-channel
* download), but CROSS-ORIGIN: the hub federates content from many origins, so
- * this worker caches transcript/subs/summaries shards from ANY origin — not
+ * this worker caches every published archive document from ANY origin — not
* just its own. The only origins it ever reaches are ones the user added, whose
* JSON is CORS-open (Access-Control-Allow-Origin: *); a cross-origin fetch is
* therefore a readable `cors` response that can be cached (never `no-cors` —
@@ -21,9 +21,19 @@ const SHELL = `shell-${VERSION}`;
const PAGES = `pages-${VERSION}`;
const META = `meta-${VERSION}`;
-// Matches /transcripts/<slug>/... and the parallel /subs, /summaries trees, on
-// any origin (we test url.pathname, so it's origin-independent).
-const SHARD_RE = /^\/(transcripts|subs|posts|summaries)\/([^/]+)\/(.+)$/;
+// THE THREE URL FAMILIES OF THE PUBLISHED CONTRACT, matched on url.pathname so
+// they are origin-independent. A service worker cannot import, so these lists
+// are hand-written — and `common/lib/archive/contract.test.ts` reads THIS FILE
+// (and site-sw.js) and fails if they drift from CONTRACT.layers,
+// PER_CHANNEL_TREES and ROOT_FILES. `digests` was missing here entirely: the
+// hub federated AI digests it could never cache.
+
+// Per-channel trees: /<tree>/<slug>/(manifest.json|page-NNNN.json).
+const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/;
+// Flat trees: one manifest and its pages at the tree root, no channel level.
+const FLAT_RE = /^\/(summaries|stats)\/(.+)$/;
+// The root documents a reader fetches by name.
+const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/;
self.addEventListener("install", () => {
self.skipWaiting();
@@ -57,6 +67,15 @@ self.addEventListener("fetch", (event) => {
return;
}
+ // Site-wide archive documents, also from ANY origin — a member site's
+ // summaries index and its /duplicates.json are as federated as its shards.
+ // Network-first: no per-channel generatedAt to evict them against, so fresh
+ // online and present offline rather than stale-forever.
+ if (FLAT_RE.test(url.pathname) || ROOT_RE.test(url.pathname)) {
+ event.respondWith(networkFirst(req, PAGES));
+ return;
+ }
+
// App shell is same-origin only (the hub's own bundle).
if (url.origin !== self.location.origin) return;
if (url.pathname.startsWith("/_next/static/") || url.pathname.startsWith("/icons/")) {
@@ -83,6 +102,20 @@ async function cacheFirst(req, cacheName) {
}
}
+// Network-first: serve the fresh copy and remember it; fall back to whatever is
+// cached when the network is gone.
+async function networkFirst(req, cacheName) {
+ const cache = await caches.open(cacheName);
+ try {
+ const res = await fetch(req);
+ if (res.ok) cache.put(req, res.clone());
+ return res;
+ } catch (err) {
+ const hit = await cache.match(req);
+ return hit || Response.error();
+ }
+}
+
async function networkFirstDoc(req) {
const cache = await caches.open(SHELL);
try {
@@ -133,10 +166,13 @@ function belongsToChannel(entryUrl, origin, slug) {
const u = new URL(entryUrl);
const wantOrigin = origin || self.location.origin;
if (u.origin !== wantOrigin) return false;
+ // The PER-CHANNEL trees, and only those: /summaries/<slug>/ never existed
+ // (summaries is flat) and /posts, /digests were missing.
const prefixes = [
`/transcripts/${slug}/`,
`/subs/${slug}/`,
- `/summaries/${slug}/`,
+ `/posts/${slug}/`,
+ `/digests/${slug}/`,
];
return prefixes.some(
(pre) => u.pathname.startsWith(pre) && !u.pathname.endsWith("manifest.json"),
@@ -151,6 +187,26 @@ async function evictChannelPages(origin, slug) {
);
}
+// One member origin's site-wide documents — the flat trees and the root files,
+// exactly what the page downloads ONCE per origin. Scoped by origin like
+// everything else here, so evicting one member never touches another's. The
+// per-channel sweep above only knows /<tree>/<slug>/ prefixes, so without this
+// these bytes would survive every removal.
+async function evictSiteData(origin) {
+ const pages = await caches.open(PAGES);
+ const wantOrigin = origin || self.location.origin;
+ const keys = await pages.keys();
+ await Promise.all(
+ keys.map((k) => {
+ const u = new URL(k.url);
+ if (u.origin !== wantOrigin) return null;
+ return FLAT_RE.test(u.pathname) || ROOT_RE.test(u.pathname)
+ ? pages.delete(k)
+ : null;
+ }),
+ );
+}
+
// Per-(origin, channel) generatedAt stored as a synthetic Response in META.
function metaKey(origin, slug) {
return `/__gen__/${encodeURIComponent(origin || "")}/${slug}`;
@@ -178,6 +234,10 @@ self.addEventListener("message", (event) => {
() => port && port.postMessage({ ok: true }),
),
);
+ } else if (data.type === "EVICT_SITE") {
+ event.waitUntil(
+ evictSiteData(origin).then(() => port && port.postMessage({ ok: true })),
+ );
} else if (data.type === "CHANNEL_STATUS") {
event.waitUntil(channelStatus(origin, data.slug, port));
}
diff --git a/plans/one-core-phase-2.md b/plans/one-core-phase-2.md
@@ -780,6 +780,114 @@ orchestration, `scanPlan.test.ts` sits beside it, and it calls the single
`passesFilters` from `lib/search/evalTree` — so the pruner and the scanner
still cannot disagree about what matches, which was the one property that
mattered.
+### S2a — shipped 2026-09-12
+
+Branch `one-core/phase-2-s2a`, off `7f86aef` (the
+`integrate/2026-09-storage-priority` tip, S1 merged). Four commits,
+three code commits `dc3aef1` → `2eb9c3b` plus this note, unmerged. **No URL shape moved, no `corpus.json` byte
+moved, no CONTRACT version moved, no architecture allow-list entry added or
+burned, and no `common/package.json` exports line added.**
+
+| commit | what |
+|---|---|
+| `dc3aef1` | the eight caches — plus `SearchDataContext`, `siteRegistry` and `DuplicatesClient` — onto one `RemoteSource` per origin; `lib/archive/readers.ts`; the reader's raw document reads |
+| `8ca7022` | `offlineCache` enumerates `ARCHIVE_TREES` + `ROOT_FILES`; both service workers widened; the SW and search-worker guard cases in `contract.test.ts` |
+| `2eb9c3b` | `archive/readers.test.ts` + `components/archiveCaches.test.ts` |
+| (this commit) | this note |
+
+#### Seven divergences from the plan above, each because the code said so
+
+**1. The caches do NOT adopt `ArchiveReader`'s tolerant methods. The reader
+gained a block of RAW DOCUMENT READS underneath them instead.** The slice was
+specified as `memo(originId, () => reader.X(...))`, and for transcripts that is
+exactly what it is. For everything else `ArchiveReader` is the wrong surface,
+twice over.
+
+It has no method at all for four of the documents the viewer reads:
+`/subs/manifest.json` and `/posts/manifest.json` (the SITE-level index of which
+channels ship a tree — a different document from a channel's own manifest), and
+the raw summaries/stats manifest-plus-pages. What it exposes for those is
+`videoIndex()` / `statsIndex()`, folded slug-keyed maps, which are the wrong
+shape for a UI that renders arrays and loads page by page under react-query.
+
+And where a method does exist, its answer to absence is the opposite of the
+viewer's. `subsManifest`/`postsManifest`/`digestsManifest` resolve `null` for
+anything that is not a 200 — right for a tool probing thirty channels, wrong
+for a browser, because **react-query retries a rejected query and will never
+retry a resolved `null`**. Folding both policies into one method is exactly how
+a transient blip becomes a panel that is empty for the rest of the session.
+
+So `RemoteSource` now has twelve one-to-three-line reads that do URL + fetch +
+parse and THROW, and every tolerance policy sits on top and picks its own
+answer. **The MCP's behaviour is unchanged by construction**: each tolerant
+method is rebased on the raw read it duplicated, and `buildVideoIndex` /
+`buildStatsIndex` already treat a throw from their readers as absence, so the
+degradation paths are the same code they were. This is not the `record(layer,
+slug, id)` the plan forbids — these are the DOCUMENTS of the walk, not records
+inside a shard, and no read count moves.
+
+`ArchiveHttpError` came with them, for the one caller that has to tell a 404
+from a dropped connection (the alias dictionary, whose query has `staleTime:
+Infinity`). Its `message` is byte-identical to the `Error` it replaced.
+
+**2. Two memos stay in `components/`, and neither is a leftover.** The
+subs/posts CHANNEL-manifest memos keep the throwing policy of (1). The posts
+and digest PAGE memos exist because the reader caches subs and transcript pages
+but not those two — and `fetchThread` walks every page of a channel, so without
+a memo opening two posts in one thread re-downloads the whole channel. Adding
+those caches to `RemoteSource` instead would have moved the bench's read counts,
+which is not a thing to do in a slice that claims the walk is all that changed.
+
+**3. `io-stats` blind spots are preserved deliberately.** The raw reads take an
+OPTIONAL `kind`, and the ones that pass none are exactly the reads that
+recorded nothing before (the summaries/stats/duplicates/alias folds used a bare
+`fetch`). Closing them would move the bench's structural counters for a reason
+unrelated to the walk. That is also why the bench was not re-run for this slice
+— nothing in the reader's semantics or read counts moved.
+
+**4. `contract.ts` needed `treeManifestUrl`, and `ARCHIVE_TREES` /
+`PER_CHANNEL_TREES` beside it.** `manifestUrl("subs")` throws with no slug, and
+correctly so — but `/subs/manifest.json` is a real published document. A
+per-channel tree has BOTH a root manifest and a per-channel one; `manifestUrl`
+builds the second, `treeManifestUrl` the first, and for a flat tree they are the
+same file. The two lists are what the offline cache and the SW guard test
+enumerate; `CONTRACT.layers` stays the frozen published list.
+
+**5. The service workers needed more than a wider `SHARD_RE`, and the plan's
+stated goal is why.** `SHARD_RE` matches three path segments, so
+`/duplicates.json` and `/summaries/*` never matched it and the fetch handler let
+them through to the network — downloading a root file into the cache and never
+serving it from there closes nothing. Both workers gain `FLAT_RE` and `ROOT_RE`
+answered NETWORK-FIRST (those documents have no per-channel `generatedAt` to
+evict against, so cache-first would be stale until the SW version moved).
+
+Three drifts surfaced while writing the lists down. `summaries` was in
+site-sw's `SHARD_RE` and in both eviction prefix lists: dead in both places,
+because it is a flat tree and `/summaries/<slug>/` does not exist. `/posts/` and
+`/digests/` were absent from both eviction lists, so a rebuilt channel kept
+serving its old posts and digests from cache. And sw-hub had no `digests` at
+all — the hub federated a layer it could never cache.
+
+**6. `offlineCache` keeps its own `cache: "no-store"` fetch** rather than taking
+the shared reader. The reader is a plain `fetch`, so it would be answered from
+the very service-worker cache that function exists to refill, and would compute
+its download list from the stale copy. What is shared is the thing that actually
+drifted: the URL shape.
+
+**7. One new `components/*.ts`, against the invariant, and why it is not the
+thing the invariant protects.** `components/archiveCaches.test.ts` is a test: it
+is imported by nothing, the `"./components/*"` catch-all maps to `.tsx`, and it
+needs no `package.json` exports line — which is the cost the invariant names. It
+cannot live under `lib/` because S3 adds `"components"` to `FORBIDDEN.lib` in
+the architecture test, and a test for the caches has to import them.
+
+Three walk sites beyond the eight caches moved too, all of them covered by the
+brief's "only the fetch walk moves": `SearchDataContext.tsx` (four URLs, state
+untouched), `siteRegistry.ts` (two), and `export/app/duplicates/
+DuplicatesClient.tsx`'s bare `fetch("/duplicates.json")`, which became a new
+`fetchDuplicateReport()` export on `duplicatesCache`. A repo-wide sweep for
+hand-written archive paths now returns only `searchIndex.worker.ts` — the
+deliberate copy, and it is pinned.
#### Gates
@@ -903,3 +1011,125 @@ settings, social links or the footer.
load-bearing prose: they are the only thing stopping the two evaluations of
the same algebra from drifting, because no test can compare a streaming
slug-set walk against a per-record boolean.
+- `pnpm --filter yt-dlp-transcript-common test` — **1095 passed / 0 failed**
+ (baseline 1077 at `7f86aef`, + 3 `contract.test.ts`, + 6 `readers.test.ts`,
+ + 9 `archiveCaches.test.ts`; none lost). Architecture test green, allow-list
+ untouched.
+- `pnpm --filter yt-dlp-transcript-mcp test` — **205 passed / 0 failed**.
+- `pnpm test:scripts` — **71 passed / 1 skipped**.
+- `pnpm --filter export exec next build` — **compiled successfully**, 11 static
+ pages. The only proof `reader-fs.ts` is unreachable from a client module, and
+ it now has eight more `components/` modules reaching `lib/archive/` to prove
+ it about. (The one warning is S1's pre-existing NFT trace.)
+- **compose-site byte-identity** over the committed fixture: `IDENTICAL modulo
+ the build clock` for both trees — 16 files under `public/`, 7 under `index/`.
+- **The mcp bench was NOT run**, per the brief — the reader's semantics and read
+ counts do not move (see divergence 3).
+- **e2e**, behind the queue lock from worktree #10 (ports 4001/4011/4010/4020):
+ export `e2e` **172 passed / 0 failed** (S1's baseline exactly), `e2e:hub`
+ **5 passed / 0 failed**. `e2e:2origin` was skipped: it is known-red on the
+ base for the hub `/ask` prerender, which S1 verified at `c7f7b90` and which
+ nothing here is in the path of.
+- The editor specs `export-search` + `export-player-platform-cache`: **20 passed
+ / 1 failed**, and the one failure is a SUBSET-RUN ARTIFACT, not a regression.
+ `export-search.spec.ts:612` opens `editor/test-settings.json` with `readJson`;
+ that file is gitignored and is only ever written by `writeSettings()`, i.e. by
+ an earlier spec in the full suite. A fresh worktree running just these two
+ files has never had one written, so the test dies on `ENOENT` before it
+ touches any code. **Confirmed by re-running the same two spec files at
+ `7f86aef` in the same worktree: the same spec fails the same way.** Worth
+ fixing independently — the spec should seed the file rather than assume a
+ predecessor left one — but it is not this slice's.
+
+#### What this slice deliberately leaves for S2c
+
+A hub caching a member's `/duplicates.json` or `/digests/*` still depends on
+those paths being in the composed `_headers` CORS set, which they are not.
+S2c's `_headers` commit is what makes the widened `sw-hub.js` reach them
+cross-origin; same-origin (the site SW) works today.
+
+The offline DOWNLOAD path has no e2e coverage and gains none here: the service
+worker registers in production builds only, so `pwa.spec` can assert the control
+renders but never that a download lands. The guard that the SW lists and the
+download list agree with the contract is `contract.test.ts`, verified by
+mutation (dropping `digests` from sw-hub's `SHARD_RE` turns it red).
+
+#### Review fixes — `9451be8`, `5ac97b3`
+
+Three items came back. One was a real defect and it is worth recording as a
+lesson about the shape of the thing, not just as a bug.
+
+**F1 — an offline copy is TWO lists, and shipping it as one put site-wide bytes
+on a per-channel bill.** `downloadChannelOffline` appended every `ARCHIVE_TREES`
+entry — the flat, site-wide `summaries` and `stats` included — plus all four
+`ROOT_FILES` to EACH channel's list. Measured on jeralyzer: summaries ~14 MB,
+stats ~21.6 MB (its `page-0000` alone is 20.97 MB), `/duplicates.json` 5.9 MB.
+**~41.5 MB of byte-identical data per channel pinned**, genuinely re-fetched
+because `cacheUrls` bulk-caches with `cache: "reload"`.
+
+The download was the smaller half. **None of it was removable.** Both evict
+paths sweep `/<tree>/<slug>/` prefixes, and not one of those URLs lives under a
+channel prefix — so "remove offline copy" left every byte in `PAGES` forever,
+with no path in the app that could ever reclaim it. Enumerating the contract was
+the right instinct and it was applied at the wrong granularity: the contract has
+two kinds of tree, and the cost model follows that split exactly.
+
+So the lists split, in `common/lib/archive/offlineUrls.ts`:
+`channelArchiveUrls` (the per-channel trees of one channel) and
+`siteArchiveUrls` (the flat trees plus the root files of one origin). Site data
+rides with the FIRST channel pinned on an origin and is skipped after; a new
+`EVICT_SITE` message in both workers removes it when that origin's last pinned
+channel goes. Downloaded once, evicted once. **The per-channel download is
+byte-for-byte what it was.**
+
+"Already present?" is answered from the pinned registry rather than a new
+status round-trip: the two lists are written and removed together, so one pinned
+channel means one copy of the site data. That inherits the registry's existing
+drift — a viewer who clears site data while keeping `localStorage` — which is
+the same gap `channelCachedPages()` already exists for, and which a re-pin
+repairs. A per-origin `SITE_STATUS` message would make it truthful; it did not
+seem worth a second SW round-trip for a case that self-heals.
+
+The builders live in `common/` and not `export/` for a reason beyond tidiness:
+**`export/` has no test runner**, and this split is precisely the thing that
+needs one. Six cases, and the load-bearing ones are structural rather than
+literal — the channel list contains no flat tree and no root file; every channel
+URL matches `/<tree>/<slug>/`, the only shape either worker's per-channel sweep
+can remove; the two lists are disjoint and together cover `ARCHIVE_TREES`
+exactly.
+
+**F2 — a browser reader now gets a browser's budget.** `RemoteSource` defaults
+to 48 MB per page cache and holds two, so the registry was handing every origin
+96 MB: reasonable for one long-lived MCP process, not for a tab, and least of
+all for a hub page holding a reader per member with no shared ceiling and no way
+for a browser to reach the env knob. `readers.ts` now passes 8 MB explicitly —
+16 MB per origin, so five federated members are 80 MB rather than 480 MB. It can
+be this small at no cost in reads because the viewer memoises every RECORD it
+has seen (`transcriptCache.resolved` plus IndexedDB), so an evicted page is only
+re-read for a video nobody has opened yet; and `MIN_CACHED_PAGES` still floors
+it at two entries where one page exceeds the whole budget. **`mcp/src/source.ts`
+constructs `RemoteSource` directly and keeps the 48 MB default**, so no bench
+counter moves.
+
+**F4 — `aliasesCache`'s comment now says what the code does.** It claimed
+"404 → data" while the catch returns `[]` for any `ArchiveHttpError`. The code
+is right (it is the old `if (!r.ok) return []`, and a server answering 500
+should cost the suggestion chip rather than the search); the comment was the
+part that was wrong.
+
+Noted for the merger, no action taken here: `contract.ts`'s `manifestUrl` flat
+branch now calls `treeManifestUrl`, and S2c edits nearby.
+
+##### Gates after the fixes
+
+- `pnpm -r exec tsc --noEmit` — clean in all six packages.
+- common — **1101 passed / 0 failed** (1095 + 6 `offlineUrls.test.ts`).
+- mcp — **205 passed / 0 failed**. `pnpm test:scripts` — **71 passed / 1
+ skipped**.
+- `pnpm --filter export exec next build` — **compiled successfully**, 11 static
+ pages.
+- **e2e** re-run behind the queue lock: export `e2e` **172 passed / 0 failed**,
+ `e2e:hub` **5 passed / 0 failed** — unchanged from before the fixes, which is
+ the point: the per-channel download and every URL shape are what they were.
+ The editor pair and `e2e:2origin` were not re-run; nothing in these two
+ commits touches the editor, and 2origin is red at base.