commit c70b355a074669b0c18de909a8cc925434b67e4b
parent ea2d3d04d126af72cf52b77dfe0d48a78e791d88
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 12 Sep 2026 02:42:59 -0400
common,export: the viewer's archive walks become one ArchiveReader per origin
The eight component caches each carried their own copy of the published
shard walk — manifest -> slugToPage -> page-NNNN.json — spelled as template
literals over `idBaseUrl(origin)`. They now call a RemoteSource from
`lib/archive/readers.ts`, one per origin, so the URL shape is the contract's
and the promise-coalescing and byte-budgeted LRU are the reader's.
What stays in `components/` is what is genuinely the viewer's: the per-id
memo, the in-flight dedupe, the IndexedDB warming, and the two memos whose
ERROR POLICY differs from the reader's. That difference is the reason the
reader gained a block of raw document reads rather than the caches adopting
its tolerant methods: a tool scanning thirty channels wants `null` for a
missing posts tree, but react-query retries a rejected query and will never
retry a resolved `null`, so folding both into one tolerant method is how a
transient blip becomes a permanently empty panel. The raw reads throw; every
tolerance policy — including this file's own empty-when-absent folds, now
rebased on them — sits on top and picks its own answer to "absent".
`ArchiveHttpError` exists for the one caller that needs to tell a 404 from a
dropped connection (the alias dictionary, whose query has staleTime:
Infinity). Its message is byte-identical to the Error it replaced.
No URL moved. `archiveUrl` leaves a path root-relative when there is no base,
so `new RemoteSource("")` emits exactly the root-relative strings the caches
built before, and the export e2e's shard route-mocks still match. The
io-stats blind spots are preserved deliberately: the reads that never
recorded a read (the summaries/stats/duplicates/alias folds) still pass no
`kind`, so the mcp bench's structural counters cannot move for a reason that
has nothing to do with the walk.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
14 files changed, 322 insertions(+), 284 deletions(-)
diff --git a/common/components/SearchDataContext.tsx b/common/components/SearchDataContext.tsx
@@ -20,11 +20,11 @@ import { useSubsManifest } from "./subsCache";
import { usePostsManifest } from "./postsCache";
import { useSearchAliases, fetchAliases } from "./aliasesCache";
import { mergeAliases, type SearchAlias } from "../lib/searchAliases";
-import { idBaseUrl, makeId } from "./originId";
+import { makeId } from "./originId";
import type { DisplaySummary } from "../lib/transcripts";
import type { Manifest, SubsManifest } from "../lib/manifest";
import type { PostsManifest } from "../lib/posts";
-import { pageFileName } from "../lib/manifest";
+import { readerFor } from "../lib/archive/readers";
import {
DEFAULT_GROUP_FALLBACK_ID,
FALLBACK_GROUP,
@@ -158,12 +158,6 @@ export function SingleSiteDataProvider({ children }: { children: ReactNode }) {
);
}
-async function fetchJson<T>(url: string): Promise<T> {
- const r = await fetch(url);
- if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`);
- return (await r.json()) as T;
-}
-
// A federated origin the hub reads from. `origin` is "" for the hub's own
// same-origin pool, else a full origin ("https://x.com"). `siteTitle` labels
// its channel group; `accent` is carried for provenance in the UI.
@@ -194,8 +188,7 @@ export function MultiSiteDataProvider({
const manifestQueries = useQueries({
queries: sites.map((s) => ({
queryKey: ["manifest", s.origin],
- queryFn: () =>
- fetchJson<Manifest>(`${idBaseUrl(s.origin)}/summaries/manifest.json`),
+ queryFn: () => readerFor(s.origin).readSummariesManifest(),
})),
});
const manifestsSettled = manifestQueries.every(
@@ -218,10 +211,7 @@ export function MultiSiteDataProvider({
const pageQueries = useQueries({
queries: pageDescriptors.map((d) => ({
queryKey: ["summaries-page", d.origin, d.index],
- queryFn: () =>
- fetchJson<DisplaySummary[]>(
- `${idBaseUrl(d.origin)}/summaries/${pageFileName(d.index)}`,
- ),
+ queryFn: () => readerFor(d.origin).readSummariesPage(d.index),
})),
});
@@ -300,8 +290,7 @@ export function MultiSiteDataProvider({
const subsQueries = useQueries({
queries: sites.map((s) => ({
queryKey: ["subs-manifest", s.origin],
- queryFn: () =>
- fetchJson<SubsManifest>(`${idBaseUrl(s.origin)}/subs/manifest.json`),
+ queryFn: () => readerFor(s.origin).readSubsSiteManifest(),
})),
});
const subsSettled = subsQueries.every((q) => q.isSuccess || q.isError);
@@ -339,9 +328,9 @@ export function MultiSiteDataProvider({
queries: sites.map((s) => ({
queryKey: ["posts-manifest", s.origin],
queryFn: () =>
- fetchJson<PostsManifest>(
- `${idBaseUrl(s.origin)}/posts/manifest.json`,
- ).catch(
+ readerFor(s.origin)
+ .readPostsSiteManifest()
+ .catch(
(): PostsManifest => ({
version: 0,
channels: [],
diff --git a/common/components/aliasesCache.ts b/common/components/aliasesCache.ts
@@ -7,15 +7,23 @@
// purely additive, so their absence must never break search.
import { useQuery } from "@tanstack/react-query";
-import { idBaseUrl } from "./originId";
+import { ArchiveHttpError } from "../lib/archive/reader";
+import { readerFor } from "../lib/archive/readers";
import { coerceAliasConfig, type SearchAlias } from "../lib/searchAliases";
const EMPTY: SearchAlias[] = [];
+// A site that ships no dictionary answers 404, and that is data: resolve empty.
+// A TRANSPORT failure is not an answer and is rethrown, so react-query retries
+// it rather than caching "this site has no aliases" for the session (the query
+// below has staleTime: Infinity, so a swallowed blip would be permanent).
export async function fetchAliases(origin = ""): Promise<SearchAlias[]> {
- const r = await fetch(`${idBaseUrl(origin)}/search-aliases.json`);
- if (!r.ok) return [];
- return coerceAliasConfig(await r.json()).aliases;
+ try {
+ return coerceAliasConfig(await readerFor(origin).readAliasConfig()).aliases;
+ } catch (err) {
+ if (err instanceof ArchiveHttpError) return [];
+ throw err;
+ }
}
export function useSearchAliases(origin = ""): SearchAlias[] {
diff --git a/common/components/digestCache.ts b/common/components/digestCache.ts
@@ -1,9 +1,11 @@
"use client";
import type { ChannelDigestsManifest, VideoDigest } from "../lib/digests";
-import { digestPageFileName, manifestHasDigest } from "../lib/digests";
+import { manifestHasDigest } from "../lib/digests";
+import { PromiseMap } from "../lib/archive/reader";
+import { channelRef, readerFor } from "../lib/archive/readers";
import { idbGet, idbPut } from "./digestStore";
-import { makeId, splitId, idBaseUrl } from "./originId";
+import { makeId, splitId } from "./originId";
// Fetch layer for the derived corpus, mirroring transcriptCache.ts: a memory
// map, in-flight dedupe, manifest -> page resolution and opportunistic warming
@@ -23,11 +25,12 @@ import { makeId, splitId, idBaseUrl } from "./originId";
const resolved = new Map<string, VideoDigest | null>();
const inFlight = new Map<string, Promise<VideoDigest | null>>();
-const channelManifests = new Map<
- string,
- Promise<ChannelDigestsManifest | null>
->();
-const pagePromises = new Map<string, Promise<VideoDigest[]>>();
+// PromiseMap memoises a RESOLVED value (including `null` — the cached negative
+// this layer depends on) and drops a REJECTION, which is exactly the policy
+// this file hand-rolled: absence is an answer worth keeping, a transport
+// failure is not proof of absence.
+const channelManifests = new PromiseMap<ChannelDigestsManifest | null>();
+const pagePromises = new PromiseMap<VideoDigest[]>();
// `id` is an OriginId: a bare "channelSlug/videoId" for same-origin content, or
// "origin\tchannelSlug/videoId" for a federated cross-origin video.
@@ -87,26 +90,9 @@ function fetchChannelManifest(
channelSlug: string,
origin: string,
): Promise<ChannelDigestsManifest | null> {
- const key = makeId(origin, channelSlug);
- let p = channelManifests.get(key);
- if (!p) {
- p = fetch(`${idBaseUrl(origin)}/digests/${channelSlug}/manifest.json`)
- .then((r) => {
- if (r.status === 404) return null;
- if (!r.ok) {
- throw new Error(`Failed to fetch digests manifest for ${channelSlug}`);
- }
- return r.json() as Promise<ChannelDigestsManifest>;
- })
- .catch((err) => {
- // A transport failure is not proof of absence, so it must not be
- // cached as one — drop the entry and let the next caller retry.
- channelManifests.delete(key);
- throw err;
- });
- channelManifests.set(key, p);
- }
- return p;
+ return channelManifests.take(makeId(origin, channelSlug), () =>
+ readerFor(origin).readChannelDigestsManifest(channelSlug),
+ );
}
function fetchPage(
@@ -114,21 +100,9 @@ function fetchPage(
pageIndex: number,
origin: string,
): Promise<VideoDigest[]> {
- const key = `${makeId(origin, channelSlug)}:${pageIndex}`;
- let p = pagePromises.get(key);
- if (!p) {
- p = fetch(
- `${idBaseUrl(origin)}/digests/${channelSlug}/${digestPageFileName(pageIndex)}`,
- ).then((r) => {
- if (!r.ok) {
- throw new Error(`Failed to fetch digest page ${channelSlug}/${pageIndex}`);
- }
- return r.json() as Promise<VideoDigest[]>;
- });
- p.catch(() => pagePromises.delete(key));
- pagePromises.set(key, p);
- }
- return p;
+ return pagePromises.take(`${makeId(origin, channelSlug)}:${pageIndex}`, () =>
+ readerFor(origin).digestPage(channelRef(channelSlug, origin), pageIndex),
+ );
}
async function load(id: string): Promise<VideoDigest | null> {
diff --git a/common/components/duplicatesCache.ts b/common/components/duplicatesCache.ts
@@ -16,7 +16,7 @@
// per member is `aligned` — see duplicateSiblings below.
import { useQuery } from "@tanstack/react-query";
-import { idBaseUrl } from "./originId";
+import { readerFor } from "../lib/archive/readers";
import type {
DuplicateCluster,
DuplicateReport,
@@ -27,15 +27,23 @@ export type DuplicateLookup = ReadonlyMap<string, DuplicateCluster>;
const EMPTY: DuplicateLookup = new Map();
-export async function fetchDuplicates(origin = ""): Promise<DuplicateLookup> {
- let report: DuplicateReport | null = null;
+// The raw shipped report, or null when the site ships none (the common case —
+// compose-site writes the file only for a site with at least one publishable
+// cluster). Absent, unreachable and unparseable all fold to null: every
+// duplicate affordance is additive, so none of them may break a page.
+export async function fetchDuplicateReport(
+ origin = "",
+): Promise<DuplicateReport | null> {
try {
- const r = await fetch(`${idBaseUrl(origin)}/duplicates.json`);
- if (!r.ok) return EMPTY;
- report = (await r.json()) as DuplicateReport;
+ return await readerFor(origin).readDuplicates();
} catch {
- return EMPTY;
+ return null;
}
+}
+
+export async function fetchDuplicates(origin = ""): Promise<DuplicateLookup> {
+ const report = await fetchDuplicateReport(origin);
+ if (!report) return EMPTY;
const out = new Map<string, DuplicateCluster>();
for (const cluster of report?.clusters ?? []) {
for (const ref of cluster?.videoRefs ?? []) {
diff --git a/common/components/postsCache.ts b/common/components/postsCache.ts
@@ -5,24 +5,21 @@
// federating hub can hold several origins' posts at once without collisions.
import { useQueries, useQuery } from "@tanstack/react-query";
-import {
- postsPageFileName,
- type ChannelPostsManifest,
- type Post,
- type PostsManifest,
-} from "../lib/posts";
-import { makeId, splitId, idBaseUrl } from "./originId";
-
-async function fetchJson<T>(url: string): Promise<T> {
- const r = await fetch(url);
- if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`);
- return (await r.json()) as T;
-}
+import type { ChannelPostsManifest, Post, PostsManifest } from "../lib/posts";
+import { PromiseMap } from "../lib/archive/reader";
+import { channelRef, readerFor } from "../lib/archive/readers";
+import { makeId, splitId } from "./originId";
const resolved = new Map<string, Post>();
const inFlight = new Map<string, Promise<Post>>();
-const channelManifests = new Map<string, Promise<ChannelPostsManifest>>();
-const pagePromises = new Map<string, Promise<Post[]>>();
+// Two memos the reader does NOT provide, for two different reasons. The
+// manifest one wants a THROW where the reader's tolerant `postsManifest` wants
+// a null (see subsCache). The page one exists because the reader caches subs
+// and transcript pages but not posts pages, and fetchThread below walks EVERY
+// page of a channel — without a memo, opening two posts in a thread would
+// re-download the whole channel.
+const channelManifests = new PromiseMap<ChannelPostsManifest>();
+const pagePromises = new PromiseMap<Post[]>();
export type PostsManifestRef = { channelSlug: string; origin?: string };
@@ -53,16 +50,9 @@ export function fetchChannelPostsManifest(
channelSlug: string,
origin = "",
): Promise<ChannelPostsManifest> {
- const key = makeId(origin, channelSlug);
- let p = channelManifests.get(key);
- if (!p) {
- p = fetchJson<ChannelPostsManifest>(
- `${idBaseUrl(origin)}/posts/${channelSlug}/manifest.json`,
- );
- p.catch(() => channelManifests.delete(key));
- channelManifests.set(key, p);
- }
- return p;
+ return channelManifests.take(makeId(origin, channelSlug), () =>
+ readerFor(origin).readChannelPostsManifest(channelSlug),
+ );
}
export function fetchPostsPage(
@@ -70,23 +60,21 @@ export function fetchPostsPage(
pageIndex: number,
origin = "",
): Promise<Post[]> {
- const key = `${makeId(origin, channelSlug)}:${pageIndex}`;
- let p = pagePromises.get(key);
- if (!p) {
- p = fetchJson<Post[]>(
- `${idBaseUrl(origin)}/posts/${channelSlug}/${postsPageFileName(pageIndex)}`,
- ).then((page) => {
+ return pagePromises.take(
+ `${makeId(origin, channelSlug)}:${pageIndex}`,
+ async () => {
+ const page = await readerFor(origin).postsPage(
+ channelRef(channelSlug, origin),
+ pageIndex,
+ );
// Warm the by-id cache so a later fetchPost() for any post on this page
// is synchronous.
for (const entry of page) {
resolved.set(makeId(origin, entry.slug), entry);
}
return page;
- });
- p.catch(() => pagePromises.delete(key));
- pagePromises.set(key, p);
- }
- return p;
+ },
+ );
}
async function load(id: string): Promise<Post> {
@@ -136,7 +124,8 @@ export function usePostsManifest(origin = "") {
return useQuery<PostsManifest>({
queryKey: ["posts-manifest", origin],
queryFn: () =>
- fetchJson<PostsManifest>(`${idBaseUrl(origin)}/posts/manifest.json`)
+ readerFor(origin)
+ .readPostsSiteManifest()
// A site with no social channels ships no posts manifest; treat that as
// an empty corpus rather than an error, so the UI degrades quietly.
.catch(
diff --git a/common/components/siteRegistry.ts b/common/components/siteRegistry.ts
@@ -19,6 +19,7 @@
// and stay in sync when a site is added or removed.
import { useEffect, useSyncExternalStore } from "react";
+import { manifestUrl, rootFileUrl } from "../lib/archive/contract";
import { SITE_DESCRIPTOR_VERSION } from "../lib/siteDescriptor";
import type { PublicSiteDescriptor } from "../lib/siteDescriptor";
@@ -259,7 +260,7 @@ export async function validateSite(input: string): Promise<ValidateResult> {
let descriptor: PublicSiteDescriptor;
try {
- const res = await fetch(`${origin}/site.json`);
+ const res = await fetch(rootFileUrl("site.json", origin));
if (!res.ok) return { ok: false, reason: UNREADABLE };
descriptor = (await res.json()) as PublicSiteDescriptor;
} catch {
@@ -284,7 +285,7 @@ export async function validateSite(input: string): Promise<ValidateResult> {
// whose descriptor is fine but whose data isn't reachable/valid would fail
// silently later, so reject it now with a clear message.
try {
- const res = await fetch(`${origin}/summaries/manifest.json`);
+ const res = await fetch(manifestUrl("summaries", undefined, origin));
if (!res.ok) return { ok: false, reason: UNREADABLE };
const manifest = (await res.json()) as { version?: unknown; channels?: unknown };
if (
diff --git a/common/components/statsCache.ts b/common/components/statsCache.ts
@@ -3,14 +3,7 @@
import { useMemo } from "react";
import { useQueries, useQuery } from "@tanstack/react-query";
import type { StatsManifest, VideoStat } from "../lib/stats";
-import { statsPageFileName } from "../lib/stats";
-import { idBaseUrl } from "./originId";
-
-async function fetchJson<T>(url: string): Promise<T> {
- const r = await fetch(url);
- if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`);
- return (await r.json()) as T;
-}
+import { readerFor } from "../lib/archive/readers";
export type StatsState = {
manifest: StatsManifest | null;
@@ -24,8 +17,7 @@ export type StatsState = {
export function useStatsManifest(origin = "") {
return useQuery<StatsManifest>({
queryKey: ["stats-manifest", origin],
- queryFn: () =>
- fetchJson<StatsManifest>(`${idBaseUrl(origin)}/stats/manifest.json`),
+ queryFn: () => readerFor(origin).readStatsManifest(),
});
}
@@ -39,10 +31,7 @@ export function useStats(origin = ""): StatsState {
const pageQueries = useQueries({
queries: Array.from({ length: pageCount }, (_, i) => ({
queryKey: ["stats-page", origin, i],
- queryFn: () =>
- fetchJson<VideoStat[]>(
- `${idBaseUrl(origin)}/stats/${statsPageFileName(i)}`,
- ),
+ queryFn: () => readerFor(origin).readStatsPage(i),
enabled: pageCount > 0,
})),
});
diff --git a/common/components/subsCache.ts b/common/components/subsCache.ts
@@ -3,19 +3,17 @@
import { useQueries, useQuery } from "@tanstack/react-query";
import type { SubsDetail } from "../lib/subs";
import type { ChannelSubsManifest, SubsManifest } from "../lib/manifest";
-import { pageFileName } from "../lib/manifest";
-import { makeId, splitId, idBaseUrl } from "./originId";
-
-async function fetchJson<T>(url: string): Promise<T> {
- const r = await fetch(url);
- if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`);
- return (await r.json()) as T;
-}
+import { PromiseMap } from "../lib/archive/reader";
+import { channelRef, readerFor } from "../lib/archive/readers";
+import { makeId, splitId } from "./originId";
const resolved = new Map<string, SubsDetail>();
const inFlight = new Map<string, Promise<SubsDetail>>();
-const channelManifests = new Map<string, Promise<ChannelSubsManifest>>();
-const pagePromises = new Map<string, Promise<SubsDetail[]>>();
+// The reader has a per-channel subs-manifest cache too, but its answer for an
+// unreachable manifest is `null` (the right answer for a scanner probing thirty
+// channels, the wrong one for a UI that wants react-query to retry). So the
+// memo with the THROWING policy stays here, over the reader's raw read.
+const channelManifests = new PromiseMap<ChannelSubsManifest>();
// A per-channel subs reference: channel slug + the origin it lives on
// ("" = same-origin). Used by the multi-origin manifest hook.
@@ -42,33 +40,9 @@ function fetchChannelSubsManifest(
channelSlug: string,
origin = "",
): Promise<ChannelSubsManifest> {
- const key = makeId(origin, channelSlug);
- let p = channelManifests.get(key);
- if (!p) {
- p = fetchJson<ChannelSubsManifest>(
- `${idBaseUrl(origin)}/subs/${channelSlug}/manifest.json`,
- );
- p.catch(() => channelManifests.delete(key));
- channelManifests.set(key, p);
- }
- return p;
-}
-
-function fetchPage(
- channelSlug: string,
- pageIndex: number,
- origin = "",
-): Promise<SubsDetail[]> {
- const key = `${makeId(origin, channelSlug)}:${pageIndex}`;
- let p = pagePromises.get(key);
- if (!p) {
- p = fetchJson<SubsDetail[]>(
- `${idBaseUrl(origin)}/subs/${channelSlug}/${pageFileName(pageIndex)}`,
- );
- p.catch(() => pagePromises.delete(key));
- pagePromises.set(key, p);
- }
- return p;
+ return channelManifests.take(makeId(origin, channelSlug), () =>
+ readerFor(origin).readChannelSubsManifest(channelSlug),
+ );
}
async function load(id: string): Promise<SubsDetail> {
@@ -81,7 +55,10 @@ async function load(id: string): Promise<SubsDetail> {
const pageIndex = manifest.slugToPage[videoId];
if (pageIndex === undefined)
throw new Error(`Unknown subs slug: ${slug}`);
- const page = await fetchPage(channelSlug, pageIndex, origin);
+ const page = await readerFor(origin).subsPage(
+ channelRef(channelSlug, origin),
+ pageIndex,
+ );
let found: SubsDetail | undefined;
for (const entry of page) {
const entryId = makeId(origin, entry.slug);
@@ -95,7 +72,7 @@ async function load(id: string): Promise<SubsDetail> {
export function useSubsManifest(origin = "") {
return useQuery<SubsManifest>({
queryKey: ["subs-manifest", origin],
- queryFn: () => fetchJson<SubsManifest>(`${idBaseUrl(origin)}/subs/manifest.json`),
+ queryFn: () => readerFor(origin).readSubsSiteManifest(),
});
}
diff --git a/common/components/summariesCache.ts b/common/components/summariesCache.ts
@@ -4,14 +4,7 @@ import { useMemo } from "react";
import { useQueries, useQuery } from "@tanstack/react-query";
import type { DisplaySummary } from "../lib/transcripts";
import type { Manifest } from "../lib/manifest";
-import { pageFileName } from "../lib/manifest";
-import { idBaseUrl } from "./originId";
-
-async function fetchJson<T>(url: string): Promise<T> {
- const r = await fetch(url);
- if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`);
- return (await r.json()) as T;
-}
+import { readerFor } from "../lib/archive/readers";
export type SummariesState = {
manifest: Manifest | null;
@@ -25,8 +18,7 @@ export type SummariesState = {
export function useManifest(origin = "") {
return useQuery<Manifest>({
queryKey: ["manifest", origin],
- queryFn: () =>
- fetchJson<Manifest>(`${idBaseUrl(origin)}/summaries/manifest.json`),
+ queryFn: () => readerFor(origin).readSummariesManifest(),
});
}
@@ -38,10 +30,7 @@ export function useSummaries(origin = ""): SummariesState {
const pageQueries = useQueries({
queries: Array.from({ length: pageCount }, (_, i) => ({
queryKey: ["summaries-page", origin, i],
- queryFn: () =>
- fetchJson<DisplaySummary[]>(
- `${idBaseUrl(origin)}/summaries/${pageFileName(i)}`,
- ),
+ queryFn: () => readerFor(origin).readSummariesPage(i),
enabled: pageCount > 0,
})),
});
diff --git a/common/components/transcriptCache.ts b/common/components/transcriptCache.ts
@@ -1,18 +1,17 @@
"use client";
import type { TranscriptDetail } from "../lib/transcripts";
-import type { ChannelTranscriptsManifest } from "../lib/manifest";
-import { pageFileName } from "../lib/manifest";
+import { channelRef, readerFor } from "../lib/archive/readers";
import { idbGet, idbPutBatch } from "./transcriptStore";
-import { makeId, splitId, idBaseUrl } from "./originId";
+import { makeId, splitId } from "./originId";
+// The manifest -> slugToPage -> page walk is the ArchiveReader's
+// (lib/archive/reader.ts), including the per-channel manifest PromiseMap and
+// the byte-budgeted page LRU. What stays here is what is genuinely the
+// viewer's: the per-id memo, the in-flight dedupe, and warming IndexedDB with
+// every record of a page we already paid for.
const resolved = new Map<string, TranscriptDetail>();
const inFlight = new Map<string, Promise<TranscriptDetail>>();
-const channelManifests = new Map<
- string,
- Promise<ChannelTranscriptsManifest>
->();
-const pagePromises = new Map<string, Promise<TranscriptDetail[]>>();
// `id` is an OriginId: a bare "channelSlug/videoId" for same-origin content, or
// "origin\tchannelSlug/videoId" for a federated cross-origin video. Maps and
@@ -35,49 +34,6 @@ export function fetchTranscript(id: string): Promise<TranscriptDetail> {
return p;
}
-function fetchChannelManifest(
- channelSlug: string,
- origin: string,
-): Promise<ChannelTranscriptsManifest> {
- const key = makeId(origin, channelSlug);
- let p = channelManifests.get(key);
- if (!p) {
- p = fetch(`${idBaseUrl(origin)}/transcripts/${channelSlug}/manifest.json`).then((r) => {
- if (!r.ok)
- throw new Error(
- `Failed to fetch transcripts manifest for ${channelSlug}`,
- );
- return r.json() as Promise<ChannelTranscriptsManifest>;
- });
- p.catch(() => channelManifests.delete(key));
- channelManifests.set(key, p);
- }
- return p;
-}
-
-function fetchPage(
- channelSlug: string,
- pageIndex: number,
- origin: string,
-): Promise<TranscriptDetail[]> {
- const key = `${makeId(origin, channelSlug)}:${pageIndex}`;
- let p = pagePromises.get(key);
- if (!p) {
- p = fetch(
- `${idBaseUrl(origin)}/transcripts/${channelSlug}/${pageFileName(pageIndex)}`,
- ).then((r) => {
- if (!r.ok)
- throw new Error(
- `Failed to fetch transcript page ${channelSlug}/${pageIndex}`,
- );
- return r.json() as Promise<TranscriptDetail[]>;
- });
- p.catch(() => pagePromises.delete(key));
- pagePromises.set(key, p);
- }
- return p;
-}
-
async function load(id: string): Promise<TranscriptDetail> {
const stored = await idbGet(id);
if (stored) return stored;
@@ -86,11 +42,13 @@ async function load(id: string): Promise<TranscriptDetail> {
if (slashIdx < 0) throw new Error(`Malformed transcript slug: ${slug}`);
const channelSlug = slug.slice(0, slashIdx);
const videoId = slug.slice(slashIdx + 1);
- const manifest = await fetchChannelManifest(channelSlug, origin);
+ const reader = readerFor(origin);
+ const ch = channelRef(channelSlug, origin);
+ const manifest = await reader.transcriptsManifest(ch);
const pageIndex = manifest.slugToPage[videoId];
if (pageIndex === undefined)
throw new Error(`Unknown transcript slug: ${slug}`);
- const page = await fetchPage(channelSlug, pageIndex, origin);
+ const page = await reader.transcriptPage(ch, pageIndex);
let found: TranscriptDetail | undefined;
for (const entry of page) {
// Re-key warmed entries by their OriginId so a cross-origin
diff --git a/common/lib/archive/contract.ts b/common/lib/archive/contract.ts
@@ -58,6 +58,14 @@ export type ContractLayer = (typeof CONTRACT.layers)[number];
// builders take the wider ArchiveTree and CONTRACT.layers stays frozen.
export type ArchiveTree = ContractLayer | "stats";
+// Every served tree, contract layers plus the undocumented `stats`. This is the
+// list a cache walker enumerates (the offline cache, the two service workers);
+// CONTRACT.layers stays the frozen PUBLISHED list.
+export const ARCHIVE_TREES: readonly ArchiveTree[] = [
+ ...CONTRACT.layers,
+ "stats",
+];
+
// The trees with no per-channel level: one manifest at the tree root and pages
// beside it. Everything else is /<tree>/<slug>/….
const FLAT_TREES: ReadonlySet<string> = new Set(["summaries", "stats"]);
@@ -66,6 +74,23 @@ export function isFlatTree(tree: ArchiveTree): boolean {
return FLAT_TREES.has(tree);
}
+// The per-channel trees: /<tree>/<slug>/…. What a per-channel cache eviction
+// has to sweep, and the half of ARCHIVE_TREES that takes a slug.
+export const PER_CHANNEL_TREES: readonly ArchiveTree[] = ARCHIVE_TREES.filter(
+ (t) => !isFlatTree(t),
+);
+
+// A tree's ROOT manifest, /<tree>/manifest.json.
+//
+// Every tree ships one, and for a per-channel tree it is a DIFFERENT document
+// from the per-channel manifest: /subs/manifest.json is the site-level index of
+// which channels ship live chat (SubsManifest), while /subs/<slug>/manifest.json
+// is that channel's slugToPage. manifestUrl() below builds the second; this
+// builds the first, and for a flat tree the two are the same file.
+export function treeManifestUrl(tree: ArchiveTree, base?: string): string {
+ return archiveUrl(base, `/${tree}/manifest.json`);
+}
+
// THE page-shard file name, for every layer. `page-0.json` is a 404 on every
// published archive, so a copy that lost the padding would 404 silently against
// a real site and pass every unit test. lib/manifest.ts re-exports this (and
@@ -117,7 +142,7 @@ export function manifestUrl(
slug?: string,
base?: string,
): string {
- if (isFlatTree(tree)) return archiveUrl(base, `/${tree}/manifest.json`);
+ if (isFlatTree(tree)) return treeManifestUrl(tree, base);
if (!slug) throw new Error(`manifestUrl(${tree}) needs a channel slug`);
return archiveUrl(base, `/${tree}/${slug}/manifest.json`);
}
diff --git a/common/lib/archive/reader.ts b/common/lib/archive/reader.ts
@@ -19,12 +19,13 @@
import {
type ChannelTranscriptsManifest,
type ChannelSubsManifest,
+ type SubsManifest,
type Manifest,
} from "../manifest";
import type { TranscriptDetail, DisplaySummary } from "../transcripts";
import { summaryState, type VideoState } from "../availability";
import type { SubsDetail } from "../subs";
-import type { ChannelPostsManifest, Post } from "../posts";
+import type { ChannelPostsManifest, Post, PostsManifest } from "../posts";
import { coerceAliasConfig, type SearchAlias } from "../searchAliases";
import {
resolveCanonicalSlug,
@@ -39,7 +40,13 @@ import {
DEFAULT_GROUP_FALLBACK_ID,
type ChannelGroup,
} from "../channelGroups";
-import { manifestUrl, pageUrl, rootFileUrl, type HubSite } from "./contract";
+import {
+ manifestUrl,
+ pageUrl,
+ rootFileUrl,
+ treeManifestUrl,
+ type HubSite,
+} from "./contract";
import { recordRead } from "./io-stats";
// A channel the source can serve. `siteId`/`siteUrl` are only populated in hub
@@ -373,6 +380,23 @@ export interface ArchiveReader {
digestPage?(ch: ChannelRef, page: number): Promise<VideoDigest[]>;
}
+// An HTTP STATUS the archive itself returned, as opposed to a transport
+// failure. A caller that treats absence as data — a site ships no
+// /search-aliases.json, a channel ships no digests — needs to tell the two
+// apart: a 404 is an answer and is worth caching, a dropped connection is
+// neither. `message` is byte-identical to the plain Error this replaced, so
+// anything that only reads the message is unaffected.
+export class ArchiveHttpError extends Error {
+ constructor(
+ readonly status: number,
+ readonly url: string,
+ statusText: string,
+ ) {
+ super(`GET ${url} -> ${status} ${statusText}`);
+ this.name = "ArchiveHttpError";
+ }
+}
+
// Used when a source states no preference. Deliberately modest: each in-flight
// page costs its raw bytes plus ~2.7× that once parsed, and this box is shared.
export const DEFAULT_PAGE_CONCURRENCY = 4;
@@ -590,8 +614,7 @@ export class RemoteSource implements ArchiveReader {
subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
return this.subsManifests.take(ch.slug, async () => {
try {
- const res = await fetch(manifestUrl("subs", ch.slug, this.base));
- return res.ok ? ((await res.json()) as ChannelSubsManifest) : null;
+ return await this.readChannelSubsManifest(ch.slug);
} catch {
return null;
}
@@ -609,8 +632,7 @@ export class RemoteSource implements ArchiveReader {
postsManifest(ch: ChannelRef): Promise<ChannelPostsManifest | null> {
return this.postsManifests.take(ch.slug, async () => {
try {
- const res = await fetch(manifestUrl("posts", ch.slug, this.base));
- return res.ok ? ((await res.json()) as ChannelPostsManifest) : null;
+ return await this.readChannelPostsManifest(ch.slug);
} catch {
return null;
}
@@ -622,15 +644,12 @@ export class RemoteSource implements ArchiveReader {
}
videoIndex(): Promise<VideoIndex> {
+ // buildVideoIndex already treats a throw from either reader as absence
+ // (empty index / skipped page), so the raw reads' rejections land exactly
+ // where the old inline `res.ok ? … : null` did.
this.index ??= buildVideoIndex(
- async () => {
- const res = await fetch(manifestUrl("summaries", undefined, this.base));
- return res.ok ? ((await res.json()) as Manifest) : null;
- },
- async (page) => {
- const res = await fetch(pageUrl("summaries", undefined, page, this.base));
- return res.ok ? ((await res.json()) as DisplaySummary[]) : null;
- },
+ () => this.readSummariesManifest(),
+ (page) => this.readSummariesPage(page),
);
return this.index;
}
@@ -642,8 +661,7 @@ export class RemoteSource implements ArchiveReader {
digestsManifest(ch: ChannelRef): Promise<ChannelDigestsManifest | null> {
return this.digestManifests.take(ch.slug, async () => {
try {
- const res = await fetch(manifestUrl("digests", ch.slug, this.base));
- return res.ok ? ((await res.json()) as ChannelDigestsManifest) : null;
+ return await this.readChannelDigestsManifest(ch.slug);
} catch {
return null;
}
@@ -657,10 +675,7 @@ export class RemoteSource implements ArchiveReader {
duplicateIndex(): Promise<DuplicateIndex> {
this.duplicates ??= (async () => {
try {
- const res = await fetch(rootFileUrl(DUPLICATES_FILENAME, this.base));
- return buildDuplicateIndex(
- res.ok ? ((await res.json()) as DuplicateReport) : null,
- );
+ return buildDuplicateIndex(await this.readDuplicates());
} catch {
return buildDuplicateIndex(null);
}
@@ -670,14 +685,8 @@ export class RemoteSource implements ArchiveReader {
statsIndex(): Promise<ReadonlyMap<string, VideoStat>> {
this.stats ??= buildStatsIndex(
- async () => {
- const res = await fetch(manifestUrl("stats", undefined, this.base));
- return res.ok ? ((await res.json()) as StatsManifest) : null;
- },
- async (page) => {
- const res = await fetch(pageUrl("stats", undefined, page, this.base));
- return res.ok ? ((await res.json()) as VideoStat[]) : null;
- },
+ () => this.readStatsManifest(),
+ (page) => this.readStatsPage(page),
);
return this.stats;
}
@@ -685,10 +694,7 @@ export class RemoteSource implements ArchiveReader {
async loadAliases(): Promise<SearchAlias[]> {
if (this.aliases) return this.aliases;
try {
- const res = await fetch(rootFileUrl("search-aliases.json", this.base));
- this.aliases = res.ok
- ? coerceAliasConfig(await res.json()).aliases
- : [];
+ this.aliases = coerceAliasConfig(await this.readAliasConfig()).aliases;
} catch {
this.aliases = [];
}
@@ -698,10 +704,7 @@ export class RemoteSource implements ArchiveReader {
async loadGroups(): Promise<ChannelGroups> {
if (this.groups) return this.groups;
try {
- const res = await fetch(manifestUrl("summaries", undefined, this.base));
- this.groups = res.ok
- ? parseGroupsManifest(await res.json())
- : EMPTY_GROUPS;
+ this.groups = parseGroupsManifest(await this.readSummariesManifest());
} catch {
this.groups = EMPTY_GROUPS;
}
@@ -713,23 +716,106 @@ export class RemoteSource implements ArchiveReader {
//
// `p` is a ROOT-RELATIVE path from the contract builders; the base is joined
// here so the thrown error names the full URL, exactly as before.
+ //
+ // `kind` is the io-stats bucket. It is OPTIONAL, and the reads that pass none
+ // are the ones that never recorded a read before this file owned them (the
+ // summaries/stats/duplicates/alias folds all used a bare `fetch`). Adding
+ // them to the ledger would move the bench's structural counters for a reason
+ // that has nothing to do with the walk, so the blind spots are preserved
+ // deliberately rather than quietly closed.
private async getJsonSized<T>(
p: string,
- kind: string,
+ kind?: string,
): Promise<{ value: T; bytes: number }> {
const res = await fetch(`${this.base}${p}`);
if (!res.ok) {
- throw new Error(`GET ${this.base}${p} -> ${res.status} ${res.statusText}`);
+ throw new ArchiveHttpError(res.status, `${this.base}${p}`, res.statusText);
}
const raw = await res.text();
- recordRead(kind, raw.length);
+ if (kind !== undefined) recordRead(kind, raw.length);
return { value: JSON.parse(raw) as T, bytes: raw.length };
}
- private async getJson<T>(p: string, kind = "json"): Promise<T> {
+ private async getJson<T>(p: string, kind?: string): Promise<T> {
return (await this.getJsonSized<T>(p, kind)).value;
}
+ // ─── Raw document reads ───
+ //
+ // One URL, one fetch, one parse, and a THROW on anything but a 200. Every
+ // tolerance policy is built on top of these — this class's own
+ // empty-when-absent folds below, and the viewer's component caches — so each
+ // published document's URL is written once and each caller picks its own
+ // answer to "absent".
+ //
+ // That split is not cosmetic: a tool scanning 30 channels wants `null` for a
+ // missing posts tree, while the viewer wants a rejection, because react-query
+ // retries a rejected query and will never retry a resolved `null`. Folding
+ // both into one tolerant method is how a transient network blip becomes a
+ // permanently empty panel.
+
+ readCorpus(): Promise<SiteCorpusJson> {
+ return this.getJson<SiteCorpusJson>(rootFileUrl("corpus.json"), "corpus");
+ }
+
+ readAliasConfig(): Promise<unknown> {
+ return this.getJson<unknown>(rootFileUrl("search-aliases.json"));
+ }
+
+ readDuplicates(): Promise<DuplicateReport> {
+ return this.getJson<DuplicateReport>(rootFileUrl(DUPLICATES_FILENAME));
+ }
+
+ readSummariesManifest(): Promise<Manifest> {
+ return this.getJson<Manifest>(manifestUrl("summaries"));
+ }
+
+ readSummariesPage(page: number): Promise<DisplaySummary[]> {
+ return this.getJson<DisplaySummary[]>(pageUrl("summaries", undefined, page));
+ }
+
+ readStatsManifest(): Promise<StatsManifest> {
+ return this.getJson<StatsManifest>(manifestUrl("stats"));
+ }
+
+ readStatsPage(page: number): Promise<VideoStat[]> {
+ return this.getJson<VideoStat[]>(pageUrl("stats", undefined, page));
+ }
+
+ // The SITE-level index of which channels ship a per-channel tree — a
+ // different document from a channel's own manifest (see treeManifestUrl).
+ readSubsSiteManifest(): Promise<SubsManifest> {
+ return this.getJson<SubsManifest>(treeManifestUrl("subs"));
+ }
+
+ readPostsSiteManifest(): Promise<PostsManifest> {
+ return this.getJson<PostsManifest>(treeManifestUrl("posts"));
+ }
+
+ readChannelSubsManifest(slug: string): Promise<ChannelSubsManifest> {
+ return this.getJson<ChannelSubsManifest>(manifestUrl("subs", slug));
+ }
+
+ readChannelPostsManifest(slug: string): Promise<ChannelPostsManifest> {
+ return this.getJson<ChannelPostsManifest>(manifestUrl("posts", slug));
+ }
+
+ // 404 → null, everything else → throw. The digest corpus is SPARSE BY DESIGN
+ // (a channel with no digests ships no manifest at all), so the 404 is the
+ // document's own answer and belongs in the raw read; a 500 or a dropped
+ // connection is not proof of absence and must stay distinguishable.
+ async readChannelDigestsManifest(
+ slug: string,
+ ): Promise<ChannelDigestsManifest | null> {
+ const p = manifestUrl("digests", slug);
+ const res = await fetch(`${this.base}${p}`);
+ if (res.status === 404) return null;
+ if (!res.ok) {
+ throw new ArchiveHttpError(res.status, `${this.base}${p}`, res.statusText);
+ }
+ return (await res.json()) as ChannelDigestsManifest;
+ }
+
private channelList?: Promise<ChannelRef[]>;
listChannels(opts: { refresh?: boolean } = {}): Promise<ChannelRef[]> {
@@ -745,10 +831,7 @@ export class RemoteSource implements ArchiveReader {
}
private async readChannels(): Promise<ChannelRef[]> {
- const corpus = await this.getJson<SiteCorpusJson>(
- rootFileUrl("corpus.json"),
- "corpus",
- );
+ const corpus = await this.readCorpus();
return (corpus.channels ?? []).map((c) => ({
key: c.slug,
slug: c.slug,
diff --git a/common/lib/archive/readers.ts b/common/lib/archive/readers.ts
@@ -0,0 +1,52 @@
+// One ArchiveReader per origin, for the code that reads an archive from a
+// BROWSER.
+//
+// The viewer's component caches are keyed by origin ("" = same-origin, a full
+// origin like "https://x.example" for a federated hub member — see
+// components/originId.ts). Each one used to carry its own hand-written walk of
+// the shard scheme; now each one calls a RemoteSource from here, so the URL
+// shape is the contract's and the promise-coalescing/LRU behaviour is the
+// reader's rather than eight near-copies of it.
+//
+// WHY THIS LIVES IN lib/archive/ AND NOT components/: it is the reader
+// registry, not a React concern — no hooks, no JSX, no node imports — and a new
+// components/*.ts file would need its own `exports` entry in common's
+// package.json. lib/* is wildcarded, so this file costs nothing to reach from
+// editor/ or export/.
+//
+// RemoteSource with an EMPTY base is the same-origin case: archiveUrl() leaves a
+// path root-relative when there is no base, which is byte-for-byte the
+// `${idBaseUrl(origin)}/transcripts/…` the caches built before. So the
+// single-site export app's URLs do not move at all.
+
+import { RemoteSource, type ChannelRef } from "./reader";
+
+const byOrigin = new Map<string, RemoteSource>();
+
+// The reader for an origin, created on first use and kept for the life of the
+// page — the readers hold the manifest/page caches, so a fresh one per call
+// would be a fresh cache per call.
+export function readerFor(origin = ""): RemoteSource {
+ let reader = byOrigin.get(origin);
+ if (reader === undefined) {
+ reader = new RemoteSource(origin);
+ byOrigin.set(origin, reader);
+ }
+ return reader;
+}
+
+// A minimal ChannelRef for the per-channel reader calls. The viewer addresses a
+// channel by slug alone; `key`/`name` exist for the MCP's channel list and are
+// not read on this path. Reader caches key off `slug`, so a fresh object per
+// call is free.
+export function channelRef(slug: string, origin = ""): ChannelRef {
+ return origin
+ ? { key: slug, slug, name: slug, siteUrl: origin }
+ : { key: slug, slug, name: slug };
+}
+
+// Drop every reader and its caches. For tests, and for any future "the archive
+// was rebuilt under us" reset — not called on a normal page.
+export function resetReaders(): void {
+ byOrigin.clear();
+}
diff --git a/export/app/duplicates/DuplicatesClient.tsx b/export/app/duplicates/DuplicatesClient.tsx
@@ -10,6 +10,7 @@ import { useMediaQuery } from "yt-dlp-transcript-common/lib/useMediaQuery";
import { Checkbox } from "yt-dlp-transcript-common/components/ui/checkbox";
import { VirtualRow } from "yt-dlp-transcript-common/components/VirtualRow";
import { fetchTranscript } from "yt-dlp-transcript-common/components/transcriptCache";
+import { fetchDuplicateReport } from "yt-dlp-transcript-common/components/duplicatesCache";
import { formatDate } from "yt-dlp-transcript-common/lib/format";
import type {
DuplicateCluster,
@@ -102,14 +103,9 @@ export function DuplicatesClient() {
// state rather than an error.
useEffect(() => {
let alive = true;
- fetch("/duplicates.json")
- .then((r) => (r.ok ? r.json() : null))
- .then((report: DuplicateReport | null) => {
- if (alive) setState({ status: "ready", report });
- })
- .catch(() => {
- if (alive) setState({ status: "ready", report: null });
- });
+ fetchDuplicateReport().then((report) => {
+ if (alive) setState({ status: "ready", report });
+ });
return () => {
alive = false;
};