Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit c70b355a074669b0c18de909a8cc925434b67e4b
parent ea2d3d04d126af72cf52b77dfe0d48a78e791d88
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 12 Sep 2026 02:42:59 -0400

common,export: the viewer's archive walks become one ArchiveReader per origin

The eight component caches each carried their own copy of the published
shard walk — manifest -> slugToPage -> page-NNNN.json — spelled as template
literals over `idBaseUrl(origin)`. They now call a RemoteSource from
`lib/archive/readers.ts`, one per origin, so the URL shape is the contract's
and the promise-coalescing and byte-budgeted LRU are the reader's.

What stays in `components/` is what is genuinely the viewer's: the per-id
memo, the in-flight dedupe, the IndexedDB warming, and the two memos whose
ERROR POLICY differs from the reader's. That difference is the reason the
reader gained a block of raw document reads rather than the caches adopting
its tolerant methods: a tool scanning thirty channels wants `null` for a
missing posts tree, but react-query retries a rejected query and will never
retry a resolved `null`, so folding both into one tolerant method is how a
transient blip becomes a permanently empty panel. The raw reads throw; every
tolerance policy — including this file's own empty-when-absent folds, now
rebased on them — sits on top and picks its own answer to "absent".

`ArchiveHttpError` exists for the one caller that needs to tell a 404 from a
dropped connection (the alias dictionary, whose query has staleTime:
Infinity). Its message is byte-identical to the Error it replaced.

No URL moved. `archiveUrl` leaves a path root-relative when there is no base,
so `new RemoteSource("")` emits exactly the root-relative strings the caches
built before, and the export e2e's shard route-mocks still match. The
io-stats blind spots are preserved deliberately: the reads that never
recorded a read (the summaries/stats/duplicates/alias folds) still pass no
`kind`, so the mcp bench's structural counters cannot move for a reason that
has nothing to do with the walk.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/components/SearchDataContext.tsx | 27++++++++-------------------
Mcommon/components/aliasesCache.ts | 16++++++++++++----
Mcommon/components/digestCache.ts | 58++++++++++++++++------------------------------------------
Mcommon/components/duplicatesCache.ts | 22+++++++++++++++-------
Mcommon/components/postsCache.ts | 63++++++++++++++++++++++++++-------------------------------------
Mcommon/components/siteRegistry.ts | 5+++--
Mcommon/components/statsCache.ts | 17+++--------------
Mcommon/components/subsCache.ts | 55++++++++++++++++---------------------------------------
Mcommon/components/summariesCache.ts | 17+++--------------
Mcommon/components/transcriptCache.ts | 64+++++++++++-----------------------------------------------------
Mcommon/lib/archive/contract.ts | 27++++++++++++++++++++++++++-
Mcommon/lib/archive/reader.ts | 171++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------------------
Acommon/lib/archive/readers.ts | 52++++++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/app/duplicates/DuplicatesClient.tsx | 12++++--------
14 files changed, 322 insertions(+), 284 deletions(-)

diff --git a/common/components/SearchDataContext.tsx b/common/components/SearchDataContext.tsx @@ -20,11 +20,11 @@ import { useSubsManifest } from "./subsCache"; import { usePostsManifest } from "./postsCache"; import { useSearchAliases, fetchAliases } from "./aliasesCache"; import { mergeAliases, type SearchAlias } from "../lib/searchAliases"; -import { idBaseUrl, makeId } from "./originId"; +import { makeId } from "./originId"; import type { DisplaySummary } from "../lib/transcripts"; import type { Manifest, SubsManifest } from "../lib/manifest"; import type { PostsManifest } from "../lib/posts"; -import { pageFileName } from "../lib/manifest"; +import { readerFor } from "../lib/archive/readers"; import { DEFAULT_GROUP_FALLBACK_ID, FALLBACK_GROUP, @@ -158,12 +158,6 @@ export function SingleSiteDataProvider({ children }: { children: ReactNode }) { ); } -async function fetchJson<T>(url: string): Promise<T> { - const r = await fetch(url); - if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); - return (await r.json()) as T; -} - // A federated origin the hub reads from. `origin` is "" for the hub's own // same-origin pool, else a full origin ("https://x.com"). `siteTitle` labels // its channel group; `accent` is carried for provenance in the UI. @@ -194,8 +188,7 @@ export function MultiSiteDataProvider({ const manifestQueries = useQueries({ queries: sites.map((s) => ({ queryKey: ["manifest", s.origin], - queryFn: () => - fetchJson<Manifest>(`${idBaseUrl(s.origin)}/summaries/manifest.json`), + queryFn: () => readerFor(s.origin).readSummariesManifest(), })), }); const manifestsSettled = manifestQueries.every( @@ -218,10 +211,7 @@ export function MultiSiteDataProvider({ const pageQueries = useQueries({ queries: pageDescriptors.map((d) => ({ queryKey: ["summaries-page", d.origin, d.index], - queryFn: () => - fetchJson<DisplaySummary[]>( - `${idBaseUrl(d.origin)}/summaries/${pageFileName(d.index)}`, - ), + queryFn: () => readerFor(d.origin).readSummariesPage(d.index), })), }); @@ -300,8 +290,7 @@ export function MultiSiteDataProvider({ const subsQueries = useQueries({ queries: sites.map((s) => ({ queryKey: ["subs-manifest", s.origin], - queryFn: () => - fetchJson<SubsManifest>(`${idBaseUrl(s.origin)}/subs/manifest.json`), + queryFn: () => readerFor(s.origin).readSubsSiteManifest(), })), }); const subsSettled = subsQueries.every((q) => q.isSuccess || q.isError); @@ -339,9 +328,9 @@ export function MultiSiteDataProvider({ queries: sites.map((s) => ({ queryKey: ["posts-manifest", s.origin], queryFn: () => - fetchJson<PostsManifest>( - `${idBaseUrl(s.origin)}/posts/manifest.json`, - ).catch( + readerFor(s.origin) + .readPostsSiteManifest() + .catch( (): PostsManifest => ({ version: 0, channels: [], diff --git a/common/components/aliasesCache.ts b/common/components/aliasesCache.ts @@ -7,15 +7,23 @@ // purely additive, so their absence must never break search. import { useQuery } from "@tanstack/react-query"; -import { idBaseUrl } from "./originId"; +import { ArchiveHttpError } from "../lib/archive/reader"; +import { readerFor } from "../lib/archive/readers"; import { coerceAliasConfig, type SearchAlias } from "../lib/searchAliases"; const EMPTY: SearchAlias[] = []; +// A site that ships no dictionary answers 404, and that is data: resolve empty. +// A TRANSPORT failure is not an answer and is rethrown, so react-query retries +// it rather than caching "this site has no aliases" for the session (the query +// below has staleTime: Infinity, so a swallowed blip would be permanent). export async function fetchAliases(origin = ""): Promise<SearchAlias[]> { - const r = await fetch(`${idBaseUrl(origin)}/search-aliases.json`); - if (!r.ok) return []; - return coerceAliasConfig(await r.json()).aliases; + try { + return coerceAliasConfig(await readerFor(origin).readAliasConfig()).aliases; + } catch (err) { + if (err instanceof ArchiveHttpError) return []; + throw err; + } } export function useSearchAliases(origin = ""): SearchAlias[] { diff --git a/common/components/digestCache.ts b/common/components/digestCache.ts @@ -1,9 +1,11 @@ "use client"; import type { ChannelDigestsManifest, VideoDigest } from "../lib/digests"; -import { digestPageFileName, manifestHasDigest } from "../lib/digests"; +import { manifestHasDigest } from "../lib/digests"; +import { PromiseMap } from "../lib/archive/reader"; +import { channelRef, readerFor } from "../lib/archive/readers"; import { idbGet, idbPut } from "./digestStore"; -import { makeId, splitId, idBaseUrl } from "./originId"; +import { makeId, splitId } from "./originId"; // Fetch layer for the derived corpus, mirroring transcriptCache.ts: a memory // map, in-flight dedupe, manifest -> page resolution and opportunistic warming @@ -23,11 +25,12 @@ import { makeId, splitId, idBaseUrl } from "./originId"; const resolved = new Map<string, VideoDigest | null>(); const inFlight = new Map<string, Promise<VideoDigest | null>>(); -const channelManifests = new Map< - string, - Promise<ChannelDigestsManifest | null> ->(); -const pagePromises = new Map<string, Promise<VideoDigest[]>>(); +// PromiseMap memoises a RESOLVED value (including `null` — the cached negative +// this layer depends on) and drops a REJECTION, which is exactly the policy +// this file hand-rolled: absence is an answer worth keeping, a transport +// failure is not proof of absence. +const channelManifests = new PromiseMap<ChannelDigestsManifest | null>(); +const pagePromises = new PromiseMap<VideoDigest[]>(); // `id` is an OriginId: a bare "channelSlug/videoId" for same-origin content, or // "origin\tchannelSlug/videoId" for a federated cross-origin video. @@ -87,26 +90,9 @@ function fetchChannelManifest( channelSlug: string, origin: string, ): Promise<ChannelDigestsManifest | null> { - const key = makeId(origin, channelSlug); - let p = channelManifests.get(key); - if (!p) { - p = fetch(`${idBaseUrl(origin)}/digests/${channelSlug}/manifest.json`) - .then((r) => { - if (r.status === 404) return null; - if (!r.ok) { - throw new Error(`Failed to fetch digests manifest for ${channelSlug}`); - } - return r.json() as Promise<ChannelDigestsManifest>; - }) - .catch((err) => { - // A transport failure is not proof of absence, so it must not be - // cached as one — drop the entry and let the next caller retry. - channelManifests.delete(key); - throw err; - }); - channelManifests.set(key, p); - } - return p; + return channelManifests.take(makeId(origin, channelSlug), () => + readerFor(origin).readChannelDigestsManifest(channelSlug), + ); } function fetchPage( @@ -114,21 +100,9 @@ function fetchPage( pageIndex: number, origin: string, ): Promise<VideoDigest[]> { - const key = `${makeId(origin, channelSlug)}:${pageIndex}`; - let p = pagePromises.get(key); - if (!p) { - p = fetch( - `${idBaseUrl(origin)}/digests/${channelSlug}/${digestPageFileName(pageIndex)}`, - ).then((r) => { - if (!r.ok) { - throw new Error(`Failed to fetch digest page ${channelSlug}/${pageIndex}`); - } - return r.json() as Promise<VideoDigest[]>; - }); - p.catch(() => pagePromises.delete(key)); - pagePromises.set(key, p); - } - return p; + return pagePromises.take(`${makeId(origin, channelSlug)}:${pageIndex}`, () => + readerFor(origin).digestPage(channelRef(channelSlug, origin), pageIndex), + ); } async function load(id: string): Promise<VideoDigest | null> { diff --git a/common/components/duplicatesCache.ts b/common/components/duplicatesCache.ts @@ -16,7 +16,7 @@ // per member is `aligned` — see duplicateSiblings below. import { useQuery } from "@tanstack/react-query"; -import { idBaseUrl } from "./originId"; +import { readerFor } from "../lib/archive/readers"; import type { DuplicateCluster, DuplicateReport, @@ -27,15 +27,23 @@ export type DuplicateLookup = ReadonlyMap<string, DuplicateCluster>; const EMPTY: DuplicateLookup = new Map(); -export async function fetchDuplicates(origin = ""): Promise<DuplicateLookup> { - let report: DuplicateReport | null = null; +// The raw shipped report, or null when the site ships none (the common case — +// compose-site writes the file only for a site with at least one publishable +// cluster). Absent, unreachable and unparseable all fold to null: every +// duplicate affordance is additive, so none of them may break a page. +export async function fetchDuplicateReport( + origin = "", +): Promise<DuplicateReport | null> { try { - const r = await fetch(`${idBaseUrl(origin)}/duplicates.json`); - if (!r.ok) return EMPTY; - report = (await r.json()) as DuplicateReport; + return await readerFor(origin).readDuplicates(); } catch { - return EMPTY; + return null; } +} + +export async function fetchDuplicates(origin = ""): Promise<DuplicateLookup> { + const report = await fetchDuplicateReport(origin); + if (!report) return EMPTY; const out = new Map<string, DuplicateCluster>(); for (const cluster of report?.clusters ?? []) { for (const ref of cluster?.videoRefs ?? []) { diff --git a/common/components/postsCache.ts b/common/components/postsCache.ts @@ -5,24 +5,21 @@ // federating hub can hold several origins' posts at once without collisions. import { useQueries, useQuery } from "@tanstack/react-query"; -import { - postsPageFileName, - type ChannelPostsManifest, - type Post, - type PostsManifest, -} from "../lib/posts"; -import { makeId, splitId, idBaseUrl } from "./originId"; - -async function fetchJson<T>(url: string): Promise<T> { - const r = await fetch(url); - if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); - return (await r.json()) as T; -} +import type { ChannelPostsManifest, Post, PostsManifest } from "../lib/posts"; +import { PromiseMap } from "../lib/archive/reader"; +import { channelRef, readerFor } from "../lib/archive/readers"; +import { makeId, splitId } from "./originId"; const resolved = new Map<string, Post>(); const inFlight = new Map<string, Promise<Post>>(); -const channelManifests = new Map<string, Promise<ChannelPostsManifest>>(); -const pagePromises = new Map<string, Promise<Post[]>>(); +// Two memos the reader does NOT provide, for two different reasons. The +// manifest one wants a THROW where the reader's tolerant `postsManifest` wants +// a null (see subsCache). The page one exists because the reader caches subs +// and transcript pages but not posts pages, and fetchThread below walks EVERY +// page of a channel — without a memo, opening two posts in a thread would +// re-download the whole channel. +const channelManifests = new PromiseMap<ChannelPostsManifest>(); +const pagePromises = new PromiseMap<Post[]>(); export type PostsManifestRef = { channelSlug: string; origin?: string }; @@ -53,16 +50,9 @@ export function fetchChannelPostsManifest( channelSlug: string, origin = "", ): Promise<ChannelPostsManifest> { - const key = makeId(origin, channelSlug); - let p = channelManifests.get(key); - if (!p) { - p = fetchJson<ChannelPostsManifest>( - `${idBaseUrl(origin)}/posts/${channelSlug}/manifest.json`, - ); - p.catch(() => channelManifests.delete(key)); - channelManifests.set(key, p); - } - return p; + return channelManifests.take(makeId(origin, channelSlug), () => + readerFor(origin).readChannelPostsManifest(channelSlug), + ); } export function fetchPostsPage( @@ -70,23 +60,21 @@ export function fetchPostsPage( pageIndex: number, origin = "", ): Promise<Post[]> { - const key = `${makeId(origin, channelSlug)}:${pageIndex}`; - let p = pagePromises.get(key); - if (!p) { - p = fetchJson<Post[]>( - `${idBaseUrl(origin)}/posts/${channelSlug}/${postsPageFileName(pageIndex)}`, - ).then((page) => { + return pagePromises.take( + `${makeId(origin, channelSlug)}:${pageIndex}`, + async () => { + const page = await readerFor(origin).postsPage( + channelRef(channelSlug, origin), + pageIndex, + ); // Warm the by-id cache so a later fetchPost() for any post on this page // is synchronous. for (const entry of page) { resolved.set(makeId(origin, entry.slug), entry); } return page; - }); - p.catch(() => pagePromises.delete(key)); - pagePromises.set(key, p); - } - return p; + }, + ); } async function load(id: string): Promise<Post> { @@ -136,7 +124,8 @@ export function usePostsManifest(origin = "") { return useQuery<PostsManifest>({ queryKey: ["posts-manifest", origin], queryFn: () => - fetchJson<PostsManifest>(`${idBaseUrl(origin)}/posts/manifest.json`) + readerFor(origin) + .readPostsSiteManifest() // A site with no social channels ships no posts manifest; treat that as // an empty corpus rather than an error, so the UI degrades quietly. .catch( diff --git a/common/components/siteRegistry.ts b/common/components/siteRegistry.ts @@ -19,6 +19,7 @@ // and stay in sync when a site is added or removed. import { useEffect, useSyncExternalStore } from "react"; +import { manifestUrl, rootFileUrl } from "../lib/archive/contract"; import { SITE_DESCRIPTOR_VERSION } from "../lib/siteDescriptor"; import type { PublicSiteDescriptor } from "../lib/siteDescriptor"; @@ -259,7 +260,7 @@ export async function validateSite(input: string): Promise<ValidateResult> { let descriptor: PublicSiteDescriptor; try { - const res = await fetch(`${origin}/site.json`); + const res = await fetch(rootFileUrl("site.json", origin)); if (!res.ok) return { ok: false, reason: UNREADABLE }; descriptor = (await res.json()) as PublicSiteDescriptor; } catch { @@ -284,7 +285,7 @@ export async function validateSite(input: string): Promise<ValidateResult> { // whose descriptor is fine but whose data isn't reachable/valid would fail // silently later, so reject it now with a clear message. try { - const res = await fetch(`${origin}/summaries/manifest.json`); + const res = await fetch(manifestUrl("summaries", undefined, origin)); if (!res.ok) return { ok: false, reason: UNREADABLE }; const manifest = (await res.json()) as { version?: unknown; channels?: unknown }; if ( diff --git a/common/components/statsCache.ts b/common/components/statsCache.ts @@ -3,14 +3,7 @@ import { useMemo } from "react"; import { useQueries, useQuery } from "@tanstack/react-query"; import type { StatsManifest, VideoStat } from "../lib/stats"; -import { statsPageFileName } from "../lib/stats"; -import { idBaseUrl } from "./originId"; - -async function fetchJson<T>(url: string): Promise<T> { - const r = await fetch(url); - if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); - return (await r.json()) as T; -} +import { readerFor } from "../lib/archive/readers"; export type StatsState = { manifest: StatsManifest | null; @@ -24,8 +17,7 @@ export type StatsState = { export function useStatsManifest(origin = "") { return useQuery<StatsManifest>({ queryKey: ["stats-manifest", origin], - queryFn: () => - fetchJson<StatsManifest>(`${idBaseUrl(origin)}/stats/manifest.json`), + queryFn: () => readerFor(origin).readStatsManifest(), }); } @@ -39,10 +31,7 @@ export function useStats(origin = ""): StatsState { const pageQueries = useQueries({ queries: Array.from({ length: pageCount }, (_, i) => ({ queryKey: ["stats-page", origin, i], - queryFn: () => - fetchJson<VideoStat[]>( - `${idBaseUrl(origin)}/stats/${statsPageFileName(i)}`, - ), + queryFn: () => readerFor(origin).readStatsPage(i), enabled: pageCount > 0, })), }); diff --git a/common/components/subsCache.ts b/common/components/subsCache.ts @@ -3,19 +3,17 @@ import { useQueries, useQuery } from "@tanstack/react-query"; import type { SubsDetail } from "../lib/subs"; import type { ChannelSubsManifest, SubsManifest } from "../lib/manifest"; -import { pageFileName } from "../lib/manifest"; -import { makeId, splitId, idBaseUrl } from "./originId"; - -async function fetchJson<T>(url: string): Promise<T> { - const r = await fetch(url); - if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); - return (await r.json()) as T; -} +import { PromiseMap } from "../lib/archive/reader"; +import { channelRef, readerFor } from "../lib/archive/readers"; +import { makeId, splitId } from "./originId"; const resolved = new Map<string, SubsDetail>(); const inFlight = new Map<string, Promise<SubsDetail>>(); -const channelManifests = new Map<string, Promise<ChannelSubsManifest>>(); -const pagePromises = new Map<string, Promise<SubsDetail[]>>(); +// The reader has a per-channel subs-manifest cache too, but its answer for an +// unreachable manifest is `null` (the right answer for a scanner probing thirty +// channels, the wrong one for a UI that wants react-query to retry). So the +// memo with the THROWING policy stays here, over the reader's raw read. +const channelManifests = new PromiseMap<ChannelSubsManifest>(); // A per-channel subs reference: channel slug + the origin it lives on // ("" = same-origin). Used by the multi-origin manifest hook. @@ -42,33 +40,9 @@ function fetchChannelSubsManifest( channelSlug: string, origin = "", ): Promise<ChannelSubsManifest> { - const key = makeId(origin, channelSlug); - let p = channelManifests.get(key); - if (!p) { - p = fetchJson<ChannelSubsManifest>( - `${idBaseUrl(origin)}/subs/${channelSlug}/manifest.json`, - ); - p.catch(() => channelManifests.delete(key)); - channelManifests.set(key, p); - } - return p; -} - -function fetchPage( - channelSlug: string, - pageIndex: number, - origin = "", -): Promise<SubsDetail[]> { - const key = `${makeId(origin, channelSlug)}:${pageIndex}`; - let p = pagePromises.get(key); - if (!p) { - p = fetchJson<SubsDetail[]>( - `${idBaseUrl(origin)}/subs/${channelSlug}/${pageFileName(pageIndex)}`, - ); - p.catch(() => pagePromises.delete(key)); - pagePromises.set(key, p); - } - return p; + return channelManifests.take(makeId(origin, channelSlug), () => + readerFor(origin).readChannelSubsManifest(channelSlug), + ); } async function load(id: string): Promise<SubsDetail> { @@ -81,7 +55,10 @@ async function load(id: string): Promise<SubsDetail> { const pageIndex = manifest.slugToPage[videoId]; if (pageIndex === undefined) throw new Error(`Unknown subs slug: ${slug}`); - const page = await fetchPage(channelSlug, pageIndex, origin); + const page = await readerFor(origin).subsPage( + channelRef(channelSlug, origin), + pageIndex, + ); let found: SubsDetail | undefined; for (const entry of page) { const entryId = makeId(origin, entry.slug); @@ -95,7 +72,7 @@ async function load(id: string): Promise<SubsDetail> { export function useSubsManifest(origin = "") { return useQuery<SubsManifest>({ queryKey: ["subs-manifest", origin], - queryFn: () => fetchJson<SubsManifest>(`${idBaseUrl(origin)}/subs/manifest.json`), + queryFn: () => readerFor(origin).readSubsSiteManifest(), }); } diff --git a/common/components/summariesCache.ts b/common/components/summariesCache.ts @@ -4,14 +4,7 @@ import { useMemo } from "react"; import { useQueries, useQuery } from "@tanstack/react-query"; import type { DisplaySummary } from "../lib/transcripts"; import type { Manifest } from "../lib/manifest"; -import { pageFileName } from "../lib/manifest"; -import { idBaseUrl } from "./originId"; - -async function fetchJson<T>(url: string): Promise<T> { - const r = await fetch(url); - if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); - return (await r.json()) as T; -} +import { readerFor } from "../lib/archive/readers"; export type SummariesState = { manifest: Manifest | null; @@ -25,8 +18,7 @@ export type SummariesState = { export function useManifest(origin = "") { return useQuery<Manifest>({ queryKey: ["manifest", origin], - queryFn: () => - fetchJson<Manifest>(`${idBaseUrl(origin)}/summaries/manifest.json`), + queryFn: () => readerFor(origin).readSummariesManifest(), }); } @@ -38,10 +30,7 @@ export function useSummaries(origin = ""): SummariesState { const pageQueries = useQueries({ queries: Array.from({ length: pageCount }, (_, i) => ({ queryKey: ["summaries-page", origin, i], - queryFn: () => - fetchJson<DisplaySummary[]>( - `${idBaseUrl(origin)}/summaries/${pageFileName(i)}`, - ), + queryFn: () => readerFor(origin).readSummariesPage(i), enabled: pageCount > 0, })), }); diff --git a/common/components/transcriptCache.ts b/common/components/transcriptCache.ts @@ -1,18 +1,17 @@ "use client"; import type { TranscriptDetail } from "../lib/transcripts"; -import type { ChannelTranscriptsManifest } from "../lib/manifest"; -import { pageFileName } from "../lib/manifest"; +import { channelRef, readerFor } from "../lib/archive/readers"; import { idbGet, idbPutBatch } from "./transcriptStore"; -import { makeId, splitId, idBaseUrl } from "./originId"; +import { makeId, splitId } from "./originId"; +// The manifest -> slugToPage -> page walk is the ArchiveReader's +// (lib/archive/reader.ts), including the per-channel manifest PromiseMap and +// the byte-budgeted page LRU. What stays here is what is genuinely the +// viewer's: the per-id memo, the in-flight dedupe, and warming IndexedDB with +// every record of a page we already paid for. const resolved = new Map<string, TranscriptDetail>(); const inFlight = new Map<string, Promise<TranscriptDetail>>(); -const channelManifests = new Map< - string, - Promise<ChannelTranscriptsManifest> ->(); -const pagePromises = new Map<string, Promise<TranscriptDetail[]>>(); // `id` is an OriginId: a bare "channelSlug/videoId" for same-origin content, or // "origin\tchannelSlug/videoId" for a federated cross-origin video. Maps and @@ -35,49 +34,6 @@ export function fetchTranscript(id: string): Promise<TranscriptDetail> { return p; } -function fetchChannelManifest( - channelSlug: string, - origin: string, -): Promise<ChannelTranscriptsManifest> { - const key = makeId(origin, channelSlug); - let p = channelManifests.get(key); - if (!p) { - p = fetch(`${idBaseUrl(origin)}/transcripts/${channelSlug}/manifest.json`).then((r) => { - if (!r.ok) - throw new Error( - `Failed to fetch transcripts manifest for ${channelSlug}`, - ); - return r.json() as Promise<ChannelTranscriptsManifest>; - }); - p.catch(() => channelManifests.delete(key)); - channelManifests.set(key, p); - } - return p; -} - -function fetchPage( - channelSlug: string, - pageIndex: number, - origin: string, -): Promise<TranscriptDetail[]> { - const key = `${makeId(origin, channelSlug)}:${pageIndex}`; - let p = pagePromises.get(key); - if (!p) { - p = fetch( - `${idBaseUrl(origin)}/transcripts/${channelSlug}/${pageFileName(pageIndex)}`, - ).then((r) => { - if (!r.ok) - throw new Error( - `Failed to fetch transcript page ${channelSlug}/${pageIndex}`, - ); - return r.json() as Promise<TranscriptDetail[]>; - }); - p.catch(() => pagePromises.delete(key)); - pagePromises.set(key, p); - } - return p; -} - async function load(id: string): Promise<TranscriptDetail> { const stored = await idbGet(id); if (stored) return stored; @@ -86,11 +42,13 @@ async function load(id: string): Promise<TranscriptDetail> { if (slashIdx < 0) throw new Error(`Malformed transcript slug: ${slug}`); const channelSlug = slug.slice(0, slashIdx); const videoId = slug.slice(slashIdx + 1); - const manifest = await fetchChannelManifest(channelSlug, origin); + const reader = readerFor(origin); + const ch = channelRef(channelSlug, origin); + const manifest = await reader.transcriptsManifest(ch); const pageIndex = manifest.slugToPage[videoId]; if (pageIndex === undefined) throw new Error(`Unknown transcript slug: ${slug}`); - const page = await fetchPage(channelSlug, pageIndex, origin); + const page = await reader.transcriptPage(ch, pageIndex); let found: TranscriptDetail | undefined; for (const entry of page) { // Re-key warmed entries by their OriginId so a cross-origin diff --git a/common/lib/archive/contract.ts b/common/lib/archive/contract.ts @@ -58,6 +58,14 @@ export type ContractLayer = (typeof CONTRACT.layers)[number]; // builders take the wider ArchiveTree and CONTRACT.layers stays frozen. export type ArchiveTree = ContractLayer | "stats"; +// Every served tree, contract layers plus the undocumented `stats`. This is the +// list a cache walker enumerates (the offline cache, the two service workers); +// CONTRACT.layers stays the frozen PUBLISHED list. +export const ARCHIVE_TREES: readonly ArchiveTree[] = [ + ...CONTRACT.layers, + "stats", +]; + // The trees with no per-channel level: one manifest at the tree root and pages // beside it. Everything else is /<tree>/<slug>/…. const FLAT_TREES: ReadonlySet<string> = new Set(["summaries", "stats"]); @@ -66,6 +74,23 @@ export function isFlatTree(tree: ArchiveTree): boolean { return FLAT_TREES.has(tree); } +// The per-channel trees: /<tree>/<slug>/…. What a per-channel cache eviction +// has to sweep, and the half of ARCHIVE_TREES that takes a slug. +export const PER_CHANNEL_TREES: readonly ArchiveTree[] = ARCHIVE_TREES.filter( + (t) => !isFlatTree(t), +); + +// A tree's ROOT manifest, /<tree>/manifest.json. +// +// Every tree ships one, and for a per-channel tree it is a DIFFERENT document +// from the per-channel manifest: /subs/manifest.json is the site-level index of +// which channels ship live chat (SubsManifest), while /subs/<slug>/manifest.json +// is that channel's slugToPage. manifestUrl() below builds the second; this +// builds the first, and for a flat tree the two are the same file. +export function treeManifestUrl(tree: ArchiveTree, base?: string): string { + return archiveUrl(base, `/${tree}/manifest.json`); +} + // THE page-shard file name, for every layer. `page-0.json` is a 404 on every // published archive, so a copy that lost the padding would 404 silently against // a real site and pass every unit test. lib/manifest.ts re-exports this (and @@ -117,7 +142,7 @@ export function manifestUrl( slug?: string, base?: string, ): string { - if (isFlatTree(tree)) return archiveUrl(base, `/${tree}/manifest.json`); + if (isFlatTree(tree)) return treeManifestUrl(tree, base); if (!slug) throw new Error(`manifestUrl(${tree}) needs a channel slug`); return archiveUrl(base, `/${tree}/${slug}/manifest.json`); } diff --git a/common/lib/archive/reader.ts b/common/lib/archive/reader.ts @@ -19,12 +19,13 @@ import { type ChannelTranscriptsManifest, type ChannelSubsManifest, + type SubsManifest, type Manifest, } from "../manifest"; import type { TranscriptDetail, DisplaySummary } from "../transcripts"; import { summaryState, type VideoState } from "../availability"; import type { SubsDetail } from "../subs"; -import type { ChannelPostsManifest, Post } from "../posts"; +import type { ChannelPostsManifest, Post, PostsManifest } from "../posts"; import { coerceAliasConfig, type SearchAlias } from "../searchAliases"; import { resolveCanonicalSlug, @@ -39,7 +40,13 @@ import { DEFAULT_GROUP_FALLBACK_ID, type ChannelGroup, } from "../channelGroups"; -import { manifestUrl, pageUrl, rootFileUrl, type HubSite } from "./contract"; +import { + manifestUrl, + pageUrl, + rootFileUrl, + treeManifestUrl, + type HubSite, +} from "./contract"; import { recordRead } from "./io-stats"; // A channel the source can serve. `siteId`/`siteUrl` are only populated in hub @@ -373,6 +380,23 @@ export interface ArchiveReader { digestPage?(ch: ChannelRef, page: number): Promise<VideoDigest[]>; } +// An HTTP STATUS the archive itself returned, as opposed to a transport +// failure. A caller that treats absence as data — a site ships no +// /search-aliases.json, a channel ships no digests — needs to tell the two +// apart: a 404 is an answer and is worth caching, a dropped connection is +// neither. `message` is byte-identical to the plain Error this replaced, so +// anything that only reads the message is unaffected. +export class ArchiveHttpError extends Error { + constructor( + readonly status: number, + readonly url: string, + statusText: string, + ) { + super(`GET ${url} -> ${status} ${statusText}`); + this.name = "ArchiveHttpError"; + } +} + // Used when a source states no preference. Deliberately modest: each in-flight // page costs its raw bytes plus ~2.7× that once parsed, and this box is shared. export const DEFAULT_PAGE_CONCURRENCY = 4; @@ -590,8 +614,7 @@ export class RemoteSource implements ArchiveReader { subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { return this.subsManifests.take(ch.slug, async () => { try { - const res = await fetch(manifestUrl("subs", ch.slug, this.base)); - return res.ok ? ((await res.json()) as ChannelSubsManifest) : null; + return await this.readChannelSubsManifest(ch.slug); } catch { return null; } @@ -609,8 +632,7 @@ export class RemoteSource implements ArchiveReader { postsManifest(ch: ChannelRef): Promise<ChannelPostsManifest | null> { return this.postsManifests.take(ch.slug, async () => { try { - const res = await fetch(manifestUrl("posts", ch.slug, this.base)); - return res.ok ? ((await res.json()) as ChannelPostsManifest) : null; + return await this.readChannelPostsManifest(ch.slug); } catch { return null; } @@ -622,15 +644,12 @@ export class RemoteSource implements ArchiveReader { } videoIndex(): Promise<VideoIndex> { + // buildVideoIndex already treats a throw from either reader as absence + // (empty index / skipped page), so the raw reads' rejections land exactly + // where the old inline `res.ok ? … : null` did. this.index ??= buildVideoIndex( - async () => { - const res = await fetch(manifestUrl("summaries", undefined, this.base)); - return res.ok ? ((await res.json()) as Manifest) : null; - }, - async (page) => { - const res = await fetch(pageUrl("summaries", undefined, page, this.base)); - return res.ok ? ((await res.json()) as DisplaySummary[]) : null; - }, + () => this.readSummariesManifest(), + (page) => this.readSummariesPage(page), ); return this.index; } @@ -642,8 +661,7 @@ export class RemoteSource implements ArchiveReader { digestsManifest(ch: ChannelRef): Promise<ChannelDigestsManifest | null> { return this.digestManifests.take(ch.slug, async () => { try { - const res = await fetch(manifestUrl("digests", ch.slug, this.base)); - return res.ok ? ((await res.json()) as ChannelDigestsManifest) : null; + return await this.readChannelDigestsManifest(ch.slug); } catch { return null; } @@ -657,10 +675,7 @@ export class RemoteSource implements ArchiveReader { duplicateIndex(): Promise<DuplicateIndex> { this.duplicates ??= (async () => { try { - const res = await fetch(rootFileUrl(DUPLICATES_FILENAME, this.base)); - return buildDuplicateIndex( - res.ok ? ((await res.json()) as DuplicateReport) : null, - ); + return buildDuplicateIndex(await this.readDuplicates()); } catch { return buildDuplicateIndex(null); } @@ -670,14 +685,8 @@ export class RemoteSource implements ArchiveReader { statsIndex(): Promise<ReadonlyMap<string, VideoStat>> { this.stats ??= buildStatsIndex( - async () => { - const res = await fetch(manifestUrl("stats", undefined, this.base)); - return res.ok ? ((await res.json()) as StatsManifest) : null; - }, - async (page) => { - const res = await fetch(pageUrl("stats", undefined, page, this.base)); - return res.ok ? ((await res.json()) as VideoStat[]) : null; - }, + () => this.readStatsManifest(), + (page) => this.readStatsPage(page), ); return this.stats; } @@ -685,10 +694,7 @@ export class RemoteSource implements ArchiveReader { async loadAliases(): Promise<SearchAlias[]> { if (this.aliases) return this.aliases; try { - const res = await fetch(rootFileUrl("search-aliases.json", this.base)); - this.aliases = res.ok - ? coerceAliasConfig(await res.json()).aliases - : []; + this.aliases = coerceAliasConfig(await this.readAliasConfig()).aliases; } catch { this.aliases = []; } @@ -698,10 +704,7 @@ export class RemoteSource implements ArchiveReader { async loadGroups(): Promise<ChannelGroups> { if (this.groups) return this.groups; try { - const res = await fetch(manifestUrl("summaries", undefined, this.base)); - this.groups = res.ok - ? parseGroupsManifest(await res.json()) - : EMPTY_GROUPS; + this.groups = parseGroupsManifest(await this.readSummariesManifest()); } catch { this.groups = EMPTY_GROUPS; } @@ -713,23 +716,106 @@ export class RemoteSource implements ArchiveReader { // // `p` is a ROOT-RELATIVE path from the contract builders; the base is joined // here so the thrown error names the full URL, exactly as before. + // + // `kind` is the io-stats bucket. It is OPTIONAL, and the reads that pass none + // are the ones that never recorded a read before this file owned them (the + // summaries/stats/duplicates/alias folds all used a bare `fetch`). Adding + // them to the ledger would move the bench's structural counters for a reason + // that has nothing to do with the walk, so the blind spots are preserved + // deliberately rather than quietly closed. private async getJsonSized<T>( p: string, - kind: string, + kind?: string, ): Promise<{ value: T; bytes: number }> { const res = await fetch(`${this.base}${p}`); if (!res.ok) { - throw new Error(`GET ${this.base}${p} -> ${res.status} ${res.statusText}`); + throw new ArchiveHttpError(res.status, `${this.base}${p}`, res.statusText); } const raw = await res.text(); - recordRead(kind, raw.length); + if (kind !== undefined) recordRead(kind, raw.length); return { value: JSON.parse(raw) as T, bytes: raw.length }; } - private async getJson<T>(p: string, kind = "json"): Promise<T> { + private async getJson<T>(p: string, kind?: string): Promise<T> { return (await this.getJsonSized<T>(p, kind)).value; } + // ─── Raw document reads ─── + // + // One URL, one fetch, one parse, and a THROW on anything but a 200. Every + // tolerance policy is built on top of these — this class's own + // empty-when-absent folds below, and the viewer's component caches — so each + // published document's URL is written once and each caller picks its own + // answer to "absent". + // + // That split is not cosmetic: a tool scanning 30 channels wants `null` for a + // missing posts tree, while the viewer wants a rejection, because react-query + // retries a rejected query and will never retry a resolved `null`. Folding + // both into one tolerant method is how a transient network blip becomes a + // permanently empty panel. + + readCorpus(): Promise<SiteCorpusJson> { + return this.getJson<SiteCorpusJson>(rootFileUrl("corpus.json"), "corpus"); + } + + readAliasConfig(): Promise<unknown> { + return this.getJson<unknown>(rootFileUrl("search-aliases.json")); + } + + readDuplicates(): Promise<DuplicateReport> { + return this.getJson<DuplicateReport>(rootFileUrl(DUPLICATES_FILENAME)); + } + + readSummariesManifest(): Promise<Manifest> { + return this.getJson<Manifest>(manifestUrl("summaries")); + } + + readSummariesPage(page: number): Promise<DisplaySummary[]> { + return this.getJson<DisplaySummary[]>(pageUrl("summaries", undefined, page)); + } + + readStatsManifest(): Promise<StatsManifest> { + return this.getJson<StatsManifest>(manifestUrl("stats")); + } + + readStatsPage(page: number): Promise<VideoStat[]> { + return this.getJson<VideoStat[]>(pageUrl("stats", undefined, page)); + } + + // The SITE-level index of which channels ship a per-channel tree — a + // different document from a channel's own manifest (see treeManifestUrl). + readSubsSiteManifest(): Promise<SubsManifest> { + return this.getJson<SubsManifest>(treeManifestUrl("subs")); + } + + readPostsSiteManifest(): Promise<PostsManifest> { + return this.getJson<PostsManifest>(treeManifestUrl("posts")); + } + + readChannelSubsManifest(slug: string): Promise<ChannelSubsManifest> { + return this.getJson<ChannelSubsManifest>(manifestUrl("subs", slug)); + } + + readChannelPostsManifest(slug: string): Promise<ChannelPostsManifest> { + return this.getJson<ChannelPostsManifest>(manifestUrl("posts", slug)); + } + + // 404 → null, everything else → throw. The digest corpus is SPARSE BY DESIGN + // (a channel with no digests ships no manifest at all), so the 404 is the + // document's own answer and belongs in the raw read; a 500 or a dropped + // connection is not proof of absence and must stay distinguishable. + async readChannelDigestsManifest( + slug: string, + ): Promise<ChannelDigestsManifest | null> { + const p = manifestUrl("digests", slug); + const res = await fetch(`${this.base}${p}`); + if (res.status === 404) return null; + if (!res.ok) { + throw new ArchiveHttpError(res.status, `${this.base}${p}`, res.statusText); + } + return (await res.json()) as ChannelDigestsManifest; + } + private channelList?: Promise<ChannelRef[]>; listChannels(opts: { refresh?: boolean } = {}): Promise<ChannelRef[]> { @@ -745,10 +831,7 @@ export class RemoteSource implements ArchiveReader { } private async readChannels(): Promise<ChannelRef[]> { - const corpus = await this.getJson<SiteCorpusJson>( - rootFileUrl("corpus.json"), - "corpus", - ); + const corpus = await this.readCorpus(); return (corpus.channels ?? []).map((c) => ({ key: c.slug, slug: c.slug, diff --git a/common/lib/archive/readers.ts b/common/lib/archive/readers.ts @@ -0,0 +1,52 @@ +// One ArchiveReader per origin, for the code that reads an archive from a +// BROWSER. +// +// The viewer's component caches are keyed by origin ("" = same-origin, a full +// origin like "https://x.example" for a federated hub member — see +// components/originId.ts). Each one used to carry its own hand-written walk of +// the shard scheme; now each one calls a RemoteSource from here, so the URL +// shape is the contract's and the promise-coalescing/LRU behaviour is the +// reader's rather than eight near-copies of it. +// +// WHY THIS LIVES IN lib/archive/ AND NOT components/: it is the reader +// registry, not a React concern — no hooks, no JSX, no node imports — and a new +// components/*.ts file would need its own `exports` entry in common's +// package.json. lib/* is wildcarded, so this file costs nothing to reach from +// editor/ or export/. +// +// RemoteSource with an EMPTY base is the same-origin case: archiveUrl() leaves a +// path root-relative when there is no base, which is byte-for-byte the +// `${idBaseUrl(origin)}/transcripts/…` the caches built before. So the +// single-site export app's URLs do not move at all. + +import { RemoteSource, type ChannelRef } from "./reader"; + +const byOrigin = new Map<string, RemoteSource>(); + +// The reader for an origin, created on first use and kept for the life of the +// page — the readers hold the manifest/page caches, so a fresh one per call +// would be a fresh cache per call. +export function readerFor(origin = ""): RemoteSource { + let reader = byOrigin.get(origin); + if (reader === undefined) { + reader = new RemoteSource(origin); + byOrigin.set(origin, reader); + } + return reader; +} + +// A minimal ChannelRef for the per-channel reader calls. The viewer addresses a +// channel by slug alone; `key`/`name` exist for the MCP's channel list and are +// not read on this path. Reader caches key off `slug`, so a fresh object per +// call is free. +export function channelRef(slug: string, origin = ""): ChannelRef { + return origin + ? { key: slug, slug, name: slug, siteUrl: origin } + : { key: slug, slug, name: slug }; +} + +// Drop every reader and its caches. For tests, and for any future "the archive +// was rebuilt under us" reset — not called on a normal page. +export function resetReaders(): void { + byOrigin.clear(); +} diff --git a/export/app/duplicates/DuplicatesClient.tsx b/export/app/duplicates/DuplicatesClient.tsx @@ -10,6 +10,7 @@ import { useMediaQuery } from "yt-dlp-transcript-common/lib/useMediaQuery"; import { Checkbox } from "yt-dlp-transcript-common/components/ui/checkbox"; import { VirtualRow } from "yt-dlp-transcript-common/components/VirtualRow"; import { fetchTranscript } from "yt-dlp-transcript-common/components/transcriptCache"; +import { fetchDuplicateReport } from "yt-dlp-transcript-common/components/duplicatesCache"; import { formatDate } from "yt-dlp-transcript-common/lib/format"; import type { DuplicateCluster, @@ -102,14 +103,9 @@ export function DuplicatesClient() { // state rather than an error. useEffect(() => { let alive = true; - fetch("/duplicates.json") - .then((r) => (r.ok ? r.json() : null)) - .then((report: DuplicateReport | null) => { - if (alive) setState({ status: "ready", report }); - }) - .catch(() => { - if (alive) setState({ status: "ready", report: null }); - }); + fetchDuplicateReport().then((report) => { + if (alive) setState({ status: "ready", report }); + }); return () => { alive = false; };