Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit d8de0fd14841b55a8012744cd7cd9e0357334e21
parent 701c83c28cd777290f5e07aa230c625c7375aff9
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri, 25 Sep 2026 20:13:15 -0400

common: MultiSiteDataProvider exposes per-archive state — scope, loading/ready/failed, Retry

`sites[].enabled` false takes an archive out of the search (none of its feeds
fetched, nothing cached merged). `federation.sites` carries each archive's
status (off/loading/ready/failed), transcript count and error; `retry(origin)`
refetches whatever of it failed. An archive's records join the merged list only
once all its pages are in; `progressive` turns summariesReady on at the first
ready archive (the hub page), otherwise it waits until every in-scope archive
has settled (the /ask chat). Pages are fetched per archive only after its
manifest, at most 6 at a time per origin. `summariesState.error` is set when
every in-scope archive failed. `siteTitleOf` names a result's source.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/components/SearchDataContext.tsx | 333++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-------------
1 file changed, 281 insertions(+), 52 deletions(-)

diff --git a/common/components/SearchDataContext.tsx b/common/components/SearchDataContext.tsx @@ -14,7 +14,7 @@ import { useMemo, type ReactNode, } from "react"; -import { useQueries } from "@tanstack/react-query"; +import { useQueries, useQueryClient } from "@tanstack/react-query"; import { useSummaries, type SummariesState } from "./summariesCache"; import { useSubsManifest } from "./subsCache"; import { usePostsManifest } from "./postsCache"; @@ -77,6 +77,41 @@ export type SearchDataValue = { // both modes: a hub's counts are the hub's, and summing member documents here // would put a number in front of someone that no site can reproduce. curatedTags: PublishedTag[]; + // Hub mode only (absent single-site): the per-archive state of the + // federation — which archives are in scope, loading, ready or failed — and a + // Retry per archive. Drives the scope chips and the "N of M archives + // answered." line. + federation?: FederationState; + // Hub mode only: the source archive's title for a content origin, so a result + // card can name where it came from in text, not by its accent alone. + siteTitleOf?: (origin: string) => string | undefined; +}; + +// One federated archive's state, as the hub shows it on its scope chip. +// off — the visitor took it out of the search; nothing is fetched. +// loading — its manifest or pages are still arriving (or being retried). +// ready — every summaries page arrived; its records are in the search. +// failed — its manifest or a summaries page could not be read; its records +// are NOT in the search until a Retry succeeds. +export type FederatedSiteStatus = "loading" | "ready" | "failed" | "off"; + +export type FederatedSiteState = { + origin: string; + siteTitle: string; + accent?: string; + status: FederatedSiteStatus; + // The archive's transcript count (its summaries manifest's totalCount), once + // the manifest has arrived. + count?: number; + // Why it failed, when it did. + error?: string; +}; + +export type FederationState = { + // In registry order, off archives included. + sites: FederatedSiteState[]; + // Refetch everything of this archive that failed. + retry: (origin: string) => void; }; // Single-site mode has no provenance accents; a module constant keeps the @@ -172,13 +207,51 @@ export function SingleSiteDataProvider({ children }: { children: ReactNode }) { // A federated origin the hub reads from. `origin` is "" for the hub's own // same-origin pool, else a full origin ("https://x.com"). `siteTitle` labels -// its channel group; `accent` is carried for provenance in the UI. +// its channel group; `accent` is carried for provenance in the UI. `enabled` +// false takes it out of the search: none of its feeds are fetched and nothing +// already cached from it is merged (default true). export type FederatedSite = { origin: string; siteTitle: string; accent?: string; + enabled?: boolean; }; +// At most this many summaries-page fetches in flight per archive, so one big +// member (Jeralyzer: dozens of pages) cannot take every connection the browser +// has and starve the small ones. Per ORIGIN, not global: different origins do +// not share a connection pool, so a global cap would only slow the whole hub. +const PAGE_FETCHES_PER_SITE = 6; + +type Gate = { active: number; queue: Array<() => void> }; +const pageGates = new Map<string, Gate>(); + +async function gated<T>(origin: string, fn: () => Promise<T>): Promise<T> { + let g = pageGates.get(origin); + if (!g) { + g = { active: 0, queue: [] }; + pageGates.set(origin, g); + } + if (g.active >= PAGE_FETCHES_PER_SITE) { + // The releasing fetch hands its slot straight to us (active unchanged). + await new Promise<void>((resolve) => g.queue.push(resolve)); + } else { + g.active++; + } + try { + return await fn(); + } finally { + const next = g.queue.shift(); + if (next) next(); + else g.active--; + } +} + +function errorText(e: unknown): string | undefined { + if (!e) return undefined; + return e instanceof Error ? e.message : String(e); +} + // Multi-site data source for the hub: fans out the same summaries/subs feeds // across many origins and merges them into ONE origin-qualified view, so // TranscriptSearch stays mode-agnostic (it reads exactly the same context shape @@ -189,56 +262,136 @@ export type FederatedSite = { // model, and downstream fetches (fetchTranscript decodes the origin back out) // never collide across sites. Grouping falls out of the existing grouped- // checkbox UI: one ChannelGroup per site (id = origin, label = siteTitle). +// +// Per archive, in order: its summaries manifest, then — only once that has +// arrived, and only while the archive is in scope — its pages, at most +// PAGE_FETCHES_PER_SITE at a time. An archive's records join the merged list +// only when ALL its pages are in (status "ready"), so the list grows one whole +// archive at a time and a search re-runs once per archive, not once per page. +// A failed archive contributes nothing and says so (`federation`). +// +// `progressive`: summariesReady turns true as soon as ONE in-scope archive is +// ready (the hub's search page, where results should not wait for the slowest +// member). Without it, summariesReady waits until every in-scope archive has +// settled, ready or failed (the /ask chat, which grounds in what it was given). export function MultiSiteDataProvider({ sites, + progressive = false, children, }: { sites: FederatedSite[]; + progressive?: boolean; children: ReactNode; }) { + const queryClient = useQueryClient(); + const inScope = (s: FederatedSite) => s.enabled !== false; + // 1. Per-origin summaries manifests (channels/groups/pageCount/freshness). const manifestQueries = useQueries({ queries: sites.map((s) => ({ queryKey: ["manifest", s.origin], queryFn: () => readerFor(s.origin).readSummariesManifest(), + enabled: inScope(s), })), }); - const manifestsSettled = manifestQueries.every( - (q) => q.isSuccess || q.isError, - ); - // 2. Flat page descriptors across every origin whose manifest loaded, so a - // single useQueries can fan out all pages regardless of per-site counts. + // 2. Page descriptors for every in-scope origin whose manifest has arrived. + // Keyed by a string of (origin, pageCount) so the list's identity moves + // only when an archive's manifest lands or the scope changes. + const pagePlanKey = sites + .map((s, i) => { + const m = manifestQueries[i]?.data; + return inScope(s) && m ? `${s.origin}\n${m.pageCount}` : ""; + }) + .join("\u0000"); const pageDescriptors = useMemo<{ origin: string; index: number }[]>(() => { const out: { origin: string; index: number }[] = []; sites.forEach((s, i) => { + if (!inScope(s)) return; const pc = manifestQueries[i]?.data?.pageCount ?? 0; for (let p = 0; p < pc; p++) out.push({ origin: s.origin, index: p }); }); return out; - // manifestQueries identity churns; key off settled + the site set. + // manifestQueries identity churns; pagePlanKey is its content. // eslint-disable-next-line react-hooks/exhaustive-deps - }, [sites, manifestsSettled]); + }, [pagePlanKey]); const pageQueries = useQueries({ queries: pageDescriptors.map((d) => ({ queryKey: ["summaries-page", d.origin, d.index], - queryFn: () => readerFor(d.origin).readSummariesPage(d.index), + queryFn: () => + gated(d.origin, () => readerFor(d.origin).readSummariesPage(d.index)), })), }); + // 3. Each archive's state, from its manifest query and its page queries. + const pagesByOrigin = new Map<string, (typeof pageQueries)[number][]>(); + pageQueries.forEach((q, i) => { + const origin = pageDescriptors[i]?.origin; + if (origin === undefined) return; + const list = pagesByOrigin.get(origin); + if (list) list.push(q); + else pagesByOrigin.set(origin, [q]); + }); + const rawStates: FederatedSiteState[] = sites.map((s, i) => { + const base = { + origin: s.origin, + siteTitle: s.siteTitle, + ...(s.accent ? { accent: s.accent } : {}), + }; + if (!inScope(s)) return { ...base, status: "off" }; + const m = manifestQueries[i]; + const pages = pagesByOrigin.get(s.origin) ?? []; + const count = m?.data?.totalCount; + const withCount = count === undefined ? base : { ...base, count }; + if ( + m?.isSuccess && + pages.length === m.data.pageCount && + pages.every((p) => p.isSuccess) + ) { + return { ...withCount, status: "ready" }; + } + const fetching = !!m?.isFetching || pages.some((p) => p.isFetching); + const failed = m?.isError ? m : pages.find((p) => p.isError); + if (failed && !fetching) { + return { + ...withCount, + status: "failed", + ...(errorText(failed.error) ? { error: errorText(failed.error) } : {}), + }; + } + return { ...withCount, status: "loading" }; + }); + const statusKey = rawStates + .map((s) => `${s.origin}|${s.siteTitle}|${s.accent ?? ""}|${s.status}|${s.count ?? ""}|${s.error ?? ""}`) + .join("\n"); + const siteStates = useMemo( + () => rawStates, + // eslint-disable-next-line react-hooks/exhaustive-deps + [statusKey], + ); + + const readyOrigins = useMemo( + () => + new Set(siteStates.filter((s) => s.status === "ready").map((s) => s.origin)), + [siteStates], + ); + const readyKey = Array.from(readyOrigins).join("\u0000"); + const scoped = siteStates.filter((s) => s.status !== "off"); + const allSettled = scoped.every( + (s) => s.status === "ready" || s.status === "failed", + ); + const summariesReady = allSettled || (progressive && readyOrigins.size > 0); + const loadedPages = pageQueries.filter((q) => q.data).length; - const pagesSettled = pageQueries.every((q) => q.isSuccess || q.isError); - // Ready once every manifest and every page has settled (a failing origin - // resolves to error rather than blocking the rest of the shelf). - const summariesReady = manifestsSettled && pagesSettled; - // 3. Merge summaries, rewriting ids to be origin-qualified. + // 4. Merge the READY archives' summaries, rewriting ids to be + // origin-qualified, newest first. const summaries = useMemo<DisplaySummary[]>(() => { const out: DisplaySummary[] = []; pageQueries.forEach((q, i) => { const origin = pageDescriptors[i]?.origin ?? ""; - if (!q.data) return; + if (!q.data || !readyOrigins.has(origin)) return; for (const t of q.data) { out.push( origin @@ -251,35 +404,37 @@ export function MultiSiteDataProvider({ ); } }); - if (summariesReady) { - out.sort( - (a, b) => - b.uploadDate.localeCompare(a.uploadDate) || - a.channelSlug.localeCompare(b.channelSlug) || - a.id.localeCompare(b.id), - ); - } + out.sort( + (a, b) => + b.uploadDate.localeCompare(a.uploadDate) || + a.channelSlug.localeCompare(b.channelSlug) || + a.id.localeCompare(b.id), + ); return out; // eslint-disable-next-line react-hooks/exhaustive-deps - }, [loadedPages, summariesReady, pageDescriptors]); + }, [readyKey, pageDescriptors]); - // 4. One channel group per site; channels keyed by makeId(origin, slug). + // 5. One channel group per in-scope site; channels keyed by + // makeId(origin, slug), from every in-scope manifest that has arrived. const groups = useMemo<ChannelGroup[]>( () => - sites.map((s, i) => ({ + sites.filter(inScope).map((s, i) => ({ id: s.origin, name: s.siteTitle, selectedByDefault: true, order: i, ...(s.accent ? { accent: s.accent } : {}), })), + // eslint-disable-next-line react-hooks/exhaustive-deps [sites], ); - const defaultGroupId = sites[0]?.origin ?? DEFAULT_GROUP_FALLBACK_ID; + const defaultGroupId = + sites.find(inScope)?.origin ?? sites[0]?.origin ?? DEFAULT_GROUP_FALLBACK_ID; const channels = useMemo<ChannelOption[]>(() => { const out: ChannelOption[] = []; sites.forEach((s, i) => { + if (!inScope(s)) return; const manifest = manifestQueries[i]?.data; if (!manifest) return; for (const c of manifest.channels) { @@ -295,20 +450,27 @@ export function MultiSiteDataProvider({ (a, b) => a.groupId.localeCompare(b.groupId) || a.name.localeCompare(b.name), ); // eslint-disable-next-line react-hooks/exhaustive-deps - }, [sites, manifestsSettled]); + }, [sites, pagePlanKey]); - // 5. Merge subs manifests: concat channels (slug → origin-qualified) so the - // chat scope resolves per-origin; sum the live-chat/total counts. + // 6. Merge subs manifests: concat channels (slug → origin-qualified) so the + // chat scope resolves per-origin; sum the live-chat/total counts. Merged + // from whichever in-scope archives have answered so far. const subsQueries = useQueries({ queries: sites.map((s) => ({ queryKey: ["subs-manifest", s.origin], queryFn: () => readerFor(s.origin).readSubsSiteManifest(), + enabled: inScope(s), })), }); - const subsSettled = subsQueries.every((q) => q.isSuccess || q.isError); + const subsKey = sites + .map((s, i) => (inScope(s) && subsQueries[i]?.data ? s.origin : "")) + .join("\u0000"); const subsManifest = useMemo<SubsManifest | null>(() => { const loaded = sites - .map((s, i) => ({ origin: s.origin, data: subsQueries[i]?.data })) + .map((s, i) => ({ + origin: s.origin, + data: inScope(s) ? subsQueries[i]?.data : undefined, + })) .filter((e): e is { origin: string; data: SubsManifest } => !!e.data); if (loaded.length === 0) return null; const channelsOut: SubsManifest["channels"] = []; @@ -331,11 +493,12 @@ export function MultiSiteDataProvider({ generatedAt, }; // eslint-disable-next-line react-hooks/exhaustive-deps - }, [sites, subsSettled]); + }, [subsKey]); - // 5b. Merge posts manifests, same origin-qualification as subs. A member site + // 6b. Merge posts manifests, same origin-qualification as subs. A member site // with no posts corpus 404s; treat that as an empty contribution so one - // video-only origin can't blank the hub's posts scope. + // video-only origin can't blank the hub's posts scope (and it is not a + // failure of that archive). const postsQueries = useQueries({ queries: sites.map((s) => ({ queryKey: ["posts-manifest", s.origin], @@ -350,12 +513,18 @@ export function MultiSiteDataProvider({ generatedAt: "", }), ), + enabled: inScope(s), })), }); - const postsSettled = postsQueries.every((q) => q.isSuccess || q.isError); + const postsKey = sites + .map((s, i) => (inScope(s) && postsQueries[i]?.data ? s.origin : "")) + .join("\u0000"); const postsManifest = useMemo<PostsManifest | null>(() => { const loaded = sites - .map((s, i) => ({ origin: s.origin, data: postsQueries[i]?.data })) + .map((s, i) => ({ + origin: s.origin, + data: inScope(s) ? postsQueries[i]?.data : undefined, + })) .filter((e): e is { origin: string; data: PostsManifest } => !!e.data); if (loaded.length === 0) return null; const channelsOut: PostsManifest["channels"] = []; @@ -375,7 +544,7 @@ export function MultiSiteDataProvider({ generatedAt, }; // eslint-disable-next-line react-hooks/exhaustive-deps - }, [sites, postsSettled]); + }, [postsKey]); // Alias dictionaries per federated origin, merged into one list (later origins // shadow earlier ones by id). Missing files resolve to [] — additive only. @@ -384,28 +553,36 @@ export function MultiSiteDataProvider({ queryKey: ["search-aliases", s.origin], queryFn: () => fetchAliases(s.origin), staleTime: Infinity, + enabled: inScope(s), })), }); - const aliasesSettled = aliasQueries.every((q) => q.isSuccess || q.isError); + const aliasesKey = sites + .map((s, i) => (inScope(s) && aliasQueries[i]?.data ? s.origin : "")) + .join("\u0000"); const aliases = useMemo<SearchAlias[]>(() => { let merged: SearchAlias[] = []; - for (const q of aliasQueries) if (q.data) merged = mergeAliases(merged, q.data); + sites.forEach((s, i) => { + const data = inScope(s) ? aliasQueries[i]?.data : undefined; + if (data) merged = mergeAliases(merged, data); + }); return merged; // eslint-disable-next-line react-hooks/exhaustive-deps - }, [aliasesSettled]); + }, [aliasesKey]); // The hub's OWN /tags.json, not a merge of its members'. See the field note // on SearchDataValue: a federated count nobody can reproduce is worse than no // chip, so until a hub build writes one, hub mode offers no tag chips. const curatedTags = useCuratedTags(""); - // Synthetic merged summaries manifest. TranscriptSearch reads channels/groups - // from the context (above), not from here, but the field is part of the - // SummariesState contract, so provide a coherent merged view. + // Synthetic merged summaries manifest over the READY archives — what the + // search actually covers. TranscriptSearch reads channels/groups from the + // context (above), not from here, but the field is part of the SummariesState + // contract, so provide a coherent merged view. const mergedManifest = useMemo<Manifest | null>(() => { - if (!manifestsSettled) return null; - const loaded = manifestQueries - .map((q) => q.data) + const loaded = sites + .map((s, i) => + readyOrigins.has(s.origin) ? manifestQueries[i]?.data : undefined, + ) .filter((m): m is Manifest => !!m); if (loaded.length === 0) return null; return { @@ -422,7 +599,24 @@ export function MultiSiteDataProvider({ defaultGroupId, }; // eslint-disable-next-line react-hooks/exhaustive-deps - }, [manifestsSettled, groups, defaultGroupId, pageDescriptors]); + }, [readyKey, groups, defaultGroupId, pageDescriptors]); + + // Every in-scope archive failed: say so to the consumers that read `error` + // (the /ask chat). One failure among several is the federation line's job. + const allFailed = + scoped.length > 0 && scoped.every((s) => s.status === "failed"); + const summariesError = useMemo<Error | null>( + () => + allFailed + ? new Error( + scoped.length === 1 + ? `${scoped[0].siteTitle} did not answer.` + : "None of the archives answered.", + ) + : null, + // eslint-disable-next-line react-hooks/exhaustive-deps + [allFailed, statusKey], + ); const summariesState = useMemo<SummariesState>( () => ({ @@ -431,12 +625,19 @@ export function MultiSiteDataProvider({ loadedPages, pageCount: pageDescriptors.length, summariesReady, - error: null, + error: summariesError, }), - [mergedManifest, summaries, loadedPages, pageDescriptors, summariesReady], + [ + mergedManifest, + summaries, + loadedPages, + pageDescriptors, + summariesReady, + summariesError, + ], ); - // origin → accent lookup for provenance in results. + // origin → accent / title lookups for provenance in results. const accentByOrigin = useMemo(() => { const m = new Map<string, string>(); for (const s of sites) if (s.accent) m.set(s.origin, s.accent); @@ -446,6 +647,30 @@ export function MultiSiteDataProvider({ (origin: string) => accentByOrigin.get(origin), [accentByOrigin], ); + const titleByOrigin = useMemo( + () => new Map(sites.map((s) => [s.origin, s.siteTitle])), + [sites], + ); + const siteTitleOf = useCallback( + (origin: string) => titleByOrigin.get(origin), + [titleByOrigin], + ); + + // Retry: refetch every query of this origin that ended in error — its + // manifest, failed pages, and any other feed of it that failed. + const retry = useCallback( + (origin: string) => { + void queryClient.refetchQueries({ + predicate: (q) => + q.queryKey[1] === origin && q.state.status === "error", + }); + }, + [queryClient], + ); + const federation = useMemo<FederationState>( + () => ({ sites: siteStates, retry }), + [siteStates, retry], + ); const value = useMemo<SearchDataValue>( () => ({ @@ -461,6 +686,8 @@ export function MultiSiteDataProvider({ accentOf, aliases, curatedTags, + federation, + siteTitleOf, }), [ summariesState, @@ -472,6 +699,8 @@ export function MultiSiteDataProvider({ accentOf, aliases, curatedTags, + federation, + siteTitleOf, ], );