Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 396968443b31dd22dba28246c41db1c8b00ea05f
parent 01934bcf79a1a826d58ea95b727576aa60644719
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu,  2 Jul 2026 02:02:58 -0400

Phase 6b: MultiSiteDataProvider — merge many origins into one search view

Add the hub's data source to SearchDataContext. MultiSiteDataProvider takes a
FederatedSite[] (origin/siteTitle/accent), fans out each origin's summaries
manifest + pages + subs manifest via useQueries, and merges them into one
origin-qualified SummariesState:

- every cross-origin summary's slug + channelSlug is rewritten to
  makeId(origin, …), so result ids, the channelKey selection model, and
  fetchTranscript (which decodes the origin back out) never collide across sites;
- one ChannelGroup per site (id = origin, label = siteTitle) so group-by-site
  falls out of the existing grouped-checkbox UI; channelKeyOf = t.channelSlug
  (already the qualified key);
- subs manifests concat with origin-qualified slugs + summed counts; a failing
  origin resolves to error rather than blocking the rest of the shelf.

Make TranscriptSearch's subs handling origin-aware: build {origin, channelSlug}
refs via splitId and origin-qualify the chat-scope ids via makeId. Single-site
is byte-identical (bare slug → {origin:"", slug}; makeId("",x)===x) — verified by
the live-chat + profile-row + share e2e (13 passed).

MultiSiteDataProvider isn't mounted yet (the hub page lands next), so the site
path still uses SingleSiteDataProvider: this commit is site-mode-inert.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/components/SearchDataContext.tsx | 237++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mcommon/components/TranscriptSearch.tsx | 30+++++++++++++++++++++---------
2 files changed, 257 insertions(+), 10 deletions(-)

diff --git a/common/components/SearchDataContext.tsx b/common/components/SearchDataContext.tsx @@ -8,9 +8,13 @@ // itself stays mode-agnostic and reads everything from this context. import { createContext, useContext, useMemo, type ReactNode } from "react"; +import { useQueries } from "@tanstack/react-query"; import { useSummaries, type SummariesState } from "./summariesCache"; import { useSubsManifest } from "./subsCache"; -import type { SubsManifest } from "../lib/manifest"; +import { idBaseUrl, makeId } from "./originId"; +import type { DisplaySummary } from "../lib/transcripts"; +import type { Manifest, SubsManifest } from "../lib/manifest"; +import { pageFileName } from "../lib/manifest"; import { DEFAULT_GROUP_FALLBACK_ID, FALLBACK_GROUP, @@ -115,3 +119,234 @@ export function SingleSiteDataProvider({ children }: { children: ReactNode }) { </SearchDataContext.Provider> ); } + +async function fetchJson<T>(url: string): Promise<T> { + const r = await fetch(url); + if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); + return (await r.json()) as T; +} + +// A federated origin the hub reads from. `origin` is "" for the hub's own +// same-origin pool, else a full origin ("https://x.com"). `siteTitle` labels +// its channel group; `accent` is carried for provenance in the UI. +export type FederatedSite = { + origin: string; + siteTitle: string; + accent?: string; +}; + +// Multi-site data source for the hub: fans out the same summaries/subs feeds +// across many origins and merges them into ONE origin-qualified view, so +// TranscriptSearch stays mode-agnostic (it reads exactly the same context shape +// as in single-site mode). +// +// Identity: every cross-origin summary's `slug` and `channelSlug` are rewritten +// to makeId(origin, …) at merge time, so result ids, the channelKey selection +// model, and downstream fetches (fetchTranscript decodes the origin back out) +// never collide across sites. Grouping falls out of the existing grouped- +// checkbox UI: one ChannelGroup per site (id = origin, label = siteTitle). +export function MultiSiteDataProvider({ + sites, + children, +}: { + sites: FederatedSite[]; + children: ReactNode; +}) { + // 1. Per-origin summaries manifests (channels/groups/pageCount/freshness). + const manifestQueries = useQueries({ + queries: sites.map((s) => ({ + queryKey: ["manifest", s.origin], + queryFn: () => + fetchJson<Manifest>(`${idBaseUrl(s.origin)}/summaries/manifest.json`), + })), + }); + const manifestsSettled = manifestQueries.every( + (q) => q.isSuccess || q.isError, + ); + + // 2. Flat page descriptors across every origin whose manifest loaded, so a + // single useQueries can fan out all pages regardless of per-site counts. + const pageDescriptors = useMemo<{ origin: string; index: number }[]>(() => { + const out: { origin: string; index: number }[] = []; + sites.forEach((s, i) => { + const pc = manifestQueries[i]?.data?.pageCount ?? 0; + for (let p = 0; p < pc; p++) out.push({ origin: s.origin, index: p }); + }); + return out; + // manifestQueries identity churns; key off settled + the site set. + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [sites, manifestsSettled]); + + const pageQueries = useQueries({ + queries: pageDescriptors.map((d) => ({ + queryKey: ["summaries-page", d.origin, d.index], + queryFn: () => + fetchJson<DisplaySummary[]>( + `${idBaseUrl(d.origin)}/summaries/${pageFileName(d.index)}`, + ), + })), + }); + + const loadedPages = pageQueries.filter((q) => q.data).length; + const pagesSettled = pageQueries.every((q) => q.isSuccess || q.isError); + // Ready once every manifest and every page has settled (a failing origin + // resolves to error rather than blocking the rest of the shelf). + const summariesReady = manifestsSettled && pagesSettled; + + // 3. Merge summaries, rewriting ids to be origin-qualified. + const summaries = useMemo<DisplaySummary[]>(() => { + const out: DisplaySummary[] = []; + pageQueries.forEach((q, i) => { + const origin = pageDescriptors[i]?.origin ?? ""; + if (!q.data) return; + for (const t of q.data) { + out.push( + origin + ? { + ...t, + slug: makeId(origin, t.slug), + channelSlug: makeId(origin, t.channelSlug), + } + : t, + ); + } + }); + if (summariesReady) { + out.sort( + (a, b) => + b.uploadDate.localeCompare(a.uploadDate) || + a.channelSlug.localeCompare(b.channelSlug) || + a.id.localeCompare(b.id), + ); + } + return out; + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [loadedPages, summariesReady, pageDescriptors]); + + // 4. One channel group per site; channels keyed by makeId(origin, slug). + const groups = useMemo<ChannelGroup[]>( + () => + sites.map((s, i) => ({ + id: s.origin, + name: s.siteTitle, + selectedByDefault: true, + order: i, + })), + [sites], + ); + const defaultGroupId = sites[0]?.origin ?? DEFAULT_GROUP_FALLBACK_ID; + + const channels = useMemo<ChannelOption[]>(() => { + const out: ChannelOption[] = []; + sites.forEach((s, i) => { + const manifest = manifestQueries[i]?.data; + if (!manifest) return; + for (const c of manifest.channels) { + if (!c.slug || !c.name) continue; + out.push({ + key: makeId(s.origin, c.slug), + name: c.name, + groupId: s.origin, + }); + } + }); + return out.sort( + (a, b) => a.groupId.localeCompare(b.groupId) || a.name.localeCompare(b.name), + ); + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [sites, manifestsSettled]); + + // 5. Merge subs manifests: concat channels (slug → origin-qualified) so the + // chat scope resolves per-origin; sum the live-chat/total counts. + const subsQueries = useQueries({ + queries: sites.map((s) => ({ + queryKey: ["subs-manifest", s.origin], + queryFn: () => + fetchJson<SubsManifest>(`${idBaseUrl(s.origin)}/subs/manifest.json`), + })), + }); + const subsSettled = subsQueries.every((q) => q.isSuccess || q.isError); + const subsManifest = useMemo<SubsManifest | null>(() => { + const loaded = sites + .map((s, i) => ({ origin: s.origin, data: subsQueries[i]?.data })) + .filter((e): e is { origin: string; data: SubsManifest } => !!e.data); + if (loaded.length === 0) return null; + const channelsOut: SubsManifest["channels"] = []; + let totalCount = 0; + let liveChatTotalCount = 0; + let generatedAt = ""; + for (const { origin, data } of loaded) { + for (const c of data.channels) { + channelsOut.push({ ...c, slug: makeId(origin, c.slug) }); + } + totalCount += data.totalCount ?? 0; + liveChatTotalCount += data.liveChatTotalCount ?? 0; + if (data.generatedAt > generatedAt) generatedAt = data.generatedAt; + } + return { + version: loaded[0].data.version, + channels: channelsOut, + totalCount, + liveChatTotalCount, + generatedAt, + }; + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [sites, subsSettled]); + + // Synthetic merged summaries manifest. TranscriptSearch reads channels/groups + // from the context (above), not from here, but the field is part of the + // SummariesState contract, so provide a coherent merged view. + const mergedManifest = useMemo<Manifest | null>(() => { + if (!manifestsSettled) return null; + const loaded = manifestQueries + .map((q) => q.data) + .filter((m): m is Manifest => !!m); + if (loaded.length === 0) return null; + return { + version: loaded[0].version, + totalCount: loaded.reduce((n, m) => n + (m.totalCount ?? 0), 0), + pageSize: loaded[0].pageSize, + pageCount: pageDescriptors.length, + generatedAt: loaded.reduce( + (max, m) => (m.generatedAt > max ? m.generatedAt : max), + "", + ), + channels: [], + groups, + defaultGroupId, + }; + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [manifestsSettled, groups, defaultGroupId, pageDescriptors]); + + const summariesState = useMemo<SummariesState>( + () => ({ + manifest: mergedManifest, + summaries, + loadedPages, + pageCount: pageDescriptors.length, + summariesReady, + error: null, + }), + [mergedManifest, summaries, loadedPages, pageDescriptors, summariesReady], + ); + + const value = useMemo<SearchDataValue>( + () => ({ + summariesState, + subsManifest, + channels, + groups, + defaultGroupId, + // channelSlug was rewritten to the origin-qualified key above, so it IS + // the selection key (matches ChannelOption.key). + channelKeyOf: (t) => t.channelSlug, + }), + [summariesState, subsManifest, channels, groups, defaultGroupId], + ); + + return ( + <SearchDataContext.Provider value={value}> + {children} + </SearchDataContext.Provider> + ); +} diff --git a/common/components/TranscriptSearch.tsx b/common/components/TranscriptSearch.tsx @@ -62,6 +62,7 @@ import { SearchChartPanel } from "./charts/SearchChartPanel"; import { LayerSwatch } from "./LayerSwatch"; import { ymdToInput, inputToYmd } from "../lib/ymd"; import type { DisplaySummary } from "../lib/transcripts"; +import { makeId, splitId } from "./originId"; import { sortGroups, type ChannelGroup } from "../lib/channelGroups"; type Summary = DisplaySummary; @@ -280,32 +281,43 @@ export default function TranscriptSearch() { [committedRoot], ); const needsChatManifests = draftHasChatLeaf || committedHasChatLeaf; - const subsChannelSlugs = useMemo( - () => (subsManifest ? subsManifest.channels.map((c) => c.slug) : []), + // Origin-qualified refs. In single-site mode the manifest slugs are bare, so + // splitId yields {origin:"", channelSlug} and the refs (and the chat-scope + // ids built below) are byte-identical to before. In hub mode the merged + // manifest carries makeId(origin, slug) slugs, so each ref fetches from the + // right origin and the chat-scope ids match the origin-qualified summaries. + const subsRefs = useMemo( + () => + subsManifest + ? subsManifest.channels.map((c) => { + const { origin, slug } = splitId(c.slug); + return { origin, channelSlug: slug }; + }) + : [], [subsManifest], ); const channelSubsQueries = useChannelSubsManifests( - needsChatManifests ? subsChannelSlugs : [], + needsChatManifests ? subsRefs : [], ); - // Set of (channelSlug/videoId) slugs that have live_chat content, - // resolved across all per-channel subs manifests. + // Set of origin-qualified (channelSlug/videoId) slugs that have live_chat + // content, resolved across all per-channel subs manifests. const chatScopeSlugs = useMemo<Set<string>>(() => { const set = new Set<string>(); if (!needsChatManifests) return set; for (let i = 0; i < channelSubsQueries.length; i++) { const q = channelSubsQueries[i]; - const channelSlug = subsChannelSlugs[i]; + const ref = subsRefs[i]; const data = q.data; - if (!data) continue; + if (!data || !ref) continue; for (const id of Object.keys(data.slugToPage)) { - set.add(`${channelSlug}/${id}`); + set.add(makeId(ref.origin, `${ref.channelSlug}/${id}`)); } } return set; // eslint-disable-next-line react-hooks/exhaustive-deps }, [ needsChatManifests, - subsChannelSlugs, + subsRefs, channelSubsQueries.map((q) => (q.data ? 1 : 0)).join(""), ]); const subsManifestReady =