Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 225337d1f26e6a3e1c75f2429902f82ff4c7cbdc
parent 53a1a83783c0f4e91f33972a847389b42fdff482
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Wed,  1 Jul 2026 23:35:22 -0400

Phase 2: origin-aware client caches (federation-ready)

Introduce originId.ts (makeId/splitId/idBaseUrl): a single opaque string
identity that is a bare "channelSlug/videoId" for same-origin content and
"origin\tchannelSlug/videoId" for federated cross-origin content. makeId("",
slug) === slug, so every single-site code path is byte-for-byte unchanged.

- transcriptCache/subsCache: fetchTranscript/fetchSubs take an OriginId (origin
  embedded — search pipeline call sites unchanged); internal manifest/page
  fetchers take an origin, prefix URLs with idBaseUrl(origin), and key their
  maps by origin so cross-origin channels can't collide. Warmed entries re-keyed
  by OriginId.
- summariesCache/statsCache/subsCache hooks gain an optional origin param
  (default "") and add origin to every react-query key.
- transcriptStore: out-of-line keys (store record under its OriginId, not a
  "slug" keyPath) so cross-origin content can't shadow same-origin; DB_VERSION
  2 -> 3 (self-healing wipe, established pattern).
- editor IDB platform-cache e2e updated to seed the v3 out-of-line store.

Verified: full export e2e green except the 3 known-flaky charts-metadata
recharts-timeout tests (sibling metadata chart from /stats passes fast, so
statsCache is fine); both IDB self-heal specs pass on v3.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/components/originId.ts | 38++++++++++++++++++++++++++++++++++++++
Mcommon/components/statsCache.ts | 19++++++++++++-------
Mcommon/components/subsCache.ts | 75+++++++++++++++++++++++++++++++++++++++++++++++----------------------------
Mcommon/components/summariesCache.ts | 18+++++++++++-------
Mcommon/components/transcriptCache.ts | 53++++++++++++++++++++++++++++++++---------------------
Mcommon/components/transcriptStore.ts | 33++++++++++++++++++++-------------
Meditor/e2e/export-player-platform-cache.spec.ts | 10++++++----
7 files changed, 166 insertions(+), 80 deletions(-)

diff --git a/common/components/originId.ts b/common/components/originId.ts @@ -0,0 +1,38 @@ +// Origin-qualified content identity for the federated hub. +// +// The whole search operates on a single string identity (`slug`, shaped +// "channelSlug/videoId"). To federate content from multiple origins WITHOUT +// threading (origin, slug) tuples through every call site, we keep one opaque +// string id and qualify it with an origin prefix ONLY for cross-origin content. +// +// same-origin (single-site export, the built-in pool): "channelSlug/videoId" +// cross-origin (a user-added external site): "https://x.com\tchannelSlug/videoId" +// +// The separator is a TAB — impossible in a URL origin and impossible in a slug +// (channel dir names + video ids), so splitting is unambiguous. Crucially +// makeId("", slug) === slug, so every single-site code path produces byte-for- +// byte identical strings and its behavior (URL state, filter profiles, IDB +// keys) is unchanged. + +const SEP = "\t"; + +export type OriginId = string; + +// Combine an origin ("" = same-origin) with a bare slug into an id. A bare slug +// is returned unchanged for same-origin content. +export function makeId(origin: string, slug: string): OriginId { + return origin ? `${origin}${SEP}${slug}` : slug; +} + +// Decode an id back into its origin ("" = same-origin) and bare slug. +export function splitId(id: OriginId): { origin: string; slug: string } { + const sep = id.indexOf(SEP); + if (sep < 0) return { origin: "", slug: id }; + return { origin: id.slice(0, sep), slug: id.slice(sep + 1) }; +} + +// The base URL to prefix onto a root-relative resource path for a given origin: +// "" for same-origin (paths stay root-relative, unchanged), else the origin. +export function idBaseUrl(origin: string): string { + return origin || ""; +} diff --git a/common/components/statsCache.ts b/common/components/statsCache.ts @@ -4,6 +4,7 @@ import { useMemo } from "react"; import { useQueries, useQuery } from "@tanstack/react-query"; import type { StatsManifest, VideoStat } from "../lib/stats"; import { statsPageFileName } from "../lib/stats"; +import { idBaseUrl } from "./originId"; async function fetchJson<T>(url: string): Promise<T> { const r = await fetch(url); @@ -20,24 +21,28 @@ export type StatsState = { error: Error | null; }; -export function useStatsManifest() { +export function useStatsManifest(origin = "") { return useQuery<StatsManifest>({ - queryKey: ["stats-manifest"], - queryFn: () => fetchJson<StatsManifest>("/stats/manifest.json"), + queryKey: ["stats-manifest", origin], + queryFn: () => + fetchJson<StatsManifest>(`${idBaseUrl(origin)}/stats/manifest.json`), }); } // Mirrors summariesCache.useSummaries: load the manifest, then fan out the // byte-capped page files via react-query and concatenate them. -export function useStats(): StatsState { - const manifestQuery = useStatsManifest(); +export function useStats(origin = ""): StatsState { + const manifestQuery = useStatsManifest(origin); const manifest = manifestQuery.data ?? null; const pageCount = manifest?.pageCount ?? 0; const pageQueries = useQueries({ queries: Array.from({ length: pageCount }, (_, i) => ({ - queryKey: ["stats-page", i], - queryFn: () => fetchJson<VideoStat[]>(`/stats/${statsPageFileName(i)}`), + queryKey: ["stats-page", origin, i], + queryFn: () => + fetchJson<VideoStat[]>( + `${idBaseUrl(origin)}/stats/${statsPageFileName(i)}`, + ), enabled: pageCount > 0, })), }); diff --git a/common/components/subsCache.ts b/common/components/subsCache.ts @@ -4,6 +4,7 @@ import { useQueries, useQuery } from "@tanstack/react-query"; import type { SubsDetail } from "../lib/subs"; import type { ChannelSubsManifest, SubsManifest } from "../lib/manifest"; import { subsPageFileName } from "../lib/manifest"; +import { makeId, splitId, idBaseUrl } from "./originId"; async function fetchJson<T>(url: string): Promise<T> { const r = await fetch(url); @@ -16,29 +17,39 @@ const inFlight = new Map<string, Promise<SubsDetail>>(); const channelManifests = new Map<string, Promise<ChannelSubsManifest>>(); const pagePromises = new Map<string, Promise<SubsDetail[]>>(); -export function fetchSubs(slug: string): Promise<SubsDetail> { - const hit = resolved.get(slug); +// A per-channel subs reference: channel slug + the origin it lives on +// ("" = same-origin). Used by the multi-origin manifest hook. +export type SubsManifestRef = { channelSlug: string; origin?: string }; + +// `id` is an OriginId (see originId.ts): bare slug same-origin, origin-prefixed +// cross-origin. Maps are keyed by the full id so origins don't collide. +export function fetchSubs(id: string): Promise<SubsDetail> { + const hit = resolved.get(id); if (hit) return Promise.resolve(hit); - const flying = inFlight.get(slug); + const flying = inFlight.get(id); if (flying) return flying; - const p = load(slug).then((detail) => { - resolved.set(slug, detail); - inFlight.delete(slug); + const p = load(id).then((detail) => { + resolved.set(id, detail); + inFlight.delete(id); return detail; }); - p.catch(() => inFlight.delete(slug)); - inFlight.set(slug, p); + p.catch(() => inFlight.delete(id)); + inFlight.set(id, p); return p; } function fetchChannelSubsManifest( channelSlug: string, + origin = "", ): Promise<ChannelSubsManifest> { - let p = channelManifests.get(channelSlug); + const key = makeId(origin, channelSlug); + let p = channelManifests.get(key); if (!p) { - p = fetchJson<ChannelSubsManifest>(`/subs/${channelSlug}/manifest.json`); - p.catch(() => channelManifests.delete(channelSlug)); - channelManifests.set(channelSlug, p); + p = fetchJson<ChannelSubsManifest>( + `${idBaseUrl(origin)}/subs/${channelSlug}/manifest.json`, + ); + p.catch(() => channelManifests.delete(key)); + channelManifests.set(key, p); } return p; } @@ -46,12 +57,13 @@ function fetchChannelSubsManifest( function fetchPage( channelSlug: string, pageIndex: number, + origin = "", ): Promise<SubsDetail[]> { - const key = `${channelSlug}:${pageIndex}`; + const key = `${makeId(origin, channelSlug)}:${pageIndex}`; let p = pagePromises.get(key); if (!p) { p = fetchJson<SubsDetail[]>( - `/subs/${channelSlug}/${subsPageFileName(pageIndex)}`, + `${idBaseUrl(origin)}/subs/${channelSlug}/${subsPageFileName(pageIndex)}`, ); p.catch(() => pagePromises.delete(key)); pagePromises.set(key, p); @@ -59,41 +71,48 @@ function fetchPage( return p; } -async function load(slug: string): Promise<SubsDetail> { +async function load(id: string): Promise<SubsDetail> { + const { origin, slug } = splitId(id); const slashIdx = slug.indexOf("/"); if (slashIdx < 0) throw new Error(`Malformed subs slug: ${slug}`); const channelSlug = slug.slice(0, slashIdx); const videoId = slug.slice(slashIdx + 1); - const manifest = await fetchChannelSubsManifest(channelSlug); + const manifest = await fetchChannelSubsManifest(channelSlug, origin); const pageIndex = manifest.slugToPage[videoId]; if (pageIndex === undefined) throw new Error(`Unknown subs slug: ${slug}`); - const page = await fetchPage(channelSlug, pageIndex); + const page = await fetchPage(channelSlug, pageIndex, origin); let found: SubsDetail | undefined; for (const entry of page) { + const entryId = makeId(origin, entry.slug); if (entry.slug === slug) found = entry; - resolved.set(entry.slug, entry); + resolved.set(entryId, entry); } if (!found) throw new Error(`Subs ${slug} missing from page ${pageIndex}`); return found; } -export function useSubsManifest() { +export function useSubsManifest(origin = "") { return useQuery<SubsManifest>({ - queryKey: ["subs-manifest"], - queryFn: () => fetchJson<SubsManifest>("/subs/manifest.json"), + queryKey: ["subs-manifest", origin], + queryFn: () => fetchJson<SubsManifest>(`${idBaseUrl(origin)}/subs/manifest.json`), }); } // Loads each per-channel subs manifest so we can enumerate the full set of // sub-having slugs without fetching cue pages eagerly. Returns one entry per -// channel listed in the cross-channel manifest. Errors are absorbed per -// channel so a missing manifest doesn't break the whole UI. -export function useChannelSubsManifests(channelSlugs: string[]) { +// ref. Accepts either bare channel slugs (same-origin) or {channelSlug, origin} +// refs (federated). Errors are absorbed per channel so a missing manifest +// doesn't break the whole UI. +export function useChannelSubsManifests(refs: Array<string | SubsManifestRef>) { return useQueries({ - queries: channelSlugs.map((channelSlug) => ({ - queryKey: ["channel-subs-manifest", channelSlug], - queryFn: () => fetchChannelSubsManifest(channelSlug), - })), + queries: refs.map((ref) => { + const { channelSlug, origin = "" } = + typeof ref === "string" ? { channelSlug: ref, origin: "" } : ref; + return { + queryKey: ["channel-subs-manifest", origin, channelSlug], + queryFn: () => fetchChannelSubsManifest(channelSlug, origin), + }; + }), }); } diff --git a/common/components/summariesCache.ts b/common/components/summariesCache.ts @@ -5,6 +5,7 @@ import { useQueries, useQuery } from "@tanstack/react-query"; import type { DisplaySummary } from "../lib/transcripts"; import type { Manifest } from "../lib/manifest"; import { pageFileName } from "../lib/manifest"; +import { idBaseUrl } from "./originId"; async function fetchJson<T>(url: string): Promise<T> { const r = await fetch(url); @@ -21,23 +22,26 @@ export type SummariesState = { error: Error | null; }; -export function useManifest() { +export function useManifest(origin = "") { return useQuery<Manifest>({ - queryKey: ["manifest"], - queryFn: () => fetchJson<Manifest>("/summaries/manifest.json"), + queryKey: ["manifest", origin], + queryFn: () => + fetchJson<Manifest>(`${idBaseUrl(origin)}/summaries/manifest.json`), }); } -export function useSummaries(): SummariesState { - const manifestQuery = useManifest(); +export function useSummaries(origin = ""): SummariesState { + const manifestQuery = useManifest(origin); const manifest = manifestQuery.data ?? null; const pageCount = manifest?.pageCount ?? 0; const pageQueries = useQueries({ queries: Array.from({ length: pageCount }, (_, i) => ({ - queryKey: ["summaries-page", i], + queryKey: ["summaries-page", origin, i], queryFn: () => - fetchJson<DisplaySummary[]>(`/summaries/${pageFileName(i)}`), + fetchJson<DisplaySummary[]>( + `${idBaseUrl(origin)}/summaries/${pageFileName(i)}`, + ), enabled: pageCount > 0, })), }); diff --git a/common/components/transcriptCache.ts b/common/components/transcriptCache.ts @@ -4,6 +4,7 @@ import type { TranscriptDetail } from "../lib/transcripts"; import type { ChannelTranscriptsManifest } from "../lib/manifest"; import { transcriptPageFileName } from "../lib/manifest"; import { idbGet, idbPutBatch } from "./transcriptStore"; +import { makeId, splitId, idBaseUrl } from "./originId"; const resolved = new Map<string, TranscriptDetail>(); const inFlight = new Map<string, Promise<TranscriptDetail>>(); @@ -13,38 +14,43 @@ const channelManifests = new Map< >(); const pagePromises = new Map<string, Promise<TranscriptDetail[]>>(); -export function fetchTranscript(slug: string): Promise<TranscriptDetail> { - const hit = resolved.get(slug); +// `id` is an OriginId: a bare "channelSlug/videoId" for same-origin content, or +// "origin\tchannelSlug/videoId" for a federated cross-origin video. Maps and +// the IDB store are keyed by the full id, so origins never collide. +export function fetchTranscript(id: string): Promise<TranscriptDetail> { + const hit = resolved.get(id); if (hit) return Promise.resolve(hit); - const flying = inFlight.get(slug); + const flying = inFlight.get(id); if (flying) return flying; - const p = load(slug).then((detail) => { - resolved.set(slug, detail); - inFlight.delete(slug); + const p = load(id).then((detail) => { + resolved.set(id, detail); + inFlight.delete(id); return detail; }); p.catch(() => { - inFlight.delete(slug); + inFlight.delete(id); }); - inFlight.set(slug, p); + inFlight.set(id, p); return p; } function fetchChannelManifest( channelSlug: string, + origin: string, ): Promise<ChannelTranscriptsManifest> { - let p = channelManifests.get(channelSlug); + const key = makeId(origin, channelSlug); + let p = channelManifests.get(key); if (!p) { - p = fetch(`/transcripts/${channelSlug}/manifest.json`).then((r) => { + p = fetch(`${idBaseUrl(origin)}/transcripts/${channelSlug}/manifest.json`).then((r) => { if (!r.ok) throw new Error( `Failed to fetch transcripts manifest for ${channelSlug}`, ); return r.json() as Promise<ChannelTranscriptsManifest>; }); - p.catch(() => channelManifests.delete(channelSlug)); - channelManifests.set(channelSlug, p); + p.catch(() => channelManifests.delete(key)); + channelManifests.set(key, p); } return p; } @@ -52,12 +58,13 @@ function fetchChannelManifest( function fetchPage( channelSlug: string, pageIndex: number, + origin: string, ): Promise<TranscriptDetail[]> { - const key = `${channelSlug}:${pageIndex}`; + const key = `${makeId(origin, channelSlug)}:${pageIndex}`; let p = pagePromises.get(key); if (!p) { p = fetch( - `/transcripts/${channelSlug}/${transcriptPageFileName(pageIndex)}`, + `${idBaseUrl(origin)}/transcripts/${channelSlug}/${transcriptPageFileName(pageIndex)}`, ).then((r) => { if (!r.ok) throw new Error( @@ -71,27 +78,31 @@ function fetchPage( return p; } -async function load(slug: string): Promise<TranscriptDetail> { - const stored = await idbGet(slug); +async function load(id: string): Promise<TranscriptDetail> { + const stored = await idbGet(id); if (stored) return stored; + const { origin, slug } = splitId(id); const slashIdx = slug.indexOf("/"); if (slashIdx < 0) throw new Error(`Malformed transcript slug: ${slug}`); const channelSlug = slug.slice(0, slashIdx); const videoId = slug.slice(slashIdx + 1); - const manifest = await fetchChannelManifest(channelSlug); + const manifest = await fetchChannelManifest(channelSlug, origin); const pageIndex = manifest.slugToPage[videoId]; if (pageIndex === undefined) throw new Error(`Unknown transcript slug: ${slug}`); - const page = await fetchPage(channelSlug, pageIndex); + const page = await fetchPage(channelSlug, pageIndex, origin); let found: TranscriptDetail | undefined; for (const entry of page) { + // Re-key warmed entries by their OriginId so a cross-origin + // channelSlug/videoId can't shadow a same-origin one with the same slug. + const entryId = makeId(origin, entry.slug); if (entry.slug === slug) found = entry; - // Opportunistically warm the per-slug memory + IDB caches so subsequent + // Opportunistically warm the per-id memory + IDB caches so subsequent // fetches for other videos in this page hit without a network round-trip. // Huge win for searchPipeline which iterates many slugs from the same // channel/page. - resolved.set(entry.slug, entry); - idbPutBatch(entry); + resolved.set(entryId, entry); + idbPutBatch(entryId, entry); } if (!found) throw new Error(`Transcript ${slug} missing from page ${pageIndex}`); diff --git a/common/components/transcriptStore.ts b/common/components/transcriptStore.ts @@ -4,10 +4,13 @@ import type { TranscriptDetail } from "../lib/transcripts"; import { PLATFORM_VALUES } from "../lib/platform"; const DB_NAME = "yt-dlp-transcript-browser"; -// Bump to invalidate stale cached entries on existing clients. v2 wipes -// pre-multi-platform records that were cached without a `platform` field -// (they would otherwise fall through to the YouTube player forever). -const DB_VERSION = 2; +// Bump to invalidate stale cached entries on existing clients. v2 wiped +// pre-multi-platform records cached without a `platform` field. v3 switches to +// out-of-line keys (the record is stored under an OriginId, not its `slug` +// keyPath) so cross-origin content in the hub can't collide with same-origin +// content that happens to share a channelSlug/videoId. Same-origin ids equal +// the bare slug, so behavior is unchanged after the one-time wipe. +const DB_VERSION = 3; const STORE = "transcripts"; type Mode = "pending" | "ok" | "unavailable"; @@ -38,7 +41,8 @@ function openDb(): Promise<IDBDatabase | null> { if (db.objectStoreNames.contains(STORE)) { db.deleteObjectStore(STORE); } - db.createObjectStore(STORE, { keyPath: "slug" }); + // Out-of-line keys: the caller supplies an OriginId key (see idbPutBatch). + db.createObjectStore(STORE); }; req.onsuccess = () => { mode = "ok"; @@ -65,7 +69,7 @@ function downgrade(reason: string): void { } export async function idbGet( - slug: string, + id: string, ): Promise<TranscriptDetail | null> { const db = await openDb(); if (!db) return null; @@ -73,7 +77,7 @@ export async function idbGet( let req: IDBRequest<TranscriptDetail | undefined>; try { const tx = db.transaction(STORE, "readonly"); - req = tx.objectStore(STORE).get(slug) as IDBRequest< + req = tx.objectStore(STORE).get(id) as IDBRequest< TranscriptDetail | undefined >; } catch { @@ -96,13 +100,16 @@ export async function idbGet( }); } -let pending: TranscriptDetail[] = []; +type PendingEntry = { id: string; detail: TranscriptDetail }; +let pending: PendingEntry[] = []; let flushScheduled = false; let flushInFlight: Promise<void> | null = null; -export function idbPutBatch(detail: TranscriptDetail): void { +// Store `detail` under the given OriginId (out-of-line key). Same-origin ids +// are the bare slug, so single-site behavior is unchanged. +export function idbPutBatch(id: string, detail: TranscriptDetail): void { if (mode === "unavailable") return; - pending.push(detail); + pending.push({ id, detail }); if (flushScheduled) return; flushScheduled = true; queueMicrotask(() => { @@ -116,7 +123,7 @@ async function flush(): Promise<void> { await flushInFlight; } if (pending.length === 0) return; - const batch = pending; + const batch: PendingEntry[] = pending; pending = []; flushInFlight = writeBatch(batch).finally(() => { flushInFlight = null; @@ -131,7 +138,7 @@ async function flush(): Promise<void> { await flushInFlight; } -async function writeBatch(batch: TranscriptDetail[]): Promise<void> { +async function writeBatch(batch: PendingEntry[]): Promise<void> { const db = await openDb(); if (!db) return; return new Promise<void>((resolve) => { @@ -146,7 +153,7 @@ async function writeBatch(batch: TranscriptDetail[]): Promise<void> { const store = tx.objectStore(STORE); for (const entry of batch) { try { - store.put(entry); + store.put(entry.detail, entry.id); } catch { // Per-entry errors (e.g. unclonable values) shouldn't fail the batch. } diff --git a/editor/e2e/export-player-platform-cache.spec.ts b/editor/e2e/export-player-platform-cache.spec.ts @@ -118,18 +118,20 @@ async function seedCache( ): Promise<void> { await page.evaluate(async (record) => { await new Promise<void>((resolve, reject) => { - // Match transcriptStore.ts DB_NAME / DB_VERSION / STORE. - const open = indexedDB.open("yt-dlp-transcript-browser", 2); + // Match transcriptStore.ts DB_NAME / DB_VERSION / STORE. v3 uses + // out-of-line keys: the record is stored under its OriginId (for + // same-origin content that's just the bare slug). + const open = indexedDB.open("yt-dlp-transcript-browser", 3); open.onupgradeneeded = () => { const db = open.result; if (!db.objectStoreNames.contains("transcripts")) { - db.createObjectStore("transcripts", { keyPath: "slug" }); + db.createObjectStore("transcripts"); } }; open.onsuccess = () => { const db = open.result; const tx = db.transaction("transcripts", "readwrite"); - tx.objectStore("transcripts").put(record); + tx.objectStore("transcripts").put(record, record.slug as string); tx.oncomplete = () => resolve(); tx.onerror = () => reject(tx.error); };