Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a6533a81bb2e33f9c90fa6e7b7ee04b4ae295c2e
parent 001b3cf6359702c398faa8dfccbba47c7bef3275
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 12 Sep 2026 02:53:48 -0400

common: lib/ no longer imports components/, and the guard says so

The deliverable of this slice, not a test that was bent to pass. Four
back-edges existed; all four are gone and `"components"` joins
`FORBIDDEN.lib` in `architecture.test.ts` with NOTHING added to the
allow-list. The list can only shrink, so it did not grow.

  searchEval -> components/searchPipeline   the leaf pipeline moved DOWN
      (commit 1) and is now INJECTED: `runQueryTree` takes a
      `SearchRuntime` — a fetch-bound leaf runner plus a layer memo —
      instead of importing one. `components/searchPipeline.ts` is 904
      lines lighter and supplies `searchRuntime`; the three call sites
      (SearchSessionContext, useSearchSeries, export's askRetrieval)
      pass it.
  searchEval -> components/searchLayerCache the same injection. The
      memo's `CachedResult` belongs to the tree that defines it, so it
      moved to searchEval and searchLayerCache re-exports it: an
      IndexedDB store does not get to define what a leaf result IS.
  aiHandoff -> components/searchPipeline    `LayerHit` is the shared
      pipeline's type now; the import path is the only change.
  searchQuery -> components/urlState        `SearchMode` moved to the
      module that INTERPRETS it (`buildSearchRoot` turns one into a
      query tree). `urlState` re-exports it, so its two importers are
      untouched.

No import site outside these files changed: `components/searchPipeline`
still exports every name it did, and `export/app/lib/askRetrieval.ts`
keeps importing `LayerHit` from it.

tsc clean in all six packages; common 1128 passed (1077 + 51 new, none
lost), architecture test green; export `next build` compiled
successfully — which is the only real test that none of this dragged
`reader-fs.ts` or a react-query module into a place it does not belong.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/architecture.test.ts | 11+++++++++--
Mcommon/components/SearchSessionContext.tsx | 3++-
Mcommon/components/charts/useSearchSeries.ts | 2++
Mcommon/components/searchLayerCache.ts | 13+++++++------
Mcommon/components/searchPipeline.ts | 904++++++-------------------------------------------------------------------------
Mcommon/components/urlState.ts | 5++++-
Mcommon/lib/aiHandoff.test.ts | 2+-
Mcommon/lib/aiHandoff.ts | 2+-
Mcommon/lib/searchEval.ts | 77++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------------------
Mcommon/lib/searchQuery.ts | 6+++++-
Mexport/app/lib/askRetrieval.ts | 6+++++-
11 files changed, 152 insertions(+), 879 deletions(-)

diff --git a/common/architecture.test.ts b/common/architecture.test.ts @@ -29,7 +29,13 @@ const HERE = path.dirname(fileURLToPath(import.meta.url)); // Which directory may not import which. Read as "lib/ may not import // controller/ or jobs/". const FORBIDDEN: Record<string, readonly string[]> = { - lib: ["controller", "jobs"], + // `components` joined this row in one-core phase 2 slice S3. The model layer + // importing the view layer was four edges: searchEval reaching for the leaf + // pipeline and the IndexedDB layer memo, aiHandoff for a hit type, and + // searchQuery for a mode union. All four inverted — the pipeline and the two + // types moved down, and the fetch and the memo are now injected — so the + // guard turns on with nothing added to the allow-list to pay for it. + lib: ["controller", "jobs", "components"], jobs: ["controller"], components: ["controller", "jobs", "ytdlp"], }; @@ -156,7 +162,8 @@ test("no new back-edges between common's layers", async () => { assert.deepEqual( unexpected, [], - `NEW back-edge(s) in common/. lib/ may not import controller/ or jobs/; ` + + `NEW back-edge(s) in common/. lib/ may not import controller/, jobs/ or ` + + `components/; ` + `jobs/ may not import controller/; components/ may not import ` + `controller/, jobs/ or ytdlp/. Move the type or the function down a ` + `layer instead of adding it to ALLOWED. Found: ${unexpected.join(", ")}`, diff --git a/common/components/SearchSessionContext.tsx b/common/components/SearchSessionContext.tsx @@ -26,7 +26,7 @@ import { usePlayer } from "./PlayerProvider"; import { useChannelSubsManifests } from "./subsCache"; import { peekPost, useChannelPostsManifests } from "./postsCache"; import { useSearchData } from "./SearchDataContext"; -import type { LayerHit } from "./searchPipeline"; +import { searchRuntime, type LayerHit } from "./searchPipeline"; import { runQueryTree, type GroupState, @@ -760,6 +760,7 @@ function useSearchSessionState() { const controller = runQueryTree({ root: committedRoot, + runtime: searchRuntime, globalScope: globalScopeSlugs, summaries: transcripts, chatScopeSlugs: needsChatManifests ? chatScopeSlugs : null, diff --git a/common/components/charts/useSearchSeries.ts b/common/components/charts/useSearchSeries.ts @@ -8,6 +8,7 @@ import { applyFilters, type ChartData } from "../../lib/chartAggregate"; import { seriesFromSlugs } from "../../lib/chartSeriesFromSlugs"; import { parseRoot } from "../../lib/searchQuery"; import { runQueryTree, type TreeProgress } from "../../lib/searchEval"; +import { searchRuntime } from "../searchPipeline"; export type SearchSeriesState = { data: ChartData; @@ -56,6 +57,7 @@ export function useSearchSeries( try { const controller = runQueryTree({ root, + runtime: searchRuntime, globalScope: scopeSlugs, summaries, initialHitLimit: 50000, diff --git a/common/components/searchLayerCache.ts b/common/components/searchLayerCache.ts @@ -12,7 +12,13 @@ // IndexedDB writes are batched via queueMicrotask the same way as // transcriptStore.ts. -import type { LayerHit } from "./searchPipeline"; +import type { LayerHit } from "../lib/search/leafPipeline"; +import type { CachedResult } from "../lib/searchEval"; + +// Re-exported so this module's importers keep their import site. The type +// belongs to `lib/searchEval.ts` now: it is the shape of a memoized leaf +// result, and the tree — not its storage — defines it. +export type { CachedResult }; // Separate DB from `transcriptStore.ts` (`yt-dlp-transcript-browser`) so the // two stores can evolve their schemas independently — bumping the version @@ -22,11 +28,6 @@ const DB_VERSION = 1; const STORE = "layers"; const MAX_ENTRIES = 500; -export type CachedResult = { - slugs: ReadonlySet<string>; - hits: ReadonlyMap<string, LayerHit[]>; -}; - type StoredEntry = { key: string; slugs: string[]; diff --git a/common/components/searchPipeline.ts b/common/components/searchPipeline.ts @@ -1,852 +1,68 @@ +// THE BINDING between the shared search pipeline and this app's caches. +// +// Everything that used to live here — the three worker-pool drivers, the leaf +// adapter, the cue/text matchers — is `lib/search/leafPipeline.ts` now, where +// `lib/searchEval.ts` can reach it without `lib/` importing `components/`. +// What is left is the one thing that genuinely belongs to the view layer: WHICH +// FETCH. `fetchTranscript` / `fetchSubs` / `fetchPost` are the react-query-backed +// page caches, and they are supplied here rather than imported there. +// +// Every name this module exported before is still exported from it, so the five +// files that import `LayerHit` from `components/searchPipeline` (including +// `export/app/lib/askRetrieval.ts`) are untouched. + import { fetchTranscript } from "./transcriptCache"; import { fetchSubs } from "./subsCache"; import { fetchPost } from "./postsCache"; -import type { DisplaySummary } from "../lib/transcripts"; -import type { LayerScope, LeafNode } from "../lib/searchQuery"; - -export type Hit = { start: number; text: string }; - -export type SubsHit = { track: string; start: number; text: string }; - -// Per-leaf hit shape consumed by the composite-search tree (searchEval.ts). -// Carries enough info for the result-list UI to render the hit with its -// originating layer's swatch + scope-specific decorations. -export type LayerHit = { - leafId: string; - scope: LayerScope; - track?: string; - start: number; - text: string; -}; - -export type PipelineUpdate = { - hitsBySlug: Record<string, Hit[]>; - totalHits: number; - processed: number; - totalToProcess: number; - capped: boolean; - done: boolean; -}; - -type PipelineConfig = { - slugs: string[]; - query: string; - useRegex: boolean; - regex: RegExp | null; - initialHitLimit: number; - concurrency: number; - flushIntervalMs: number; - emit: (update: PipelineUpdate) => void; - // Which field of the fetched transcript document to match. "cues" (default) - // is the transcripts scope; "description"/"tags" match those metadata fields. - // All reuse the same per-channel transcript-page fetch + worker pool. - matchField?: "cues" | "description" | "tags"; +import { + runLeafPipeline as runLeafPipelineWith, + type LeafController, + type LeafFetchers, + type LeafPipelineOptions, +} from "../lib/search/leafPipeline"; +import { + cacheKey, + getCached, + getCachedSync, + putCached, +} from "./searchLayerCache"; +import type { SearchRuntime } from "../lib/searchEval"; + +export type { + Hit, + SubsHit, + LayerHit, + PipelineUpdate, + PipelineController, + SubsPipelineUpdate, + LeafProgress, + LeafResult, + LeafController, +} from "../lib/search/leafPipeline"; +export { filterSlugs } from "../lib/search/leafPipeline"; + +// This app's three fetches, as one injectable bundle. +const CACHE_FETCHERS: LeafFetchers = { + transcript: fetchTranscript, + subs: fetchSubs, + post: fetchPost, }; -export type PipelineController = { - cancel(): void; - setHitLimit(limit: number): void; -}; - -// Runs a streaming search over the given slugs. State is closure-captured so -// `setHitLimit` can raise the cap and re-spawn workers without restarting -// traversal — workers resume from the shared `idx` cursor. -export function createSearchPipeline( - config: PipelineConfig, -): PipelineController { - const { - slugs, - query, - useRegex, - regex, - initialHitLimit, - concurrency, - flushIntervalMs, - emit, - matchField = "cues", - } = config; - - let cancelled = false; - // `idx` is the next-to-claim cursor (leading edge); `completed` counts - // slugs that workers have finished scanning (trailing edge). We report - // `completed` to the UI so progress reflects actual work done, not - // in-flight claims — and it never exceeds slugs.length. - let idx = 0; - let completed = 0; - let totalSoFar = 0; - let hitLimit = initialHitLimit; - let activeWorkers = 0; - let done = false; - const localHits: Record<string, Hit[]> = {}; - let flushTimer: number | null = null; - - const pushUpdate = (overrides: Partial<PipelineUpdate> = {}) => { - emit({ - hitsBySlug: { ...localHits }, - totalHits: totalSoFar, - processed: completed, - totalToProcess: slugs.length, - // `capped` keys off `idx` (claimed), not `completed` — raising the cap - // only helps if there are still unclaimed slugs a worker can pick up. - capped: totalSoFar >= hitLimit && idx < slugs.length, - done, - ...overrides, - }); - }; - - const scheduleFlush = () => { - if (flushTimer !== null || cancelled) return; - flushTimer = window.setTimeout(() => { - flushTimer = null; - if (cancelled) return; - pushUpdate(); - }, flushIntervalMs); - }; - - const finalize = () => { - if (done || cancelled) return; - done = true; - if (flushTimer !== null) { - window.clearTimeout(flushTimer); - flushTimer = null; - } - pushUpdate(); - }; - - const worker = async () => { - activeWorkers++; - try { - while (!cancelled) { - if (totalSoFar >= hitLimit) return; - if (idx >= slugs.length) return; - const my = idx++; - const slug = slugs[my]; - try { - const full = await fetchTranscript(slug); - if (cancelled) return; - if (totalSoFar < hitLimit) { - const remaining = hitLimit - totalSoFar; - const hits = - matchField === "description" - ? findHitsInText(full.description ?? "", query, useRegex, regex) - : matchField === "tags" - ? findHitsInText( - (full.tags ?? []).join(", "), - query, - useRegex, - regex, - ) - : full.cues - ? findHitsInCues(full.cues, query, useRegex, regex, remaining) - : []; - if (hits.length > 0) { - localHits[slug] = hits; - totalSoFar += hits.length; - } - } - } catch { - // ignore per-transcript failures - } - completed++; - scheduleFlush(); - } - } finally { - activeWorkers--; - if (activeWorkers === 0 && !cancelled) { - // Either we hit the cap or ran out of work — either way, settle. - finalize(); - } - } - }; - - const ensureWorkers = () => { - if (cancelled || done) return; - if (totalSoFar >= hitLimit) return; - if (idx >= slugs.length) return; - const needed = Math.min( - concurrency - activeWorkers, - slugs.length - idx, - ); - for (let i = 0; i < needed; i++) worker(); - }; - - // Emit initial snapshot synchronously so the UI clears previous results. - pushUpdate(); - ensureWorkers(); - - return { - cancel() { - cancelled = true; - if (flushTimer !== null) { - window.clearTimeout(flushTimer); - flushTimer = null; - } - }, - setHitLimit(limit: number) { - if (cancelled) return; - if (limit <= hitLimit) return; - hitLimit = limit; - // The prior run may have finalized because we were capped. Resume. - if (done) { - done = false; - pushUpdate(); - } - ensureWorkers(); - }, - }; -} - -function findHitsInCues( - cues: { start: number; text: string }[], - query: string, - useRegex: boolean, - regex: RegExp | null, - limit: number, -): Hit[] { - const hits: Hit[] = []; - for (let i = 0; i < cues.length && hits.length < limit; i++) { - const cur = cues[i]; - const prevText = i > 0 ? cues[i - 1].text : ""; - const nextText = i < cues.length - 1 ? cues[i + 1].text : ""; - const sep1 = prevText ? " " : ""; - const sep2 = nextText ? " " : ""; - const windowText = prevText + sep1 + cur.text + sep2 + nextText; - const curStart = prevText.length + sep1.length; - const curEnd = curStart + cur.text.length; - const m = findFirstMatchInRange( - windowText, - curStart, - curEnd, - query, - useRegex, - regex, - ); - if (!m) continue; - const crosses = m.idx < curStart || m.idx + m.length > curEnd; - hits.push({ - start: Math.round(cur.start), - text: crosses ? windowText : cur.text, - }); - } - return hits; -} - -// Single-document text match (description scope): emit at most one hit with a -// snippet window around the first match, so the result row shows context. -function findHitsInText( - text: string, - query: string, - useRegex: boolean, - regex: RegExp | null, -): Hit[] { - if (!text) return []; - const m = findFirstMatchInRange(text, 0, text.length, query, useRegex, regex); - if (!m) return []; - const PAD = 80; - const from = Math.max(0, m.idx - PAD); - const to = Math.min(text.length, m.idx + m.length + PAD); - const snippet = - (from > 0 ? "…" : "") + - text.slice(from, to).replace(/\s+/g, " ").trim() + - (to < text.length ? "…" : ""); - return [{ start: 0, text: snippet }]; -} - -function findFirstMatchInRange( - haystack: string, - rangeStart: number, - rangeEnd: number, - query: string, - useRegex: boolean, - regex: RegExp | null, -): { idx: number; length: number } | null { - if (useRegex) { - if (!regex) return null; - const flags = regex.flags.includes("g") ? regex.flags : regex.flags + "g"; - const re = new RegExp(regex.source, flags); - let m: RegExpExecArray | null; - while ((m = re.exec(haystack)) !== null) { - if (m.index >= rangeStart && m.index < rangeEnd) { - return { idx: m.index, length: m[0].length }; - } - if (m.index >= rangeEnd) return null; - if (m[0].length === 0) re.lastIndex++; - } - return null; - } - if (!query) return null; - const lower = haystack.toLowerCase(); - const ql = query.toLowerCase(); - let from = 0; - while (from <= haystack.length) { - const idx = lower.indexOf(ql, from); - if (idx === -1) return null; - if (idx >= rangeStart && idx < rangeEnd) return { idx, length: ql.length }; - if (idx >= rangeEnd) return null; - from = idx + 1; - } - return null; -} - -// Streaming search over the social-post corpus. Structurally the transcripts -// pipeline with a different fetch + match: the unit of iteration is a POST -// slug (`<channelSlug>/<postId>`), and fetchPost() resolves through the page -// cache, so the first post on a page warms every other post on it. -// -// Post bodies are short but unbounded, and unlike the cue path there is no -// natural per-line unit to clip to — so hits are truncated to a ±80-char -// window (the same one the description/tags scopes use) rather than shipping -// the whole body into a result card. -export function createPostsSearchPipeline( - config: Omit<PipelineConfig, "matchField">, -): PipelineController { - const { - slugs, - query, - useRegex, - regex, - initialHitLimit, - concurrency, - flushIntervalMs, - emit, - } = config; - - let cancelled = false; - let idx = 0; - let completed = 0; - let totalSoFar = 0; - let hitLimit = initialHitLimit; - let activeWorkers = 0; - let done = false; - const localHits: Record<string, Hit[]> = {}; - let flushTimer: number | null = null; - - const pushUpdate = (overrides: Partial<PipelineUpdate> = {}) => { - emit({ - hitsBySlug: { ...localHits }, - totalHits: totalSoFar, - processed: completed, - totalToProcess: slugs.length, - capped: totalSoFar >= hitLimit && idx < slugs.length, - done, - ...overrides, - }); - }; - - const scheduleFlush = () => { - if (flushTimer !== null || cancelled) return; - flushTimer = window.setTimeout(() => { - flushTimer = null; - if (cancelled) return; - pushUpdate(); - }, flushIntervalMs); - }; - - const finalize = () => { - if (done || cancelled) return; - done = true; - if (flushTimer !== null) { - window.clearTimeout(flushTimer); - flushTimer = null; - } - pushUpdate(); - }; - - const worker = async () => { - activeWorkers++; - try { - while (!cancelled) { - if (totalSoFar >= hitLimit) return; - if (idx >= slugs.length) return; - const my = idx++; - const slug = slugs[my]; - try { - const post = await fetchPost(slug); - if (cancelled) return; - if (totalSoFar < hitLimit) { - const hits = findHitsInText(post.text, query, useRegex, regex); - if (hits.length > 0) { - localHits[slug] = hits; - totalSoFar += hits.length; - } - } - } catch { - // ignore per-post failures - } - completed++; - scheduleFlush(); - } - } finally { - activeWorkers--; - if (activeWorkers === 0 && !cancelled) finalize(); - } - }; - - const ensureWorkers = () => { - if (cancelled || done) return; - if (totalSoFar >= hitLimit) return; - if (idx >= slugs.length) return; - const needed = Math.min(concurrency - activeWorkers, slugs.length - idx); - for (let i = 0; i < needed; i++) worker(); - }; - - pushUpdate(); - ensureWorkers(); - - return { - cancel() { - cancelled = true; - if (flushTimer !== null) { - window.clearTimeout(flushTimer); - flushTimer = null; - } - }, - setHitLimit(limit: number) { - if (cancelled) return; - if (limit <= hitLimit) return; - hitLimit = limit; - if (done) { - done = false; - pushUpdate(); - } - ensureWorkers(); - }, - }; +export function runLeafPipeline( + opts: Omit<LeafPipelineOptions, "fetchers">, +): LeafController { + return runLeafPipelineWith({ ...opts, fetchers: CACHE_FETCHERS }); } -// Helper to build slug list from summaries + filter predicate. -export function filterSlugs( - summaries: DisplaySummary[], - passes: (t: DisplaySummary) => boolean, -): string[] { - const slugs: string[] = []; - for (const t of summaries) if (passes(t)) slugs.push(t.slug); - return slugs; -} - -export type SubsPipelineUpdate = { - hitsBySlug: Record<string, SubsHit[]>; - totalHits: number; - processed: number; - totalToProcess: number; - capped: boolean; - done: boolean; -}; - -type SubsPipelineConfig = { - slugs: string[]; - query: string; - useRegex: boolean; - regex: RegExp | null; - // Tracks to exclude. Empty set means "search all tracks". - excludedTracks: Set<string>; - // Optional inclusion filter. When provided, ONLY tracks in this set are - // scanned (and `excludedTracks` is irrelevant). Used by the composite- - // search "chat" scope to limit scanning to live_chat regardless of which - // other tracks a video has. - includedTracks?: Set<string> | null; - initialHitLimit: number; - concurrency: number; - flushIntervalMs: number; - emit: (update: SubsPipelineUpdate) => void; -}; - -export function createSubsSearchPipeline( - config: SubsPipelineConfig, -): PipelineController { - const { - slugs, - query, - useRegex, - regex, - excludedTracks, - includedTracks, - initialHitLimit, - concurrency, - flushIntervalMs, - emit, - } = config; - - let cancelled = false; - let idx = 0; - let completed = 0; - let totalSoFar = 0; - let hitLimit = initialHitLimit; - let activeWorkers = 0; - let done = false; - const localHits: Record<string, SubsHit[]> = {}; - let flushTimer: number | null = null; - - const pushUpdate = (overrides: Partial<SubsPipelineUpdate> = {}) => { - emit({ - hitsBySlug: { ...localHits }, - totalHits: totalSoFar, - processed: completed, - totalToProcess: slugs.length, - capped: totalSoFar >= hitLimit && idx < slugs.length, - done, - ...overrides, - }); - }; - - const scheduleFlush = () => { - if (flushTimer !== null || cancelled) return; - flushTimer = window.setTimeout(() => { - flushTimer = null; - if (cancelled) return; - pushUpdate(); - }, flushIntervalMs); - }; - - const finalize = () => { - if (done || cancelled) return; - done = true; - if (flushTimer !== null) { - window.clearTimeout(flushTimer); - flushTimer = null; - } - pushUpdate(); - }; - - const worker = async () => { - activeWorkers++; - try { - while (!cancelled) { - if (totalSoFar >= hitLimit) return; - if (idx >= slugs.length) return; - const my = idx++; - const slug = slugs[my]; - try { - const detail = await fetchSubs(slug); - if (cancelled) return; - if (totalSoFar < hitLimit) { - const slugHits: SubsHit[] = []; - for (const [track, cueList] of Object.entries(detail.tracks)) { - if (includedTracks) { - if (!includedTracks.has(track)) continue; - } else if (excludedTracks.has(track)) continue; - if (totalSoFar + slugHits.length >= hitLimit) break; - const remaining = - hitLimit - totalSoFar - slugHits.length; - const trackHits = findHitsInCues( - cueList, - query, - useRegex, - regex, - remaining, - ); - for (const h of trackHits) { - slugHits.push({ track, start: h.start, text: h.text }); - } - } - if (slugHits.length > 0) { - // Order hits by start time so live_chat and language tracks - // interleave naturally instead of being grouped per-track. - slugHits.sort((a, b) => a.start - b.start); - localHits[slug] = slugHits; - totalSoFar += slugHits.length; - } - } - } catch { - // ignore per-video failures - } - completed++; - scheduleFlush(); - } - } finally { - activeWorkers--; - if (activeWorkers === 0 && !cancelled) finalize(); - } - }; - - const ensureWorkers = () => { - if (cancelled || done) return; - if (totalSoFar >= hitLimit) return; - if (idx >= slugs.length) return; - const needed = Math.min( - concurrency - activeWorkers, - slugs.length - idx, - ); - for (let i = 0; i < needed; i++) worker(); - }; - - pushUpdate(); - ensureWorkers(); - - return { - cancel() { - cancelled = true; - if (flushTimer !== null) { - window.clearTimeout(flushTimer); - flushTimer = null; - } - }, - setHitLimit(limit: number) { - if (cancelled) return; - if (limit <= hitLimit) return; - hitLimit = limit; - if (done) { - done = false; - pushUpdate(); - } - ensureWorkers(); - }, - }; -} - -// ─── Leaf pipeline wrapper ─── -// Thin adapter over `createSearchPipeline` / `createSubsSearchPipeline` for -// the composite-search tree (`searchEval.ts`). One controller per leaf in the -// tree. Returns a `LeafController` that the tree orchestrator can cancel / -// resize, plus a `done` promise that resolves with the final LeafResult. -// -// Scope=metadata isn't handled here — searchEval.ts evaluates it synchronously -// over the summaries cache. This wrapper only deals with the network-backed -// scopes (transcripts, chat) where the existing worker pool earns its keep. - -export type LeafProgress = { - slugs: Set<string>; - hits: Map<string, LayerHit[]>; - totalHits: number; - processed: number; - totalToProcess: number; - capped: boolean; - done: boolean; +// What `runQueryTree` is handed: the fetch-bound leaf runner plus the +// IndexedDB-backed layer memo. `lib/searchEval.ts` owns the tree algebra and +// knows nothing about either. +export const searchRuntime: SearchRuntime = { + runLeaf: runLeafPipeline, + cache: { + key: cacheKey, + getSync: getCachedSync, + get: getCached, + put: putCached, + }, }; - -export type LeafResult = { - slugs: Set<string>; - hits: Map<string, LayerHit[]>; -}; - -export type LeafController = PipelineController & { - done: Promise<LeafResult>; -}; - -export function runLeafPipeline(opts: { - leaf: LeafNode; - scopeSlugs: string[]; - initialHitLimit: number; - concurrency: number; - flushIntervalMs: number; - emit: (p: LeafProgress) => void; -}): LeafController { - const { - leaf, - scopeSlugs, - initialHitLimit, - concurrency, - flushIntervalMs, - emit, - } = opts; - - const trimmed = leaf.query.trim(); - // Caller guarantees query is non-empty before invoking us (see - // isLeafActive). Stay defensive: empty queries produce an immediate done - // with no hits. - if (!trimmed) { - const empty: LeafResult = { slugs: new Set(), hits: new Map() }; - queueMicrotask(() => { - emit({ - slugs: empty.slugs, - hits: empty.hits, - totalHits: 0, - processed: 0, - totalToProcess: 0, - capped: false, - done: true, - }); - }); - return { - cancel() { - /* no-op */ - }, - setHitLimit() { - /* no-op */ - }, - done: Promise.resolve(empty), - }; - } - - const regex = compileLeafRegex(leaf); - - let resolveDone!: (r: LeafResult) => void; - const donePromise = new Promise<LeafResult>((resolve) => { - resolveDone = resolve; - }); - - // Latest seen progress, kept locally so `done` settles with the final - // payload that the consumer already saw on the last emit. - let finalSlugs = new Set<string>(); - let finalHits = new Map<string, LayerHit[]>(); - let settled = false; - - const adaptTranscriptUpdate = (u: PipelineUpdate): LeafProgress => { - const slugs = new Set<string>(); - const hits = new Map<string, LayerHit[]>(); - for (const [slug, list] of Object.entries(u.hitsBySlug)) { - if (!list || list.length === 0) continue; - slugs.add(slug); - if (leaf.contributeHits) { - hits.set( - slug, - list.map((h) => ({ - leafId: leaf.id, - scope: leaf.scope, - start: h.start, - text: h.text, - })), - ); - } - } - return { - slugs, - hits, - totalHits: u.totalHits, - processed: u.processed, - totalToProcess: u.totalToProcess, - capped: u.capped, - done: u.done, - }; - }; - - const adaptSubsUpdate = (u: SubsPipelineUpdate): LeafProgress => { - const slugs = new Set<string>(); - const hits = new Map<string, LayerHit[]>(); - for (const [slug, list] of Object.entries(u.hitsBySlug)) { - if (!list || list.length === 0) continue; - slugs.add(slug); - if (leaf.contributeHits) { - hits.set( - slug, - list.map((h) => ({ - leafId: leaf.id, - scope: "chat" as const, - track: h.track, - start: h.start, - text: h.text, - })), - ); - } - } - return { - slugs, - hits, - totalHits: u.totalHits, - processed: u.processed, - totalToProcess: u.totalToProcess, - capped: u.capped, - done: u.done, - }; - }; - - const onProgress = (p: LeafProgress) => { - finalSlugs = p.slugs; - finalHits = p.hits; - emit(p); - if (p.done && !settled) { - settled = true; - resolveDone({ slugs: finalSlugs, hits: finalHits }); - } - }; - - let controller: PipelineController; - if ( - leaf.scope === "transcripts" || - leaf.scope === "description" || - leaf.scope === "tags" - ) { - controller = createSearchPipeline({ - slugs: scopeSlugs, - query: trimmed, - useRegex: leaf.useRegex, - regex, - initialHitLimit, - concurrency, - flushIntervalMs, - matchField: - leaf.scope === "description" - ? "description" - : leaf.scope === "tags" - ? "tags" - : "cues", - emit: (u) => onProgress(adaptTranscriptUpdate(u)), - }); - } else if (leaf.scope === "posts") { - controller = createPostsSearchPipeline({ - slugs: scopeSlugs, - query: trimmed, - useRegex: leaf.useRegex, - regex, - initialHitLimit, - concurrency, - flushIntervalMs, - // A post has no timeline, so every hit is `start: 0` — the same - // convention description/tags/metadata hits already use. - emit: (u) => onProgress(adaptTranscriptUpdate(u)), - }); - } else if (leaf.scope === "chat") { - controller = createSubsSearchPipeline({ - slugs: scopeSlugs, - query: trimmed, - useRegex: leaf.useRegex, - regex, - excludedTracks: EMPTY_TRACK_SET, - includedTracks: CHAT_ONLY_TRACK_SET, - initialHitLimit, - concurrency, - flushIntervalMs, - emit: (u) => onProgress(adaptSubsUpdate(u)), - }); - } else { - // Metadata scope is handled by searchEval.ts directly. We shouldn't be - // invoked here; resolve immediately as a safety net. - const empty: LeafResult = { slugs: new Set(), hits: new Map() }; - queueMicrotask(() => { - onProgress({ - slugs: empty.slugs, - hits: empty.hits, - totalHits: 0, - processed: 0, - totalToProcess: 0, - capped: false, - done: true, - }); - }); - return { - cancel() { - /* no-op */ - }, - setHitLimit() { - /* no-op */ - }, - done: donePromise, - }; - } - - return { - cancel() { - controller.cancel(); - if (!settled) { - settled = true; - resolveDone({ slugs: finalSlugs, hits: finalHits }); - } - }, - setHitLimit(limit: number) { - controller.setHitLimit(limit); - }, - done: donePromise, - }; -} - -function compileLeafRegex(leaf: LeafNode): RegExp | null { - if (!leaf.useRegex) return null; - try { - return new RegExp(leaf.query, "i"); - } catch { - return null; - } -} - -const EMPTY_TRACK_SET: Set<string> = new Set(); -const CHAT_ONLY_TRACK_SET: Set<string> = new Set(["live_chat"]); diff --git a/common/components/urlState.ts b/common/components/urlState.ts @@ -5,7 +5,10 @@ import { useMemo, useSyncExternalStore } from "react"; // Which corpus the legacy single-input search targets. "posts" is the social // corpus (common/lib/posts.ts); the composite query tree can mix all of them // freely, this only decides what a bare `?q=` URL means. -export type SearchMode = "transcripts" | "subs" | "posts"; +// Defined in `lib/searchQuery.ts`, which interprets it. Re-exported here so +// every existing `from "./urlState"` import site is unchanged. +export type { SearchMode } from "../lib/searchQuery"; +import type { SearchMode } from "../lib/searchQuery"; // Per-video modal content mode. Independent from the search page's `mode` so // the modal can be toggled without disturbing search state. Absence on the diff --git a/common/lib/aiHandoff.test.ts b/common/lib/aiHandoff.test.ts @@ -1,7 +1,7 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import { buildSearchHandoff, type HandoffGroup } from "./aiHandoff"; -import type { LayerHit } from "../components/searchPipeline"; +import type { LayerHit } from "./search/leafPipeline"; function hit(start: number, text: string): LayerHit { return { leafId: "l1", scope: "transcripts", start, text }; diff --git a/common/lib/aiHandoff.ts b/common/lib/aiHandoff.ts @@ -1,4 +1,4 @@ -import type { LayerHit } from "../components/searchPipeline"; +import type { LayerHit } from "./search/leafPipeline"; // Search grounding: a completed transcript search's results, serialized so the // /ask chat can answer *grounded in exactly those results* instead of running diff --git a/common/lib/searchEval.ts b/common/lib/searchEval.ts @@ -1,11 +1,24 @@ -// Composite-search tree orchestrator. +// Composite-search tree orchestrator — the SLUG-SET half of one search +// pipeline. // // Walks a QueryNode tree (from `searchQuery.ts`), kicks off per-leaf -// pipelines (from `searchPipeline.ts:runLeafPipeline`), runs metadata-scope -// leaves synchronously against summaries, memoizes per-leaf results in -// `searchLayerCache.ts`, and emits a streaming TreeProgress that the UI +// pipelines, runs metadata-scope leaves synchronously against summaries, +// memoizes per-leaf results, and emits a streaming TreeProgress that the UI // renders. // +// Its sibling is `lib/search/evalTree.ts`, which evaluates the SAME algebra +// per RECORD, for a caller that already holds the record (the MCP server). The +// two are not copies: one answers "which slugs survive", streaming, without +// the corpus in memory; the other "does this record match". When the meaning of +// AND / OR / negate changes it changes in both, and each file says so. +// +// THE FETCH AND THE MEMO ARE INJECTED (`SearchRuntime`). They were imported +// from `components/searchPipeline` and `components/searchLayerCache`, which +// made the model layer depend on the view layer — the back-edge this slice +// deletes. `components/searchPipeline.ts` now supplies both as +// `searchRuntime`; a caller with different fetches (a test, a node-side sweep) +// supplies its own. +// // Evaluation rules (see plan): // AND group → narrow scope left-to-right between children // OR group → run children in parallel against the group's scope, union @@ -24,20 +37,37 @@ import { type LeafNode, type QueryNode, } from "./searchQuery"; -import { - cacheKey, - getCached, - getCachedSync, - putCached, - type CachedResult, -} from "../components/searchLayerCache"; -import { - runLeafPipeline, - type LayerHit, - type LeafController, -} from "../components/searchPipeline"; +import type { + LayerHit, + LeafController, + LeafRunner, +} from "./search/leafPipeline"; import type { DisplaySummary } from "./transcripts"; +// One leaf's memoized result: the slugs it matched and the hits that prove it. +// Keyed by `${canonicalHash(node)}__${scopeHash}` — same node + same input +// scope = same result, wherever in the tree it sits. +export type CachedResult = { + slugs: ReadonlySet<string>; + hits: ReadonlyMap<string, LayerHit[]>; +}; + +// The memo, as a dependency. `components/searchLayerCache.ts` implements it +// over an in-memory Map plus IndexedDB; a caller with no browser backs it with +// a plain Map, or with four no-ops for a run that should not memoize at all. +export type LayerCache = { + key(nodeHash: string, scopeHash: string): string; + getSync(key: string): CachedResult | null; + get(key: string): Promise<CachedResult | null>; + put(key: string, result: CachedResult): void; +}; + +// Everything `runQueryTree` needs from the layer below it. +export type SearchRuntime = { + runLeaf: LeafRunner; + cache: LayerCache; +}; + export type LeafState = { slugCount: number; totalHits: number; @@ -73,6 +103,7 @@ export type TreeController = { type MetadataLeafScope = "transcripts" | "chat" | "metadata"; type EvalCtx = { + runtime: SearchRuntime; initialHitLimit: number; concurrency: number; flushIntervalMs: number; @@ -122,6 +153,8 @@ function buildMetadataIndex(summaries: DisplaySummary[]): MetadataIndex { export function runQueryTree(opts: { root: GroupNode; + // The fetch-bound leaf runner + layer memo. See SearchRuntime above. + runtime: SearchRuntime; globalScope: string[]; summaries: DisplaySummary[]; chatScopeSlugs?: ReadonlySet<string> | null; @@ -133,6 +166,7 @@ export function runQueryTree(opts: { }): TreeController { const { root, + runtime, globalScope, summaries, chatScopeSlugs = null, @@ -149,6 +183,7 @@ export function runQueryTree(opts: { let allDone = false; const ctx: EvalCtx = { + runtime, initialHitLimit, concurrency, flushIntervalMs, @@ -311,12 +346,12 @@ async function runLeaf( // Fully-network leaves (transcripts / chat). Try the layer cache first. const scopeArr = Array.from(effectiveScope); const scopeHash = hashSlugs(scopeArr); - const key = cacheKey(canonicalHash(leaf), scopeHash); - const cachedSync = getCachedSync(key); + const key = ctx.runtime.cache.key(canonicalHash(leaf), scopeHash); + const cachedSync = ctx.runtime.cache.getSync(key); if (cachedSync) { return applyCached(leaf, parentScope, cachedSync, ctx, /*cached*/ true); } - const cached = await getCached(key); + const cached = await ctx.runtime.cache.get(key); if (ctx.cancelled) return new Set(); if (cached) { return applyCached(leaf, parentScope, cached, ctx, /*cached*/ true); @@ -338,7 +373,7 @@ async function runLeaf( }); let lastCapped = false; - const controller = runLeafPipeline({ + const controller = ctx.runtime.runLeaf({ leaf, scopeSlugs: scopeArr, initialHitLimit, @@ -371,7 +406,7 @@ async function runLeaf( // not cancelled). A capped result is partial — caching it would prevent a // later "show more" from extending into the rest of the scope. if (!lastCapped) { - putCached(key, { + ctx.runtime.cache.put(key, { slugs: new Set(result.slugs), hits: new Map(result.hits), }); diff --git a/common/lib/searchQuery.ts b/common/lib/searchQuery.ts @@ -9,7 +9,11 @@ // and IndexedDB-backed layer cache keys. Keep field names short — they end // up in URLs. -import type { SearchMode } from "../components/urlState"; +// The three top-level search modes a share link can carry. Defined here rather +// than in `components/urlState.ts` (which re-exports it) because +// `buildSearchRoot` below turns one into a query tree — the model owns the +// vocabulary it interprets. +export type SearchMode = "transcripts" | "subs" | "posts"; export type LayerScope = | "transcripts" diff --git a/export/app/lib/askRetrieval.ts b/export/app/lib/askRetrieval.ts @@ -12,7 +12,10 @@ import { tokenizeQuery, type SearchAlias, } from "yt-dlp-transcript-common/lib/searchAliases"; -import type { LayerHit } from "yt-dlp-transcript-common/components/searchPipeline"; +import { + searchRuntime, + type LayerHit, +} from "yt-dlp-transcript-common/components/searchPipeline"; import { peekPost } from "yt-dlp-transcript-common/components/postsCache"; import type { DisplaySummary } from "yt-dlp-transcript-common/lib/transcripts"; import { formatDuration } from "yt-dlp-transcript-common/lib/format"; @@ -348,6 +351,7 @@ export function retrieve( const postSlugs = opts.postScopeSlugs ?? null; const controller = runQueryTree({ root, + runtime: searchRuntime, globalScope: [ ...summaries.map((s) => s.slug), ...(postSlugs ? Array.from(postSlugs) : []),