commit a6533a81bb2e33f9c90fa6e7b7ee04b4ae295c2e
parent 001b3cf6359702c398faa8dfccbba47c7bef3275
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 12 Sep 2026 02:53:48 -0400
common: lib/ no longer imports components/, and the guard says so
The deliverable of this slice, not a test that was bent to pass. Four
back-edges existed; all four are gone and `"components"` joins
`FORBIDDEN.lib` in `architecture.test.ts` with NOTHING added to the
allow-list. The list can only shrink, so it did not grow.
searchEval -> components/searchPipeline the leaf pipeline moved DOWN
(commit 1) and is now INJECTED: `runQueryTree` takes a
`SearchRuntime` — a fetch-bound leaf runner plus a layer memo —
instead of importing one. `components/searchPipeline.ts` is 904
lines lighter and supplies `searchRuntime`; the three call sites
(SearchSessionContext, useSearchSeries, export's askRetrieval)
pass it.
searchEval -> components/searchLayerCache the same injection. The
memo's `CachedResult` belongs to the tree that defines it, so it
moved to searchEval and searchLayerCache re-exports it: an
IndexedDB store does not get to define what a leaf result IS.
aiHandoff -> components/searchPipeline `LayerHit` is the shared
pipeline's type now; the import path is the only change.
searchQuery -> components/urlState `SearchMode` moved to the
module that INTERPRETS it (`buildSearchRoot` turns one into a
query tree). `urlState` re-exports it, so its two importers are
untouched.
No import site outside these files changed: `components/searchPipeline`
still exports every name it did, and `export/app/lib/askRetrieval.ts`
keeps importing `LayerHit` from it.
tsc clean in all six packages; common 1128 passed (1077 + 51 new, none
lost), architecture test green; export `next build` compiled
successfully — which is the only real test that none of this dragged
`reader-fs.ts` or a react-query module into a place it does not belong.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
11 files changed, 152 insertions(+), 879 deletions(-)
diff --git a/common/architecture.test.ts b/common/architecture.test.ts
@@ -29,7 +29,13 @@ const HERE = path.dirname(fileURLToPath(import.meta.url));
// Which directory may not import which. Read as "lib/ may not import
// controller/ or jobs/".
const FORBIDDEN: Record<string, readonly string[]> = {
- lib: ["controller", "jobs"],
+ // `components` joined this row in one-core phase 2 slice S3. The model layer
+ // importing the view layer was four edges: searchEval reaching for the leaf
+ // pipeline and the IndexedDB layer memo, aiHandoff for a hit type, and
+ // searchQuery for a mode union. All four inverted — the pipeline and the two
+ // types moved down, and the fetch and the memo are now injected — so the
+ // guard turns on with nothing added to the allow-list to pay for it.
+ lib: ["controller", "jobs", "components"],
jobs: ["controller"],
components: ["controller", "jobs", "ytdlp"],
};
@@ -156,7 +162,8 @@ test("no new back-edges between common's layers", async () => {
assert.deepEqual(
unexpected,
[],
- `NEW back-edge(s) in common/. lib/ may not import controller/ or jobs/; ` +
+ `NEW back-edge(s) in common/. lib/ may not import controller/, jobs/ or ` +
+ `components/; ` +
`jobs/ may not import controller/; components/ may not import ` +
`controller/, jobs/ or ytdlp/. Move the type or the function down a ` +
`layer instead of adding it to ALLOWED. Found: ${unexpected.join(", ")}`,
diff --git a/common/components/SearchSessionContext.tsx b/common/components/SearchSessionContext.tsx
@@ -26,7 +26,7 @@ import { usePlayer } from "./PlayerProvider";
import { useChannelSubsManifests } from "./subsCache";
import { peekPost, useChannelPostsManifests } from "./postsCache";
import { useSearchData } from "./SearchDataContext";
-import type { LayerHit } from "./searchPipeline";
+import { searchRuntime, type LayerHit } from "./searchPipeline";
import {
runQueryTree,
type GroupState,
@@ -760,6 +760,7 @@ function useSearchSessionState() {
const controller = runQueryTree({
root: committedRoot,
+ runtime: searchRuntime,
globalScope: globalScopeSlugs,
summaries: transcripts,
chatScopeSlugs: needsChatManifests ? chatScopeSlugs : null,
diff --git a/common/components/charts/useSearchSeries.ts b/common/components/charts/useSearchSeries.ts
@@ -8,6 +8,7 @@ import { applyFilters, type ChartData } from "../../lib/chartAggregate";
import { seriesFromSlugs } from "../../lib/chartSeriesFromSlugs";
import { parseRoot } from "../../lib/searchQuery";
import { runQueryTree, type TreeProgress } from "../../lib/searchEval";
+import { searchRuntime } from "../searchPipeline";
export type SearchSeriesState = {
data: ChartData;
@@ -56,6 +57,7 @@ export function useSearchSeries(
try {
const controller = runQueryTree({
root,
+ runtime: searchRuntime,
globalScope: scopeSlugs,
summaries,
initialHitLimit: 50000,
diff --git a/common/components/searchLayerCache.ts b/common/components/searchLayerCache.ts
@@ -12,7 +12,13 @@
// IndexedDB writes are batched via queueMicrotask the same way as
// transcriptStore.ts.
-import type { LayerHit } from "./searchPipeline";
+import type { LayerHit } from "../lib/search/leafPipeline";
+import type { CachedResult } from "../lib/searchEval";
+
+// Re-exported so this module's importers keep their import site. The type
+// belongs to `lib/searchEval.ts` now: it is the shape of a memoized leaf
+// result, and the tree — not its storage — defines it.
+export type { CachedResult };
// Separate DB from `transcriptStore.ts` (`yt-dlp-transcript-browser`) so the
// two stores can evolve their schemas independently — bumping the version
@@ -22,11 +28,6 @@ const DB_VERSION = 1;
const STORE = "layers";
const MAX_ENTRIES = 500;
-export type CachedResult = {
- slugs: ReadonlySet<string>;
- hits: ReadonlyMap<string, LayerHit[]>;
-};
-
type StoredEntry = {
key: string;
slugs: string[];
diff --git a/common/components/searchPipeline.ts b/common/components/searchPipeline.ts
@@ -1,852 +1,68 @@
+// THE BINDING between the shared search pipeline and this app's caches.
+//
+// Everything that used to live here — the three worker-pool drivers, the leaf
+// adapter, the cue/text matchers — is `lib/search/leafPipeline.ts` now, where
+// `lib/searchEval.ts` can reach it without `lib/` importing `components/`.
+// What is left is the one thing that genuinely belongs to the view layer: WHICH
+// FETCH. `fetchTranscript` / `fetchSubs` / `fetchPost` are the react-query-backed
+// page caches, and they are supplied here rather than imported there.
+//
+// Every name this module exported before is still exported from it, so the five
+// files that import `LayerHit` from `components/searchPipeline` (including
+// `export/app/lib/askRetrieval.ts`) are untouched.
+
import { fetchTranscript } from "./transcriptCache";
import { fetchSubs } from "./subsCache";
import { fetchPost } from "./postsCache";
-import type { DisplaySummary } from "../lib/transcripts";
-import type { LayerScope, LeafNode } from "../lib/searchQuery";
-
-export type Hit = { start: number; text: string };
-
-export type SubsHit = { track: string; start: number; text: string };
-
-// Per-leaf hit shape consumed by the composite-search tree (searchEval.ts).
-// Carries enough info for the result-list UI to render the hit with its
-// originating layer's swatch + scope-specific decorations.
-export type LayerHit = {
- leafId: string;
- scope: LayerScope;
- track?: string;
- start: number;
- text: string;
-};
-
-export type PipelineUpdate = {
- hitsBySlug: Record<string, Hit[]>;
- totalHits: number;
- processed: number;
- totalToProcess: number;
- capped: boolean;
- done: boolean;
-};
-
-type PipelineConfig = {
- slugs: string[];
- query: string;
- useRegex: boolean;
- regex: RegExp | null;
- initialHitLimit: number;
- concurrency: number;
- flushIntervalMs: number;
- emit: (update: PipelineUpdate) => void;
- // Which field of the fetched transcript document to match. "cues" (default)
- // is the transcripts scope; "description"/"tags" match those metadata fields.
- // All reuse the same per-channel transcript-page fetch + worker pool.
- matchField?: "cues" | "description" | "tags";
+import {
+ runLeafPipeline as runLeafPipelineWith,
+ type LeafController,
+ type LeafFetchers,
+ type LeafPipelineOptions,
+} from "../lib/search/leafPipeline";
+import {
+ cacheKey,
+ getCached,
+ getCachedSync,
+ putCached,
+} from "./searchLayerCache";
+import type { SearchRuntime } from "../lib/searchEval";
+
+export type {
+ Hit,
+ SubsHit,
+ LayerHit,
+ PipelineUpdate,
+ PipelineController,
+ SubsPipelineUpdate,
+ LeafProgress,
+ LeafResult,
+ LeafController,
+} from "../lib/search/leafPipeline";
+export { filterSlugs } from "../lib/search/leafPipeline";
+
+// This app's three fetches, as one injectable bundle.
+const CACHE_FETCHERS: LeafFetchers = {
+ transcript: fetchTranscript,
+ subs: fetchSubs,
+ post: fetchPost,
};
-export type PipelineController = {
- cancel(): void;
- setHitLimit(limit: number): void;
-};
-
-// Runs a streaming search over the given slugs. State is closure-captured so
-// `setHitLimit` can raise the cap and re-spawn workers without restarting
-// traversal — workers resume from the shared `idx` cursor.
-export function createSearchPipeline(
- config: PipelineConfig,
-): PipelineController {
- const {
- slugs,
- query,
- useRegex,
- regex,
- initialHitLimit,
- concurrency,
- flushIntervalMs,
- emit,
- matchField = "cues",
- } = config;
-
- let cancelled = false;
- // `idx` is the next-to-claim cursor (leading edge); `completed` counts
- // slugs that workers have finished scanning (trailing edge). We report
- // `completed` to the UI so progress reflects actual work done, not
- // in-flight claims — and it never exceeds slugs.length.
- let idx = 0;
- let completed = 0;
- let totalSoFar = 0;
- let hitLimit = initialHitLimit;
- let activeWorkers = 0;
- let done = false;
- const localHits: Record<string, Hit[]> = {};
- let flushTimer: number | null = null;
-
- const pushUpdate = (overrides: Partial<PipelineUpdate> = {}) => {
- emit({
- hitsBySlug: { ...localHits },
- totalHits: totalSoFar,
- processed: completed,
- totalToProcess: slugs.length,
- // `capped` keys off `idx` (claimed), not `completed` — raising the cap
- // only helps if there are still unclaimed slugs a worker can pick up.
- capped: totalSoFar >= hitLimit && idx < slugs.length,
- done,
- ...overrides,
- });
- };
-
- const scheduleFlush = () => {
- if (flushTimer !== null || cancelled) return;
- flushTimer = window.setTimeout(() => {
- flushTimer = null;
- if (cancelled) return;
- pushUpdate();
- }, flushIntervalMs);
- };
-
- const finalize = () => {
- if (done || cancelled) return;
- done = true;
- if (flushTimer !== null) {
- window.clearTimeout(flushTimer);
- flushTimer = null;
- }
- pushUpdate();
- };
-
- const worker = async () => {
- activeWorkers++;
- try {
- while (!cancelled) {
- if (totalSoFar >= hitLimit) return;
- if (idx >= slugs.length) return;
- const my = idx++;
- const slug = slugs[my];
- try {
- const full = await fetchTranscript(slug);
- if (cancelled) return;
- if (totalSoFar < hitLimit) {
- const remaining = hitLimit - totalSoFar;
- const hits =
- matchField === "description"
- ? findHitsInText(full.description ?? "", query, useRegex, regex)
- : matchField === "tags"
- ? findHitsInText(
- (full.tags ?? []).join(", "),
- query,
- useRegex,
- regex,
- )
- : full.cues
- ? findHitsInCues(full.cues, query, useRegex, regex, remaining)
- : [];
- if (hits.length > 0) {
- localHits[slug] = hits;
- totalSoFar += hits.length;
- }
- }
- } catch {
- // ignore per-transcript failures
- }
- completed++;
- scheduleFlush();
- }
- } finally {
- activeWorkers--;
- if (activeWorkers === 0 && !cancelled) {
- // Either we hit the cap or ran out of work — either way, settle.
- finalize();
- }
- }
- };
-
- const ensureWorkers = () => {
- if (cancelled || done) return;
- if (totalSoFar >= hitLimit) return;
- if (idx >= slugs.length) return;
- const needed = Math.min(
- concurrency - activeWorkers,
- slugs.length - idx,
- );
- for (let i = 0; i < needed; i++) worker();
- };
-
- // Emit initial snapshot synchronously so the UI clears previous results.
- pushUpdate();
- ensureWorkers();
-
- return {
- cancel() {
- cancelled = true;
- if (flushTimer !== null) {
- window.clearTimeout(flushTimer);
- flushTimer = null;
- }
- },
- setHitLimit(limit: number) {
- if (cancelled) return;
- if (limit <= hitLimit) return;
- hitLimit = limit;
- // The prior run may have finalized because we were capped. Resume.
- if (done) {
- done = false;
- pushUpdate();
- }
- ensureWorkers();
- },
- };
-}
-
-function findHitsInCues(
- cues: { start: number; text: string }[],
- query: string,
- useRegex: boolean,
- regex: RegExp | null,
- limit: number,
-): Hit[] {
- const hits: Hit[] = [];
- for (let i = 0; i < cues.length && hits.length < limit; i++) {
- const cur = cues[i];
- const prevText = i > 0 ? cues[i - 1].text : "";
- const nextText = i < cues.length - 1 ? cues[i + 1].text : "";
- const sep1 = prevText ? " " : "";
- const sep2 = nextText ? " " : "";
- const windowText = prevText + sep1 + cur.text + sep2 + nextText;
- const curStart = prevText.length + sep1.length;
- const curEnd = curStart + cur.text.length;
- const m = findFirstMatchInRange(
- windowText,
- curStart,
- curEnd,
- query,
- useRegex,
- regex,
- );
- if (!m) continue;
- const crosses = m.idx < curStart || m.idx + m.length > curEnd;
- hits.push({
- start: Math.round(cur.start),
- text: crosses ? windowText : cur.text,
- });
- }
- return hits;
-}
-
-// Single-document text match (description scope): emit at most one hit with a
-// snippet window around the first match, so the result row shows context.
-function findHitsInText(
- text: string,
- query: string,
- useRegex: boolean,
- regex: RegExp | null,
-): Hit[] {
- if (!text) return [];
- const m = findFirstMatchInRange(text, 0, text.length, query, useRegex, regex);
- if (!m) return [];
- const PAD = 80;
- const from = Math.max(0, m.idx - PAD);
- const to = Math.min(text.length, m.idx + m.length + PAD);
- const snippet =
- (from > 0 ? "…" : "") +
- text.slice(from, to).replace(/\s+/g, " ").trim() +
- (to < text.length ? "…" : "");
- return [{ start: 0, text: snippet }];
-}
-
-function findFirstMatchInRange(
- haystack: string,
- rangeStart: number,
- rangeEnd: number,
- query: string,
- useRegex: boolean,
- regex: RegExp | null,
-): { idx: number; length: number } | null {
- if (useRegex) {
- if (!regex) return null;
- const flags = regex.flags.includes("g") ? regex.flags : regex.flags + "g";
- const re = new RegExp(regex.source, flags);
- let m: RegExpExecArray | null;
- while ((m = re.exec(haystack)) !== null) {
- if (m.index >= rangeStart && m.index < rangeEnd) {
- return { idx: m.index, length: m[0].length };
- }
- if (m.index >= rangeEnd) return null;
- if (m[0].length === 0) re.lastIndex++;
- }
- return null;
- }
- if (!query) return null;
- const lower = haystack.toLowerCase();
- const ql = query.toLowerCase();
- let from = 0;
- while (from <= haystack.length) {
- const idx = lower.indexOf(ql, from);
- if (idx === -1) return null;
- if (idx >= rangeStart && idx < rangeEnd) return { idx, length: ql.length };
- if (idx >= rangeEnd) return null;
- from = idx + 1;
- }
- return null;
-}
-
-// Streaming search over the social-post corpus. Structurally the transcripts
-// pipeline with a different fetch + match: the unit of iteration is a POST
-// slug (`<channelSlug>/<postId>`), and fetchPost() resolves through the page
-// cache, so the first post on a page warms every other post on it.
-//
-// Post bodies are short but unbounded, and unlike the cue path there is no
-// natural per-line unit to clip to — so hits are truncated to a ±80-char
-// window (the same one the description/tags scopes use) rather than shipping
-// the whole body into a result card.
-export function createPostsSearchPipeline(
- config: Omit<PipelineConfig, "matchField">,
-): PipelineController {
- const {
- slugs,
- query,
- useRegex,
- regex,
- initialHitLimit,
- concurrency,
- flushIntervalMs,
- emit,
- } = config;
-
- let cancelled = false;
- let idx = 0;
- let completed = 0;
- let totalSoFar = 0;
- let hitLimit = initialHitLimit;
- let activeWorkers = 0;
- let done = false;
- const localHits: Record<string, Hit[]> = {};
- let flushTimer: number | null = null;
-
- const pushUpdate = (overrides: Partial<PipelineUpdate> = {}) => {
- emit({
- hitsBySlug: { ...localHits },
- totalHits: totalSoFar,
- processed: completed,
- totalToProcess: slugs.length,
- capped: totalSoFar >= hitLimit && idx < slugs.length,
- done,
- ...overrides,
- });
- };
-
- const scheduleFlush = () => {
- if (flushTimer !== null || cancelled) return;
- flushTimer = window.setTimeout(() => {
- flushTimer = null;
- if (cancelled) return;
- pushUpdate();
- }, flushIntervalMs);
- };
-
- const finalize = () => {
- if (done || cancelled) return;
- done = true;
- if (flushTimer !== null) {
- window.clearTimeout(flushTimer);
- flushTimer = null;
- }
- pushUpdate();
- };
-
- const worker = async () => {
- activeWorkers++;
- try {
- while (!cancelled) {
- if (totalSoFar >= hitLimit) return;
- if (idx >= slugs.length) return;
- const my = idx++;
- const slug = slugs[my];
- try {
- const post = await fetchPost(slug);
- if (cancelled) return;
- if (totalSoFar < hitLimit) {
- const hits = findHitsInText(post.text, query, useRegex, regex);
- if (hits.length > 0) {
- localHits[slug] = hits;
- totalSoFar += hits.length;
- }
- }
- } catch {
- // ignore per-post failures
- }
- completed++;
- scheduleFlush();
- }
- } finally {
- activeWorkers--;
- if (activeWorkers === 0 && !cancelled) finalize();
- }
- };
-
- const ensureWorkers = () => {
- if (cancelled || done) return;
- if (totalSoFar >= hitLimit) return;
- if (idx >= slugs.length) return;
- const needed = Math.min(concurrency - activeWorkers, slugs.length - idx);
- for (let i = 0; i < needed; i++) worker();
- };
-
- pushUpdate();
- ensureWorkers();
-
- return {
- cancel() {
- cancelled = true;
- if (flushTimer !== null) {
- window.clearTimeout(flushTimer);
- flushTimer = null;
- }
- },
- setHitLimit(limit: number) {
- if (cancelled) return;
- if (limit <= hitLimit) return;
- hitLimit = limit;
- if (done) {
- done = false;
- pushUpdate();
- }
- ensureWorkers();
- },
- };
+export function runLeafPipeline(
+ opts: Omit<LeafPipelineOptions, "fetchers">,
+): LeafController {
+ return runLeafPipelineWith({ ...opts, fetchers: CACHE_FETCHERS });
}
-// Helper to build slug list from summaries + filter predicate.
-export function filterSlugs(
- summaries: DisplaySummary[],
- passes: (t: DisplaySummary) => boolean,
-): string[] {
- const slugs: string[] = [];
- for (const t of summaries) if (passes(t)) slugs.push(t.slug);
- return slugs;
-}
-
-export type SubsPipelineUpdate = {
- hitsBySlug: Record<string, SubsHit[]>;
- totalHits: number;
- processed: number;
- totalToProcess: number;
- capped: boolean;
- done: boolean;
-};
-
-type SubsPipelineConfig = {
- slugs: string[];
- query: string;
- useRegex: boolean;
- regex: RegExp | null;
- // Tracks to exclude. Empty set means "search all tracks".
- excludedTracks: Set<string>;
- // Optional inclusion filter. When provided, ONLY tracks in this set are
- // scanned (and `excludedTracks` is irrelevant). Used by the composite-
- // search "chat" scope to limit scanning to live_chat regardless of which
- // other tracks a video has.
- includedTracks?: Set<string> | null;
- initialHitLimit: number;
- concurrency: number;
- flushIntervalMs: number;
- emit: (update: SubsPipelineUpdate) => void;
-};
-
-export function createSubsSearchPipeline(
- config: SubsPipelineConfig,
-): PipelineController {
- const {
- slugs,
- query,
- useRegex,
- regex,
- excludedTracks,
- includedTracks,
- initialHitLimit,
- concurrency,
- flushIntervalMs,
- emit,
- } = config;
-
- let cancelled = false;
- let idx = 0;
- let completed = 0;
- let totalSoFar = 0;
- let hitLimit = initialHitLimit;
- let activeWorkers = 0;
- let done = false;
- const localHits: Record<string, SubsHit[]> = {};
- let flushTimer: number | null = null;
-
- const pushUpdate = (overrides: Partial<SubsPipelineUpdate> = {}) => {
- emit({
- hitsBySlug: { ...localHits },
- totalHits: totalSoFar,
- processed: completed,
- totalToProcess: slugs.length,
- capped: totalSoFar >= hitLimit && idx < slugs.length,
- done,
- ...overrides,
- });
- };
-
- const scheduleFlush = () => {
- if (flushTimer !== null || cancelled) return;
- flushTimer = window.setTimeout(() => {
- flushTimer = null;
- if (cancelled) return;
- pushUpdate();
- }, flushIntervalMs);
- };
-
- const finalize = () => {
- if (done || cancelled) return;
- done = true;
- if (flushTimer !== null) {
- window.clearTimeout(flushTimer);
- flushTimer = null;
- }
- pushUpdate();
- };
-
- const worker = async () => {
- activeWorkers++;
- try {
- while (!cancelled) {
- if (totalSoFar >= hitLimit) return;
- if (idx >= slugs.length) return;
- const my = idx++;
- const slug = slugs[my];
- try {
- const detail = await fetchSubs(slug);
- if (cancelled) return;
- if (totalSoFar < hitLimit) {
- const slugHits: SubsHit[] = [];
- for (const [track, cueList] of Object.entries(detail.tracks)) {
- if (includedTracks) {
- if (!includedTracks.has(track)) continue;
- } else if (excludedTracks.has(track)) continue;
- if (totalSoFar + slugHits.length >= hitLimit) break;
- const remaining =
- hitLimit - totalSoFar - slugHits.length;
- const trackHits = findHitsInCues(
- cueList,
- query,
- useRegex,
- regex,
- remaining,
- );
- for (const h of trackHits) {
- slugHits.push({ track, start: h.start, text: h.text });
- }
- }
- if (slugHits.length > 0) {
- // Order hits by start time so live_chat and language tracks
- // interleave naturally instead of being grouped per-track.
- slugHits.sort((a, b) => a.start - b.start);
- localHits[slug] = slugHits;
- totalSoFar += slugHits.length;
- }
- }
- } catch {
- // ignore per-video failures
- }
- completed++;
- scheduleFlush();
- }
- } finally {
- activeWorkers--;
- if (activeWorkers === 0 && !cancelled) finalize();
- }
- };
-
- const ensureWorkers = () => {
- if (cancelled || done) return;
- if (totalSoFar >= hitLimit) return;
- if (idx >= slugs.length) return;
- const needed = Math.min(
- concurrency - activeWorkers,
- slugs.length - idx,
- );
- for (let i = 0; i < needed; i++) worker();
- };
-
- pushUpdate();
- ensureWorkers();
-
- return {
- cancel() {
- cancelled = true;
- if (flushTimer !== null) {
- window.clearTimeout(flushTimer);
- flushTimer = null;
- }
- },
- setHitLimit(limit: number) {
- if (cancelled) return;
- if (limit <= hitLimit) return;
- hitLimit = limit;
- if (done) {
- done = false;
- pushUpdate();
- }
- ensureWorkers();
- },
- };
-}
-
-// ─── Leaf pipeline wrapper ───
-// Thin adapter over `createSearchPipeline` / `createSubsSearchPipeline` for
-// the composite-search tree (`searchEval.ts`). One controller per leaf in the
-// tree. Returns a `LeafController` that the tree orchestrator can cancel /
-// resize, plus a `done` promise that resolves with the final LeafResult.
-//
-// Scope=metadata isn't handled here — searchEval.ts evaluates it synchronously
-// over the summaries cache. This wrapper only deals with the network-backed
-// scopes (transcripts, chat) where the existing worker pool earns its keep.
-
-export type LeafProgress = {
- slugs: Set<string>;
- hits: Map<string, LayerHit[]>;
- totalHits: number;
- processed: number;
- totalToProcess: number;
- capped: boolean;
- done: boolean;
+// What `runQueryTree` is handed: the fetch-bound leaf runner plus the
+// IndexedDB-backed layer memo. `lib/searchEval.ts` owns the tree algebra and
+// knows nothing about either.
+export const searchRuntime: SearchRuntime = {
+ runLeaf: runLeafPipeline,
+ cache: {
+ key: cacheKey,
+ getSync: getCachedSync,
+ get: getCached,
+ put: putCached,
+ },
};
-
-export type LeafResult = {
- slugs: Set<string>;
- hits: Map<string, LayerHit[]>;
-};
-
-export type LeafController = PipelineController & {
- done: Promise<LeafResult>;
-};
-
-export function runLeafPipeline(opts: {
- leaf: LeafNode;
- scopeSlugs: string[];
- initialHitLimit: number;
- concurrency: number;
- flushIntervalMs: number;
- emit: (p: LeafProgress) => void;
-}): LeafController {
- const {
- leaf,
- scopeSlugs,
- initialHitLimit,
- concurrency,
- flushIntervalMs,
- emit,
- } = opts;
-
- const trimmed = leaf.query.trim();
- // Caller guarantees query is non-empty before invoking us (see
- // isLeafActive). Stay defensive: empty queries produce an immediate done
- // with no hits.
- if (!trimmed) {
- const empty: LeafResult = { slugs: new Set(), hits: new Map() };
- queueMicrotask(() => {
- emit({
- slugs: empty.slugs,
- hits: empty.hits,
- totalHits: 0,
- processed: 0,
- totalToProcess: 0,
- capped: false,
- done: true,
- });
- });
- return {
- cancel() {
- /* no-op */
- },
- setHitLimit() {
- /* no-op */
- },
- done: Promise.resolve(empty),
- };
- }
-
- const regex = compileLeafRegex(leaf);
-
- let resolveDone!: (r: LeafResult) => void;
- const donePromise = new Promise<LeafResult>((resolve) => {
- resolveDone = resolve;
- });
-
- // Latest seen progress, kept locally so `done` settles with the final
- // payload that the consumer already saw on the last emit.
- let finalSlugs = new Set<string>();
- let finalHits = new Map<string, LayerHit[]>();
- let settled = false;
-
- const adaptTranscriptUpdate = (u: PipelineUpdate): LeafProgress => {
- const slugs = new Set<string>();
- const hits = new Map<string, LayerHit[]>();
- for (const [slug, list] of Object.entries(u.hitsBySlug)) {
- if (!list || list.length === 0) continue;
- slugs.add(slug);
- if (leaf.contributeHits) {
- hits.set(
- slug,
- list.map((h) => ({
- leafId: leaf.id,
- scope: leaf.scope,
- start: h.start,
- text: h.text,
- })),
- );
- }
- }
- return {
- slugs,
- hits,
- totalHits: u.totalHits,
- processed: u.processed,
- totalToProcess: u.totalToProcess,
- capped: u.capped,
- done: u.done,
- };
- };
-
- const adaptSubsUpdate = (u: SubsPipelineUpdate): LeafProgress => {
- const slugs = new Set<string>();
- const hits = new Map<string, LayerHit[]>();
- for (const [slug, list] of Object.entries(u.hitsBySlug)) {
- if (!list || list.length === 0) continue;
- slugs.add(slug);
- if (leaf.contributeHits) {
- hits.set(
- slug,
- list.map((h) => ({
- leafId: leaf.id,
- scope: "chat" as const,
- track: h.track,
- start: h.start,
- text: h.text,
- })),
- );
- }
- }
- return {
- slugs,
- hits,
- totalHits: u.totalHits,
- processed: u.processed,
- totalToProcess: u.totalToProcess,
- capped: u.capped,
- done: u.done,
- };
- };
-
- const onProgress = (p: LeafProgress) => {
- finalSlugs = p.slugs;
- finalHits = p.hits;
- emit(p);
- if (p.done && !settled) {
- settled = true;
- resolveDone({ slugs: finalSlugs, hits: finalHits });
- }
- };
-
- let controller: PipelineController;
- if (
- leaf.scope === "transcripts" ||
- leaf.scope === "description" ||
- leaf.scope === "tags"
- ) {
- controller = createSearchPipeline({
- slugs: scopeSlugs,
- query: trimmed,
- useRegex: leaf.useRegex,
- regex,
- initialHitLimit,
- concurrency,
- flushIntervalMs,
- matchField:
- leaf.scope === "description"
- ? "description"
- : leaf.scope === "tags"
- ? "tags"
- : "cues",
- emit: (u) => onProgress(adaptTranscriptUpdate(u)),
- });
- } else if (leaf.scope === "posts") {
- controller = createPostsSearchPipeline({
- slugs: scopeSlugs,
- query: trimmed,
- useRegex: leaf.useRegex,
- regex,
- initialHitLimit,
- concurrency,
- flushIntervalMs,
- // A post has no timeline, so every hit is `start: 0` — the same
- // convention description/tags/metadata hits already use.
- emit: (u) => onProgress(adaptTranscriptUpdate(u)),
- });
- } else if (leaf.scope === "chat") {
- controller = createSubsSearchPipeline({
- slugs: scopeSlugs,
- query: trimmed,
- useRegex: leaf.useRegex,
- regex,
- excludedTracks: EMPTY_TRACK_SET,
- includedTracks: CHAT_ONLY_TRACK_SET,
- initialHitLimit,
- concurrency,
- flushIntervalMs,
- emit: (u) => onProgress(adaptSubsUpdate(u)),
- });
- } else {
- // Metadata scope is handled by searchEval.ts directly. We shouldn't be
- // invoked here; resolve immediately as a safety net.
- const empty: LeafResult = { slugs: new Set(), hits: new Map() };
- queueMicrotask(() => {
- onProgress({
- slugs: empty.slugs,
- hits: empty.hits,
- totalHits: 0,
- processed: 0,
- totalToProcess: 0,
- capped: false,
- done: true,
- });
- });
- return {
- cancel() {
- /* no-op */
- },
- setHitLimit() {
- /* no-op */
- },
- done: donePromise,
- };
- }
-
- return {
- cancel() {
- controller.cancel();
- if (!settled) {
- settled = true;
- resolveDone({ slugs: finalSlugs, hits: finalHits });
- }
- },
- setHitLimit(limit: number) {
- controller.setHitLimit(limit);
- },
- done: donePromise,
- };
-}
-
-function compileLeafRegex(leaf: LeafNode): RegExp | null {
- if (!leaf.useRegex) return null;
- try {
- return new RegExp(leaf.query, "i");
- } catch {
- return null;
- }
-}
-
-const EMPTY_TRACK_SET: Set<string> = new Set();
-const CHAT_ONLY_TRACK_SET: Set<string> = new Set(["live_chat"]);
diff --git a/common/components/urlState.ts b/common/components/urlState.ts
@@ -5,7 +5,10 @@ import { useMemo, useSyncExternalStore } from "react";
// Which corpus the legacy single-input search targets. "posts" is the social
// corpus (common/lib/posts.ts); the composite query tree can mix all of them
// freely, this only decides what a bare `?q=` URL means.
-export type SearchMode = "transcripts" | "subs" | "posts";
+// Defined in `lib/searchQuery.ts`, which interprets it. Re-exported here so
+// every existing `from "./urlState"` import site is unchanged.
+export type { SearchMode } from "../lib/searchQuery";
+import type { SearchMode } from "../lib/searchQuery";
// Per-video modal content mode. Independent from the search page's `mode` so
// the modal can be toggled without disturbing search state. Absence on the
diff --git a/common/lib/aiHandoff.test.ts b/common/lib/aiHandoff.test.ts
@@ -1,7 +1,7 @@
import { test } from "node:test";
import assert from "node:assert/strict";
import { buildSearchHandoff, type HandoffGroup } from "./aiHandoff";
-import type { LayerHit } from "../components/searchPipeline";
+import type { LayerHit } from "./search/leafPipeline";
function hit(start: number, text: string): LayerHit {
return { leafId: "l1", scope: "transcripts", start, text };
diff --git a/common/lib/aiHandoff.ts b/common/lib/aiHandoff.ts
@@ -1,4 +1,4 @@
-import type { LayerHit } from "../components/searchPipeline";
+import type { LayerHit } from "./search/leafPipeline";
// Search grounding: a completed transcript search's results, serialized so the
// /ask chat can answer *grounded in exactly those results* instead of running
diff --git a/common/lib/searchEval.ts b/common/lib/searchEval.ts
@@ -1,11 +1,24 @@
-// Composite-search tree orchestrator.
+// Composite-search tree orchestrator — the SLUG-SET half of one search
+// pipeline.
//
// Walks a QueryNode tree (from `searchQuery.ts`), kicks off per-leaf
-// pipelines (from `searchPipeline.ts:runLeafPipeline`), runs metadata-scope
-// leaves synchronously against summaries, memoizes per-leaf results in
-// `searchLayerCache.ts`, and emits a streaming TreeProgress that the UI
+// pipelines, runs metadata-scope leaves synchronously against summaries,
+// memoizes per-leaf results, and emits a streaming TreeProgress that the UI
// renders.
//
+// Its sibling is `lib/search/evalTree.ts`, which evaluates the SAME algebra
+// per RECORD, for a caller that already holds the record (the MCP server). The
+// two are not copies: one answers "which slugs survive", streaming, without
+// the corpus in memory; the other "does this record match". When the meaning of
+// AND / OR / negate changes it changes in both, and each file says so.
+//
+// THE FETCH AND THE MEMO ARE INJECTED (`SearchRuntime`). They were imported
+// from `components/searchPipeline` and `components/searchLayerCache`, which
+// made the model layer depend on the view layer — the back-edge this slice
+// deletes. `components/searchPipeline.ts` now supplies both as
+// `searchRuntime`; a caller with different fetches (a test, a node-side sweep)
+// supplies its own.
+//
// Evaluation rules (see plan):
// AND group → narrow scope left-to-right between children
// OR group → run children in parallel against the group's scope, union
@@ -24,20 +37,37 @@ import {
type LeafNode,
type QueryNode,
} from "./searchQuery";
-import {
- cacheKey,
- getCached,
- getCachedSync,
- putCached,
- type CachedResult,
-} from "../components/searchLayerCache";
-import {
- runLeafPipeline,
- type LayerHit,
- type LeafController,
-} from "../components/searchPipeline";
+import type {
+ LayerHit,
+ LeafController,
+ LeafRunner,
+} from "./search/leafPipeline";
import type { DisplaySummary } from "./transcripts";
+// One leaf's memoized result: the slugs it matched and the hits that prove it.
+// Keyed by `${canonicalHash(node)}__${scopeHash}` — same node + same input
+// scope = same result, wherever in the tree it sits.
+export type CachedResult = {
+ slugs: ReadonlySet<string>;
+ hits: ReadonlyMap<string, LayerHit[]>;
+};
+
+// The memo, as a dependency. `components/searchLayerCache.ts` implements it
+// over an in-memory Map plus IndexedDB; a caller with no browser backs it with
+// a plain Map, or with four no-ops for a run that should not memoize at all.
+export type LayerCache = {
+ key(nodeHash: string, scopeHash: string): string;
+ getSync(key: string): CachedResult | null;
+ get(key: string): Promise<CachedResult | null>;
+ put(key: string, result: CachedResult): void;
+};
+
+// Everything `runQueryTree` needs from the layer below it.
+export type SearchRuntime = {
+ runLeaf: LeafRunner;
+ cache: LayerCache;
+};
+
export type LeafState = {
slugCount: number;
totalHits: number;
@@ -73,6 +103,7 @@ export type TreeController = {
type MetadataLeafScope = "transcripts" | "chat" | "metadata";
type EvalCtx = {
+ runtime: SearchRuntime;
initialHitLimit: number;
concurrency: number;
flushIntervalMs: number;
@@ -122,6 +153,8 @@ function buildMetadataIndex(summaries: DisplaySummary[]): MetadataIndex {
export function runQueryTree(opts: {
root: GroupNode;
+ // The fetch-bound leaf runner + layer memo. See SearchRuntime above.
+ runtime: SearchRuntime;
globalScope: string[];
summaries: DisplaySummary[];
chatScopeSlugs?: ReadonlySet<string> | null;
@@ -133,6 +166,7 @@ export function runQueryTree(opts: {
}): TreeController {
const {
root,
+ runtime,
globalScope,
summaries,
chatScopeSlugs = null,
@@ -149,6 +183,7 @@ export function runQueryTree(opts: {
let allDone = false;
const ctx: EvalCtx = {
+ runtime,
initialHitLimit,
concurrency,
flushIntervalMs,
@@ -311,12 +346,12 @@ async function runLeaf(
// Fully-network leaves (transcripts / chat). Try the layer cache first.
const scopeArr = Array.from(effectiveScope);
const scopeHash = hashSlugs(scopeArr);
- const key = cacheKey(canonicalHash(leaf), scopeHash);
- const cachedSync = getCachedSync(key);
+ const key = ctx.runtime.cache.key(canonicalHash(leaf), scopeHash);
+ const cachedSync = ctx.runtime.cache.getSync(key);
if (cachedSync) {
return applyCached(leaf, parentScope, cachedSync, ctx, /*cached*/ true);
}
- const cached = await getCached(key);
+ const cached = await ctx.runtime.cache.get(key);
if (ctx.cancelled) return new Set();
if (cached) {
return applyCached(leaf, parentScope, cached, ctx, /*cached*/ true);
@@ -338,7 +373,7 @@ async function runLeaf(
});
let lastCapped = false;
- const controller = runLeafPipeline({
+ const controller = ctx.runtime.runLeaf({
leaf,
scopeSlugs: scopeArr,
initialHitLimit,
@@ -371,7 +406,7 @@ async function runLeaf(
// not cancelled). A capped result is partial — caching it would prevent a
// later "show more" from extending into the rest of the scope.
if (!lastCapped) {
- putCached(key, {
+ ctx.runtime.cache.put(key, {
slugs: new Set(result.slugs),
hits: new Map(result.hits),
});
diff --git a/common/lib/searchQuery.ts b/common/lib/searchQuery.ts
@@ -9,7 +9,11 @@
// and IndexedDB-backed layer cache keys. Keep field names short — they end
// up in URLs.
-import type { SearchMode } from "../components/urlState";
+// The three top-level search modes a share link can carry. Defined here rather
+// than in `components/urlState.ts` (which re-exports it) because
+// `buildSearchRoot` below turns one into a query tree — the model owns the
+// vocabulary it interprets.
+export type SearchMode = "transcripts" | "subs" | "posts";
export type LayerScope =
| "transcripts"
diff --git a/export/app/lib/askRetrieval.ts b/export/app/lib/askRetrieval.ts
@@ -12,7 +12,10 @@ import {
tokenizeQuery,
type SearchAlias,
} from "yt-dlp-transcript-common/lib/searchAliases";
-import type { LayerHit } from "yt-dlp-transcript-common/components/searchPipeline";
+import {
+ searchRuntime,
+ type LayerHit,
+} from "yt-dlp-transcript-common/components/searchPipeline";
import { peekPost } from "yt-dlp-transcript-common/components/postsCache";
import type { DisplaySummary } from "yt-dlp-transcript-common/lib/transcripts";
import { formatDuration } from "yt-dlp-transcript-common/lib/format";
@@ -348,6 +351,7 @@ export function retrieve(
const postSlugs = opts.postScopeSlugs ?? null;
const controller = runQueryTree({
root,
+ runtime: searchRuntime,
globalScope: [
...summaries.map((s) => s.slug),
...(postSlugs ? Array.from(postSlugs) : []),