// ONE SEARCH PIPELINE — the query-tree evaluator and the filter predicate. // // This is the per-RECORD half of composite search: given one transcript (or // post) record and a compiled query tree, does it match, how many times, and // which excerpts prove it. It backs the MCP's `open_link` / `sweep link=` // flows and the plain scanner's single-leaf case. // // Its sibling is `lib/searchEval.ts`, which evaluates the SAME tree algebra // over SLUG SETS, streaming, because the browser cannot hold the corpus in // memory and fetches per leaf. The two are not copies of one another — one // answers "does this record match", the other "which slugs survive" — and // merging them would mean the browser materialising every record. What they // do share, and what this move makes single, is the algebra's meaning: // AND narrows, OR unions, `negate` inverts and contributes no hits, and an // inactive subtree is identity. When that changes it must change here and in // `searchEval.ts` together; each file points at the other. // // `passesFilters` stays SINGULAR. A second "cheap" predicate for scan planning // is exactly how a pruner starts silently disagreeing with the scanner about // what matches — so the page planner and the hit decision call this one // function, typed on the three fields it reads so a summaries record and a // full transcript record both satisfy it. import { forEachLeaf, isLeaf, isLeafActive, isNodeActive, type LayerScope, type QueryNode, } from "../searchQuery"; import { matchAliases, type SearchAlias } from "../searchAliases"; import { VIDEO_STATES, type VideoState } from "../availability"; import type { Cue } from "../vtt"; import { hitsAcrossTracks, type AltTrack } from "../captionTracks"; import { clock, truncate, type Matcher } from "./window"; export type { Matcher }; // ─── Matcher compilation ─── // Build the combined, alias-aware matcher for a query, mirroring the browser's // buildSearchRoot OR-of-leaves semantics: a text matches if the plain query // substring matches OR any fired alias's suggestion regex matches. An explicit // `regex` query is taken verbatim with NO alias expansion (the caller is // crafting their own pattern). Returns the fired aliases so the tool can report // which curated expansions it applied. export function buildMatcher(opts: { query: string; regex?: boolean; useAliases?: boolean; aliases?: SearchAlias[]; }): { match: Matcher; firedAliases: SearchAlias[] } { if (opts.regex) { const re = new RegExp(opts.query, "i"); return { match: (t) => re.test(t), firedAliases: [] }; } const needle = opts.query.toLowerCase(); const plain: Matcher = (t) => t.toLowerCase().includes(needle); const useAliases = opts.useAliases !== false; const fired = useAliases && opts.aliases && opts.aliases.length > 0 ? matchAliases(opts.query, "transcripts", opts.aliases) : []; if (fired.length === 0) return { match: plain, firedAliases: [] }; const aliasMatchers: Matcher[] = fired.map((a) => { if (a.useRegex) { try { const re = new RegExp(a.suggestion, "i"); return (t: string) => re.test(t); } catch { // malformed suggestion regex — fall back to substring on the literal } } const n = a.suggestion.toLowerCase(); return (t: string) => t.toLowerCase().includes(n); }); const match: Matcher = (t) => plain(t) || aliasMatchers.some((m) => m(t)); return { match, firedAliases: fired }; } export type LeafMatcher = { scope: LayerScope; test: Matcher }; // Compile a matcher per active leaf. Transcripts leaves are alias-aware (unless // they're regex); every other scope matches plain-substring / regex only. Fired // aliases are unioned for the caller to report. export function buildLeafMatchers( root: QueryNode, aliases: SearchAlias[], useAliases: boolean, ): { matchers: Map; fired: SearchAlias[] } { const matchers = new Map(); const fired: SearchAlias[] = []; forEachLeaf(root, (leaf) => { if (!isLeafActive(leaf)) return; const aliasAware = leaf.scope === "transcripts" && !leaf.useRegex && useAliases; const built = buildMatcher({ query: leaf.query, regex: leaf.useRegex, useAliases: aliasAware, aliases: aliasAware ? aliases : [], }); matchers.set(leaf.id, { scope: leaf.scope, test: built.match }); for (const a of built.firedAliases) { if (!fired.some((x) => x.id === a.id)) fired.push(a); } }); return { matchers, fired }; } // ─── Per-record evaluation ─── // A snippet tagged with the scope it came from, carrying the seconds needed to // build a moment link. `seconds` is 0 for the non-timed scopes (metadata / // description / tags / posts). export type ScopedSnippet = { scope: LayerScope; track?: string; clock: string; seconds: number; text: string; }; // Per-record context the tree evaluates against. export type RecordCtx = { title: string; channel: string; description: string; tags: string; cues: Cue[]; // The record's alternate English tracks (lib/captionTracks.ts), when it has // any: the transcripts scope reads them too. altTracks?: AltTrack[]; chatCues: Cue[]; // The post body, when this record IS a post rather than a video. Empty for a // video record, so a posts-scope leaf never matches one. postText?: string; snippetsPerVideo: number; includeSnippets: boolean; // Snippet truncation width — `SearchPolicy.snippetChars`. REQUIRED: a default // here would be one caller's budget imposed on every other, and the way a // viewer adopter discovers it is by shipping quietly clipped excerpts. snippetChars: number; }; export type LeafOutcome = { matched: boolean; count: number; hits: ScopedSnippet[] }; export function evalLeaf( leaf: QueryNode, m: LeafMatcher, ctx: RecordCtx, ): LeafOutcome { if (!isLeaf(leaf)) return { matched: false, count: 0, hits: [] }; const max = ctx.snippetChars; const hits: ScopedSnippet[] = []; const push = (s: ScopedSnippet): void => { if (ctx.includeSnippets && hits.length < ctx.snippetsPerVideo) hits.push(s); }; let count = 0; switch (m.scope) { case "transcripts": // Every English track of the record (lib/captionTracks.ts): a match only // an alternate holds wears that alternate's track id. for (const cue of hitsAcrossTracks( { cues: ctx.cues, altTracks: ctx.altTracks }, (cues) => cues.filter((c) => m.test(c.text)), )) { count++; push({ scope: "transcripts", ...(cue.track ? { track: cue.track } : {}), clock: clock(cue.start), seconds: cue.start, text: truncate(cue.text, max), }); } break; case "chat": for (const cue of ctx.chatCues) { if (!m.test(cue.text)) continue; count++; push({ scope: "chat", track: "live_chat", clock: clock(cue.start), seconds: cue.start, text: truncate(cue.text, max), }); } break; case "posts": // A post has no timeline: one hit, seconds 0 — the same convention the // metadata / description / tags scopes already use. if (ctx.postText && m.test(ctx.postText)) { count++; push({ scope: "posts", clock: clock(0), seconds: 0, text: truncate(ctx.postText, max), }); } break; case "metadata": { const titleHit = m.test(ctx.title); const channelHit = m.test(ctx.channel); if (titleHit || channelHit) { count++; if (titleHit) { push({ scope: "metadata", clock: clock(0), seconds: 0, text: truncate(ctx.title, max), }); } else { push({ scope: "metadata", clock: clock(0), seconds: 0, text: `Channel: ${ctx.channel}`, }); } } break; } case "description": if (ctx.description && m.test(ctx.description)) { count++; push({ scope: "description", clock: clock(0), seconds: 0, text: truncate(ctx.description, max), }); } break; case "tags": if (ctx.tags && m.test(ctx.tags)) { count++; push({ scope: "tags", clock: clock(0), seconds: 0, text: truncate(ctx.tags, max), }); } break; } return { matched: count > 0, count, hits }; } export type NodeOutcome = { match: boolean; count: number; hits: ScopedSnippet[] }; // Evaluate the tree against one record — the per-record mirror of searchEval's // AND/OR/negate. A negated node contributes no hits (like the browser's `diff`). // An inactive (empty) subtree is identity (matches, no hits). export function evalNode( node: QueryNode, matchers: ReadonlyMap, ctx: RecordCtx, ): NodeOutcome { if (!isNodeActive(node)) return { match: true, count: 0, hits: [] }; if (isLeaf(node)) { const m = matchers.get(node.id); if (!m) return { match: true, count: 0, hits: [] }; const r = evalLeaf(node, m, ctx); if (node.negate) return { match: !r.matched, count: 0, hits: [] }; return { match: r.matched, count: node.contributeHits ? r.count : 0, hits: node.contributeHits ? r.hits : [], }; } const active = node.children.filter(isNodeActive); if (active.length === 0) return { match: true, count: 0, hits: [] }; if (node.op === "AND") { let allMatch = true; let count = 0; const hits: ScopedSnippet[] = []; for (const c of active) { const r = evalNode(c, matchers, ctx); if (!r.match) { allMatch = false; break; } count += r.count; hits.push(...r.hits); } const match = node.negate ? !allMatch : allMatch; return match && !node.negate ? { match, count, hits } : { match, count: 0, hits: [] }; } // OR let any = false; let count = 0; const hits: ScopedSnippet[] = []; for (const c of active) { const r = evalNode(c, matchers, ctx); if (r.match) { any = true; count += r.count; hits.push(...r.hits); } } const match = node.negate ? !any : any; return match && !node.negate ? { match, count, hits } : { match, count: 0, hits: [] }; } // ─── The share-filter predicate ─── // The positive share-filter selection (parseShareV1), as a predicate input. export type SearchFilters = { // ft — video / livestream types kept. videos: boolean; livestreams: boolean; // fa — all-ages / age-restricted kept. allAges: boolean; restricted: boolean; // fav — the VideoState values kept (see lib/availability). Absent from the // set means filtered out. states: ReadonlySet; // fdf / fdt — inclusive upload-date bounds, "YYYYMMDD". dateFrom?: string; dateTo?: string; // tg — curated tag ids (common/lib/curatedTags.ts), ORed together: a record // passes when it carries ANY of them. NOT the yt-dlp keywords in // `TranscriptSummary.tags` — see the naming note there. // // Absent or empty means "no tag filter", which is why an empty array must // never read as "matches nothing": the UI hands this straight through from a // chip row that starts empty. curatedTags?: string[]; }; // Typed on the three fields it actually reads rather than on TranscriptDetail, // so the SAME predicate can be applied to a summaries index record while // planning a scan and to the full transcript record while deciding a hit. export type FilterableRecord = { isLivestream?: boolean; ageRestricted?: boolean; uploadDate: string; // Curated tag ids carried by the record. OPTIONAL and omitted-when-empty on // the wire, so a record from a site built before corpus spec 4 simply has // none — and, correctly, passes no tag filter. curatedTags?: string[]; }; export function passesFilters( rec: FilterableRecord, f: SearchFilters, avail: { state: VideoState } | undefined, ): boolean { // ft — type if (rec.isLivestream ? !f.livestreams : !f.videos) return false; // fa — audience if (rec.ageRestricted ? !f.restricted : !f.allAges) return false; // fav — presence on the source platform if (!f.states.has(avail?.state ?? "available")) return false; // fdf / fdt — upload-date range (lexicographic on YYYYMMDD) if (f.dateFrom && rec.uploadDate < f.dateFrom) return false; if (f.dateTo && rec.uploadDate > f.dateTo) return false; // tg — curated tags. OR across the selection (a video tagged either // "eva-collab" OR "eva-in-chat" passes both-chips-selected), which is the // operator's decision: the chips are a union, not an intersection. if (f.curatedTags && f.curatedTags.length > 0) { const has = rec.curatedTags; if (!has || has.length === 0) return false; if (!f.curatedTags.some((t) => has.includes(t))) return false; } return true; } // True when the fav filter could exclude something (so availability must be // fetched). If every availability bucket is kept, there's nothing to look up. export function needsAvailability(f: SearchFilters | null | undefined): boolean { return !!f && !VIDEO_STATES.every((s) => f.states.has(s)); } // True when a filter set could actually exclude something. An all-permissive // filter (every state kept, both media types, both audiences, no dates) is the // same query as no filter at all, and must NOT trigger an index read — that is // the guard against making an unfiltered query slower by planning it. export function filterIsSelective(f: SearchFilters | null | undefined): boolean { if (!f) return false; if (!VIDEO_STATES.every((s) => f.states.has(s))) return true; if (!f.videos || !f.livestreams) return true; if (!f.allAges || !f.restricted) return true; if (f.curatedTags && f.curatedTags.length > 0) return true; return Boolean(f.dateFrom || f.dateTo); }