Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 001b3cf6359702c398faa8dfccbba47c7bef3275
parent 2485b59fc1bfdb5c0f4176497017b20b9e2d8e6c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 12 Sep 2026 02:49:15 -0400

mcp: search.ts is orchestration over lib/search/, and the caps are data

1,563 lines -> 1,206. What left: the matcher compilation, the excerpt
shapes, the per-record tree evaluator, the one filter predicate, the
thread comparator and the mirror-collapse rules — all of which now live
in `common/lib/search/` beside the viewer's half of the same pipeline.
What stayed: the walk. Which pages to read, in what order, and when to
stop is the only thing this file is for.

The two scanners take a `policy` (default `MCP_POLICY`) instead of
reading three module-private constants. That is the mechanism by which
the bench cannot move: a scanner that is HANDED its ceiling cannot
quietly pick a different one, and `MCP_POLICY` holds the four numbers
verbatim — 400 pages, 2,000 videos, 200 window lines, 240 snippet
characters. `truncate(post.text, 120)` and `(…, 480)` keep their
literals; they were never the policy's to set.

Every export the thirteen tools and the two test files import is still
exported — `buildMatcher`, `filterIsSelective`, `Matcher`,
`ScopedSnippet` and `SearchFilters` as re-exports of their new homes, so
no importer moved a line. mcp tests: 205 passed, unchanged.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mmcp/src/search.ts | 531+++++++++++++------------------------------------------------------------------
1 file changed, 87 insertions(+), 444 deletions(-)

diff --git a/mcp/src/search.ts b/mcp/src/search.ts @@ -1,43 +1,62 @@ -import { formatDuration } from "yt-dlp-transcript-common/lib/format"; +// ORCHESTRATION over `yt-dlp-transcript-common/lib/search/`. +// +// Everything below plans and drives reads against a ShardSource; the matching, +// excerpting, tree algebra, filter predicate, ordering and mirror collapsing +// all live in the shared pipeline, once, alongside the viewer's half of it. +// The caps travel as `MCP_POLICY` rather than being re-derived here, which is +// what makes the bench's structural read/byte counters unchanged BY +// CONSTRUCTION: this file cannot quietly pick a different ceiling than the one +// the policy names. import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; -import type { Cue } from "yt-dlp-transcript-common/lib/vtt"; import type { Post } from "yt-dlp-transcript-common/lib/posts"; +import type { Cue } from "yt-dlp-transcript-common/lib/vtt"; import type { Platform } from "yt-dlp-transcript-common/lib/platform"; import { - forEachLeaf, - isLeaf, - isLeafActive, - isNodeActive, type GroupNode, type LayerScope, - type QueryNode, } from "yt-dlp-transcript-common/lib/searchQuery"; -import { - matchAliases, - type SearchAlias, -} from "yt-dlp-transcript-common/lib/searchAliases"; -import { - windowCues, - cuesToSnippets, - mergeSnippets, - type WindowSnippet, -} from "yt-dlp-transcript-common/lib/transcriptWindow"; +import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import { resolveChannelGroupId, type ChannelGroup, } from "yt-dlp-transcript-common/lib/channelGroups"; -import { - VIDEO_STATES, - type VideoState, -} from "yt-dlp-transcript-common/lib/availability"; import { mapConcurrent } from "yt-dlp-transcript-common/lib/concurrency"; import { + MCP_POLICY, + type SearchPolicy, +} from "yt-dlp-transcript-common/lib/search/policy"; +import { + clock, + truncate, + windowedTranscript, +} from "yt-dlp-transcript-common/lib/search/window"; +import { + buildMatcher, + buildLeafMatchers, + evalNode, + filterIsSelective, + needsAvailability, + passesFilters, + type LeafMatcher, + type Matcher, + type RecordCtx, + type ScopedSnippet, + type SearchFilters, +} from "yt-dlp-transcript-common/lib/search/evalTree"; +import { collapseDuplicates as collapseHits } from "yt-dlp-transcript-common/lib/search/collapse"; +import { rankThread } from "yt-dlp-transcript-common/lib/search/rank"; +import { DEFAULT_PAGE_CONCURRENCY, type ChannelRef, type ShardSource, - type VideoAvailability, } from "./source"; +// Re-exported so the thirteen tools and the two test files keep importing +// every name from `./search`, exactly as before. The definitions moved; the +// module's surface did not. +export { buildMatcher, filterIsSelective, MCP_POLICY }; +export type { Matcher, ScopedSnippet, SearchFilters, SearchPolicy }; + export type Snippet = { clock: string; seconds: number; @@ -200,7 +219,7 @@ export type SearchHit = { export type SearchResult = { hits: SearchHit[]; - // Total matched videos found in this scan (up to HARD_VIDEO_CAP). `hits` is + // Total matched videos found in this scan (up to policy.hardVideoCap). `hits` is // the [offset, offset+limit) slice of that set, so total is stable across // paged calls and lets a caller plan a full sweep. total: number; @@ -222,7 +241,7 @@ export type SearchResult = { // either way — distinct from "checked, found none". duplicates: { collapsed: number; clusters: number; available: boolean }; // Coverage is partial — the page cap (MAX_PAGES) or the video cap - // (HARD_VIDEO_CAP) was reached before the corpus was fully scanned. + // (policy.hardVideoCap) was reached before the corpus was fully scanned. truncated: boolean; // What was actually covered, in enough detail to act on. A cap truncates in // CHANNEL ITERATION ORDER, not at random, so "partial" on its own is @@ -254,24 +273,6 @@ export type SearchResult = { }; }; -// A hard ceiling on shard pages fetched per query so a rare term over a large -// (or hub-wide) corpus can't run away. Reaching it sets `truncated`. -const MAX_PAGES = 400; - -// A ceiling on the number of matched videos we collect before we stop counting, -// so `total` stays bounded and stable even for a very common term. Reaching it -// also sets `truncated` (the true total is higher than reported). -const HARD_VIDEO_CAP = 2000; - -// Cap on windowed excerpt lines emitted per video by getWindowedTranscript, so a -// video with hundreds of matches can't blow the batch's token budget. -const WINDOW_LINE_CAP = 200; - -function clock(seconds: number): string { - const s = Math.max(0, Math.floor(seconds)); - return s === 0 ? "0:00" : formatDuration(s); -} - // ─── Windowed page reading ─── // // Pages were read one `await` at a time, which on a local corpus leaves the @@ -335,18 +336,6 @@ export type ScanPlan = { unknownVideos: number; }; -// True when a filter set could actually exclude something. An all-permissive -// filter (every state kept, both media types, both audiences, no dates) is the -// same query as no filter at all, and must NOT trigger an index read — that is -// the guard against making an unfiltered query slower by planning it. -export function filterIsSelective(f: SearchFilters | null | undefined): boolean { - if (!f) return false; - if (!VIDEO_STATES.every((s) => f.states.has(s))) return true; - if (!f.videos || !f.livestreams) return true; - if (!f.allAges || !f.restricted) return true; - return Boolean(f.dateFrom || f.dateTo); -} - // Build the page plan for a scan. Falls back to "every page of every channel" // whenever pruning is impossible or pointless: no selective filter, or a source // with no summaries index (both in-memory test stubs, and any site that ships @@ -408,72 +397,11 @@ export async function buildScanPlan( }; } -export type Matcher = (text: string) => boolean; - -// Build the combined, alias-aware matcher for a query, mirroring the browser's -// buildSearchRoot OR-of-leaves semantics: a text matches if the plain query -// substring matches OR any fired alias's suggestion regex matches. An explicit -// `regex` query is taken verbatim with NO alias expansion (the caller is -// crafting their own pattern). Returns the fired aliases so the tool can report -// which curated expansions it applied. -export function buildMatcher(opts: { - query: string; - regex?: boolean; - useAliases?: boolean; - aliases?: SearchAlias[]; -}): { match: Matcher; firedAliases: SearchAlias[] } { - if (opts.regex) { - const re = new RegExp(opts.query, "i"); - return { match: (t) => re.test(t), firedAliases: [] }; - } - const needle = opts.query.toLowerCase(); - const plain: Matcher = (t) => t.toLowerCase().includes(needle); - - const useAliases = opts.useAliases !== false; - const fired = - useAliases && opts.aliases && opts.aliases.length > 0 - ? matchAliases(opts.query, "transcripts", opts.aliases) - : []; - if (fired.length === 0) return { match: plain, firedAliases: [] }; - - const aliasMatchers: Matcher[] = fired.map((a) => { - if (a.useRegex) { - try { - const re = new RegExp(a.suggestion, "i"); - return (t: string) => re.test(t); - } catch { - // malformed suggestion regex — fall back to substring on the literal - } - } - const n = a.suggestion.toLowerCase(); - return (t: string) => t.toLowerCase().includes(n); - }); - - const match: Matcher = (t) => plain(t) || aliasMatchers.some((m) => m(t)); - return { match, firedAliases: fired }; -} - -function truncate(text: string, max = 240): string { - const t = text.trim().replace(/\s+/g, " "); - return t.length > max ? t.slice(0, max - 1) + "…" : t; -} - -// Collapse cross-platform mirrors in a result list, IN PLACE, keeping one row -// per recording. Returns what it did so the caller can report it. -// -// Two rules make this safe to have on by default: -// -// 1. The kept row is the cluster's canonical member WHEN that member is -// itself among the matches — otherwise it is simply the first match. A -// mirror is frequently the only surviving copy of a deleted upload, and -// preferring an absent canonical would delete exactly the evidence a -// "what did the removed videos say" question is asking for. -// 2. The collapsed copies are NAMED on the row they folded into. Nothing -// vanishes; the count stops double-counting. Pass collapse_duplicates:false -// to see every upload as its own row. -// -// Timestamps are never mapped between copies here — that requires the per-pair -// `aligned` gate, and this function does not move a single second of anything. +// Fetch the duplicate index and apply the collapse rules (lib/search/collapse) +// to a result list, IN PLACE. Only the read and the "this corpus ships no +// duplicates.json" distinction live here: `available: false` means no claim +// about mirrors can be made either way, which is different from "checked, +// found none". async function collapseDuplicates( source: ShardSource, all: SearchHit[], @@ -490,51 +418,20 @@ async function collapseDuplicates( } if (index.size === 0) return { collapsed: 0, clusters: 0, available: true }; - const repIndexOf = new Map<string, number>(); // clusterId -> index in `kept` - const kept: SearchHit[] = []; - let collapsed = 0; - for (const hit of all) { - const membership = index.get(hit.slug); - if (!membership) { - kept.push(hit); - continue; - } - const at = repIndexOf.get(membership.clusterId); - if (at === undefined) { - repIndexOf.set(membership.clusterId, kept.length); - kept.push(hit); - continue; - } - collapsed++; - const rep = kept[at]; - const repIsCanonical = index.get(rep.slug)?.isCanonical === true; - const fold = (into: SearchHit, gone: SearchHit): SearchHit => ({ - ...into, - mirrors: [ - ...(into.mirrors ?? []), - ...(gone.mirrors ?? []), - { videoId: gone.videoId, channelName: gone.channelName, slug: gone.slug }, - ], - }); - // Promote the canonical member to the representative if it turns up later; - // otherwise fold this copy into the incumbent. - kept[at] = repIsCanonical || !membership.isCanonical - ? fold(rep, hit) - : fold(hit, rep); - } + const { collapsed, clusters, kept } = collapseHits(all, index); if (collapsed > 0) { all.length = 0; all.push(...kept); } - return { collapsed, clusters: repIndexOf.size, available: true }; + return { collapsed, clusters, available: true }; } // Scan a source's transcript shards for `query` (alias-aware by default), -// collecting ALL matched videos up to HARD_VIDEO_CAP so counting is stable, then +// collecting ALL matched videos up to policy.hardVideoCap so counting is stable, then // returning the [offset, offset+limit) slice with a `total`/`hasMore`. Plain // substring by default, or a caller-supplied regex; either can be paged. Set // `includeSnippets:false` for a cheap worklist (id/title/channel/date/matches, -// no cue text). Stops early at MAX_PAGES (→ truncated) and HARD_VIDEO_CAP. +// no cue text). Stops early at MAX_PAGES (→ truncated) and policy.hardVideoCap. export async function searchTranscripts( source: ShardSource, opts: { @@ -571,11 +468,17 @@ export async function searchTranscripts( // collapsed copies are named on the row they fold into, never dropped // silently. collapseDuplicates?: boolean; + // The caps this scan runs under. Defaults to MCP_POLICY — the same four + // numbers this scanner used when they were private constants — so an + // existing caller reads exactly the same pages and parses exactly the same + // bytes. A caller with a different budget passes its own. + policy?: SearchPolicy; }, ): Promise<SearchResult> { + const policy = opts.policy ?? MCP_POLICY; const limit = opts.limit ?? 20; const offset = Math.max(0, opts.offset ?? 0); - const maxPages = opts.maxPages ?? MAX_PAGES; + const maxPages = opts.maxPages ?? policy.maxPages; const includeSnippets = opts.includeSnippets !== false; const snippetsPerVideo = opts.snippetsPerVideo ?? 4; @@ -699,7 +602,7 @@ export async function searchTranscripts( push({ clock: clock(cue.start), seconds: cue.start, - text: truncate(cue.text), + text: truncate(cue.text, policy.snippetChars), }); } } @@ -722,7 +625,7 @@ export async function searchTranscripts( push({ clock: clock(0), seconds: 0, - text: truncate(rec.description), + text: truncate(rec.description, policy.snippetChars), scope: "description", }); } @@ -730,7 +633,12 @@ export async function searchTranscripts( const tags = (rec.tags ?? []).join(", "); if (tags && match(tags)) { otherHit = true; - push({ clock: clock(0), seconds: 0, text: truncate(tags), scope: "tags" }); + push({ + clock: clock(0), + seconds: 0, + text: truncate(tags, policy.snippetChars), + scope: "tags", + }); } } if (wantChat && chatCuesFor) { @@ -740,7 +648,7 @@ export async function searchTranscripts( push({ clock: clock(cue.start), seconds: cue.start, - text: truncate(cue.text), + text: truncate(cue.text, policy.snippetChars), scope: "chat", }); } @@ -761,7 +669,7 @@ export async function searchTranscripts( matches: matches || 1, snippets, }); - if (all.length >= HARD_VIDEO_CAP) { + if (all.length >= policy.hardVideoCap) { truncated = true; channelStopped = { channel: ch.name, @@ -828,7 +736,7 @@ export async function searchTranscripts( ? [{ clock: "", seconds: 0, text: truncate(post.text, 480) }] : [], }); - if (all.length >= HARD_VIDEO_CAP) { + if (all.length >= policy.hardVideoCap) { truncated = true; break postsOuter; } @@ -961,11 +869,7 @@ export async function getThread( if ((p.threadId || p.id) === threadId) thread.push(p); } } - thread.sort((a, b) => - a.createdAt === b.createdAt - ? a.id.localeCompare(b.id) - : a.createdAt.localeCompare(b.createdAt), - ); + rankThread(thread); return thread.length > 0 ? thread : [post]; } @@ -982,33 +886,15 @@ export function getWindowedTranscript( // the line's clock + start seconds. Bare clock when omitted. stamp?: (clock: string, seconds: number) => string; // Cap on merged excerpt lines emitted for this video (default - // WINDOW_LINE_CAP). The earliest lines are kept; `maxCues` above stays the - // per-window bound. + // MCP_POLICY.windowLineCap). The earliest lines are kept; `maxCues` above + // stays the per-window bound. maxLines?: number; } = {}, ): { lines: string[]; matchCount: number } { - const cues = record.cues ?? []; - const timestamps = opts.timestamps !== false; - let merged: WindowSnippet[] = []; - let matchCount = 0; - for (const cue of cues) { - if (!matcher(cue.text)) continue; - matchCount++; - const win = cuesToSnippets( - windowCues(cues, cue.start, { - before: opts.before, - after: opts.after, - maxCues: opts.maxCues, - }), - ); - merged = mergeSnippets(merged, win, opts.maxLines ?? WINDOW_LINE_CAP); - } - const lines = merged.map((s) => { - if (!timestamps) return s.text; - const stamp = opts.stamp ? opts.stamp(s.clock, s.seconds) : s.clock; - return `[${stamp}] ${s.text}`; + return windowedTranscript(record.cues ?? [], matcher, { + ...opts, + maxLines: opts.maxLines ?? MCP_POLICY.windowLineCap, }); - return { lines, matchCount }; } // Locate a single video across the source's channels via each channel's @@ -1067,26 +953,10 @@ export async function findVideo( // the simpler `searchTranscripts` scanner above (a trivial single-leaf query), // so today's callers/tests are unaffected. // -// It mirrors the browser's tree algebra (searchEval.ts) and filter predicate -// (SearchSessionContext.tsx `passesFilter`) — evaluated per record to a boolean -// instead of over slug sets — and the per-scope text extraction of -// searchPipeline.ts. - -// The positive share-filter selection (parseShareV1), as a predicate input. -export type SearchFilters = { - // ft — video / livestream types kept. - videos: boolean; - livestreams: boolean; - // fa — all-ages / age-restricted kept. - allAges: boolean; - restricted: boolean; - // fav — the VideoState values kept (see common/lib/availability). Absent - // from the set means filtered out. - states: ReadonlySet<VideoState>; - // fdf / fdt — inclusive upload-date bounds, "YYYYMMDD". - dateFrom?: string; - dateTo?: string; -}; +// The tree algebra, the leaf evaluation and the filter predicate are +// `lib/search/evalTree.ts` — the same module the viewer's `lib/searchEval.ts` +// points at for the meaning of AND / OR / negate. What stays here is the walk: +// which pages to read, in what order, and when to stop. export type SearchSpec = { // The root of the composite query (a `qt=` tree, or a synthesized single leaf @@ -1102,17 +972,6 @@ export type SearchSpec = { useAliases?: boolean; }; -// A snippet tagged with the scope it came from, carrying the seconds needed to -// build a moment link. `seconds` is 0 for the non-timed scopes (metadata / -// description / tags). -export type ScopedSnippet = { - scope: LayerScope; - track?: string; - clock: string; - seconds: number; - text: string; -}; - export type SpecHit = { videoId: string; slug: string; @@ -1139,226 +998,6 @@ export type SpecResult = { truncated: boolean; }; -type LeafMatcher = { scope: LayerScope; test: Matcher }; - -// Compile a matcher per active leaf. Transcripts leaves are alias-aware (unless -// they're regex); every other scope matches plain-substring / regex only. Fired -// aliases are unioned for the caller to report. -function buildLeafMatchers( - root: QueryNode, - aliases: SearchAlias[], - useAliases: boolean, -): { matchers: Map<string, LeafMatcher>; fired: SearchAlias[] } { - const matchers = new Map<string, LeafMatcher>(); - const fired: SearchAlias[] = []; - forEachLeaf(root, (leaf) => { - if (!isLeafActive(leaf)) return; - const aliasAware = - leaf.scope === "transcripts" && !leaf.useRegex && useAliases; - const built = buildMatcher({ - query: leaf.query, - regex: leaf.useRegex, - useAliases: aliasAware, - aliases: aliasAware ? aliases : [], - }); - matchers.set(leaf.id, { scope: leaf.scope, test: built.match }); - for (const a of built.firedAliases) { - if (!fired.some((x) => x.id === a.id)) fired.push(a); - } - }); - return { matchers, fired }; -} - -// Per-record context the tree evaluates against. -type RecordCtx = { - title: string; - channel: string; - description: string; - tags: string; - cues: Cue[]; - chatCues: Cue[]; - // The post body, when this record IS a post rather than a video. Empty for a - // video record, so a posts-scope leaf never matches one. - postText?: string; - snippetsPerVideo: number; - includeSnippets: boolean; -}; - -type LeafOutcome = { matched: boolean; count: number; hits: ScopedSnippet[] }; - -function evalLeaf(leaf: QueryNode, m: LeafMatcher, ctx: RecordCtx): LeafOutcome { - if (!isLeaf(leaf)) return { matched: false, count: 0, hits: [] }; - const hits: ScopedSnippet[] = []; - const push = (s: ScopedSnippet): void => { - if (ctx.includeSnippets && hits.length < ctx.snippetsPerVideo) hits.push(s); - }; - let count = 0; - switch (m.scope) { - case "transcripts": - for (const cue of ctx.cues) { - if (!m.test(cue.text)) continue; - count++; - push({ - scope: "transcripts", - clock: clock(cue.start), - seconds: cue.start, - text: truncate(cue.text), - }); - } - break; - case "chat": - for (const cue of ctx.chatCues) { - if (!m.test(cue.text)) continue; - count++; - push({ - scope: "chat", - track: "live_chat", - clock: clock(cue.start), - seconds: cue.start, - text: truncate(cue.text), - }); - } - break; - case "posts": - // A post has no timeline: one hit, seconds 0 — the same convention the - // metadata / description / tags scopes already use. - if (ctx.postText && m.test(ctx.postText)) { - count++; - push({ - scope: "posts", - clock: clock(0), - seconds: 0, - text: truncate(ctx.postText), - }); - } - break; - case "metadata": { - const titleHit = m.test(ctx.title); - const channelHit = m.test(ctx.channel); - if (titleHit || channelHit) { - count++; - if (titleHit) { - push({ scope: "metadata", clock: clock(0), seconds: 0, text: truncate(ctx.title) }); - } else { - push({ scope: "metadata", clock: clock(0), seconds: 0, text: `Channel: ${ctx.channel}` }); - } - } - break; - } - case "description": - if (ctx.description && m.test(ctx.description)) { - count++; - push({ scope: "description", clock: clock(0), seconds: 0, text: truncate(ctx.description) }); - } - break; - case "tags": - if (ctx.tags && m.test(ctx.tags)) { - count++; - push({ scope: "tags", clock: clock(0), seconds: 0, text: truncate(ctx.tags) }); - } - break; - } - return { matched: count > 0, count, hits }; -} - -type NodeOutcome = { match: boolean; count: number; hits: ScopedSnippet[] }; - -// Evaluate the tree against one record — mirrors searchEval's AND/OR/negate, -// per record. A negated node contributes no hits (like the browser's `diff`). -// An inactive (empty) subtree is identity (matches, no hits). -function evalNode( - node: QueryNode, - matchers: Map<string, LeafMatcher>, - ctx: RecordCtx, -): NodeOutcome { - if (!isNodeActive(node)) return { match: true, count: 0, hits: [] }; - if (isLeaf(node)) { - const m = matchers.get(node.id); - if (!m) return { match: true, count: 0, hits: [] }; - const r = evalLeaf(node, m, ctx); - if (node.negate) return { match: !r.matched, count: 0, hits: [] }; - return { - match: r.matched, - count: node.contributeHits ? r.count : 0, - hits: node.contributeHits ? r.hits : [], - }; - } - const active = node.children.filter(isNodeActive); - if (active.length === 0) return { match: true, count: 0, hits: [] }; - if (node.op === "AND") { - let allMatch = true; - let count = 0; - const hits: ScopedSnippet[] = []; - for (const c of active) { - const r = evalNode(c, matchers, ctx); - if (!r.match) { - allMatch = false; - break; - } - count += r.count; - hits.push(...r.hits); - } - const match = node.negate ? !allMatch : allMatch; - return match && !node.negate - ? { match, count, hits } - : { match, count: 0, hits: [] }; - } - // OR - let any = false; - let count = 0; - const hits: ScopedSnippet[] = []; - for (const c of active) { - const r = evalNode(c, matchers, ctx); - if (r.match) { - any = true; - count += r.count; - hits.push(...r.hits); - } - } - const match = node.negate ? !any : any; - return match && !node.negate - ? { match, count, hits } - : { match, count: 0, hits: [] }; -} - -// The share-filter predicate over a transcript record + its availability. -// Mirrors SearchSessionContext.tsx `passesFilter` exactly. -// -// Typed on the three fields it actually reads rather than on TranscriptDetail, -// so the SAME predicate can be applied to a summaries index record while -// planning a scan and to the full transcript record while deciding a hit. -// There is deliberately only one of these: a second, "cheap" predicate for -// planning is exactly how a pruner starts silently disagreeing with the -// scanner about what matches. -type FilterableRecord = { - isLivestream?: boolean; - ageRestricted?: boolean; - uploadDate: string; -}; - -function passesFilters( - rec: FilterableRecord, - f: SearchFilters, - avail: VideoAvailability | undefined, -): boolean { - // ft — type - if (rec.isLivestream ? !f.livestreams : !f.videos) return false; - // fa — audience - if (rec.ageRestricted ? !f.restricted : !f.allAges) return false; - // fav — presence on the source platform - if (!f.states.has(avail?.state ?? "available")) return false; - // fdf / fdt — upload-date range (lexicographic on YYYYMMDD) - if (f.dateFrom && rec.uploadDate < f.dateFrom) return false; - if (f.dateTo && rec.uploadDate > f.dateTo) return false; - return true; -} - -// True when the fav filter could exclude something (so availability must be -// fetched). If every availability bucket is kept, there's nothing to look up. -function needsAvailability(f: SearchFilters | null | undefined): boolean { - return !!f && !VIDEO_STATES.every((s) => f.states.has(s)); -} - // A small lazy live-chat fetcher: per-channel subs manifest + page caches, so a // chat-scope leaf only pulls the subs shards it actually touches. function makeChatFetcher(source: ShardSource) { @@ -1399,11 +1038,13 @@ export async function runSearchSpec( includeSnippets?: boolean; maxPages?: number; snippetsPerVideo?: number; + policy?: SearchPolicy; } = {}, ): Promise<SpecResult> { + const policy = opts.policy ?? MCP_POLICY; const limit = opts.limit ?? 20; const offset = Math.max(0, opts.offset ?? 0); - const maxPages = opts.maxPages ?? MAX_PAGES; + const maxPages = opts.maxPages ?? policy.maxPages; const includeSnippets = opts.includeSnippets !== false; const snippetsPerVideo = opts.snippetsPerVideo ?? 4; const filters = spec.filters ?? null; @@ -1460,6 +1101,7 @@ export async function runSearchSpec( chatCues: chatCuesFor ? await chatCuesFor(ch, rec) : [], snippetsPerVideo, includeSnippets, + snippetChars: policy.snippetChars, }; const r = evalNode(spec.tree, matchers, ctx); if (!r.match) continue; @@ -1477,7 +1119,7 @@ export async function runSearchSpec( matches: r.count || 1, snippets: r.hits, }); - if (all.length >= HARD_VIDEO_CAP) { + if (all.length >= policy.hardVideoCap) { truncated = true; break outer; } @@ -1523,6 +1165,7 @@ export async function runSearchSpec( postText: post.text, snippetsPerVideo, includeSnippets, + snippetChars: policy.snippetChars, }; const r = evalNode(spec.tree, matchers, ctx); if (!r.match) continue; @@ -1540,7 +1183,7 @@ export async function runSearchSpec( matches: r.count || 1, snippets: r.hits, }); - if (all.length >= HARD_VIDEO_CAP) { + if (all.length >= policy.hardVideoCap) { truncated = true; break postsOuter; }