commit 001b3cf6359702c398faa8dfccbba47c7bef3275
parent 2485b59fc1bfdb5c0f4176497017b20b9e2d8e6c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 12 Sep 2026 02:49:15 -0400
mcp: search.ts is orchestration over lib/search/, and the caps are data
1,563 lines -> 1,206. What left: the matcher compilation, the excerpt
shapes, the per-record tree evaluator, the one filter predicate, the
thread comparator and the mirror-collapse rules — all of which now live
in `common/lib/search/` beside the viewer's half of the same pipeline.
What stayed: the walk. Which pages to read, in what order, and when to
stop is the only thing this file is for.
The two scanners take a `policy` (default `MCP_POLICY`) instead of
reading three module-private constants. That is the mechanism by which
the bench cannot move: a scanner that is HANDED its ceiling cannot
quietly pick a different one, and `MCP_POLICY` holds the four numbers
verbatim — 400 pages, 2,000 videos, 200 window lines, 240 snippet
characters. `truncate(post.text, 120)` and `(…, 480)` keep their
literals; they were never the policy's to set.
Every export the thirteen tools and the two test files import is still
exported — `buildMatcher`, `filterIsSelective`, `Matcher`,
`ScopedSnippet` and `SearchFilters` as re-exports of their new homes, so
no importer moved a line. mcp tests: 205 passed, unchanged.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
| M | mcp/src/search.ts | | | 531 | +++++++++++++------------------------------------------------------------------ |
1 file changed, 87 insertions(+), 444 deletions(-)
diff --git a/mcp/src/search.ts b/mcp/src/search.ts
@@ -1,43 +1,62 @@
-import { formatDuration } from "yt-dlp-transcript-common/lib/format";
+// ORCHESTRATION over `yt-dlp-transcript-common/lib/search/`.
+//
+// Everything below plans and drives reads against a ShardSource; the matching,
+// excerpting, tree algebra, filter predicate, ordering and mirror collapsing
+// all live in the shared pipeline, once, alongside the viewer's half of it.
+// The caps travel as `MCP_POLICY` rather than being re-derived here, which is
+// what makes the bench's structural read/byte counters unchanged BY
+// CONSTRUCTION: this file cannot quietly pick a different ceiling than the one
+// the policy names.
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
-import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
import type { Post } from "yt-dlp-transcript-common/lib/posts";
+import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
import type { Platform } from "yt-dlp-transcript-common/lib/platform";
import {
- forEachLeaf,
- isLeaf,
- isLeafActive,
- isNodeActive,
type GroupNode,
type LayerScope,
- type QueryNode,
} from "yt-dlp-transcript-common/lib/searchQuery";
-import {
- matchAliases,
- type SearchAlias,
-} from "yt-dlp-transcript-common/lib/searchAliases";
-import {
- windowCues,
- cuesToSnippets,
- mergeSnippets,
- type WindowSnippet,
-} from "yt-dlp-transcript-common/lib/transcriptWindow";
+import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import {
resolveChannelGroupId,
type ChannelGroup,
} from "yt-dlp-transcript-common/lib/channelGroups";
-import {
- VIDEO_STATES,
- type VideoState,
-} from "yt-dlp-transcript-common/lib/availability";
import { mapConcurrent } from "yt-dlp-transcript-common/lib/concurrency";
import {
+ MCP_POLICY,
+ type SearchPolicy,
+} from "yt-dlp-transcript-common/lib/search/policy";
+import {
+ clock,
+ truncate,
+ windowedTranscript,
+} from "yt-dlp-transcript-common/lib/search/window";
+import {
+ buildMatcher,
+ buildLeafMatchers,
+ evalNode,
+ filterIsSelective,
+ needsAvailability,
+ passesFilters,
+ type LeafMatcher,
+ type Matcher,
+ type RecordCtx,
+ type ScopedSnippet,
+ type SearchFilters,
+} from "yt-dlp-transcript-common/lib/search/evalTree";
+import { collapseDuplicates as collapseHits } from "yt-dlp-transcript-common/lib/search/collapse";
+import { rankThread } from "yt-dlp-transcript-common/lib/search/rank";
+import {
DEFAULT_PAGE_CONCURRENCY,
type ChannelRef,
type ShardSource,
- type VideoAvailability,
} from "./source";
+// Re-exported so the thirteen tools and the two test files keep importing
+// every name from `./search`, exactly as before. The definitions moved; the
+// module's surface did not.
+export { buildMatcher, filterIsSelective, MCP_POLICY };
+export type { Matcher, ScopedSnippet, SearchFilters, SearchPolicy };
+
export type Snippet = {
clock: string;
seconds: number;
@@ -200,7 +219,7 @@ export type SearchHit = {
export type SearchResult = {
hits: SearchHit[];
- // Total matched videos found in this scan (up to HARD_VIDEO_CAP). `hits` is
+ // Total matched videos found in this scan (up to policy.hardVideoCap). `hits` is
// the [offset, offset+limit) slice of that set, so total is stable across
// paged calls and lets a caller plan a full sweep.
total: number;
@@ -222,7 +241,7 @@ export type SearchResult = {
// either way — distinct from "checked, found none".
duplicates: { collapsed: number; clusters: number; available: boolean };
// Coverage is partial — the page cap (MAX_PAGES) or the video cap
- // (HARD_VIDEO_CAP) was reached before the corpus was fully scanned.
+ // (policy.hardVideoCap) was reached before the corpus was fully scanned.
truncated: boolean;
// What was actually covered, in enough detail to act on. A cap truncates in
// CHANNEL ITERATION ORDER, not at random, so "partial" on its own is
@@ -254,24 +273,6 @@ export type SearchResult = {
};
};
-// A hard ceiling on shard pages fetched per query so a rare term over a large
-// (or hub-wide) corpus can't run away. Reaching it sets `truncated`.
-const MAX_PAGES = 400;
-
-// A ceiling on the number of matched videos we collect before we stop counting,
-// so `total` stays bounded and stable even for a very common term. Reaching it
-// also sets `truncated` (the true total is higher than reported).
-const HARD_VIDEO_CAP = 2000;
-
-// Cap on windowed excerpt lines emitted per video by getWindowedTranscript, so a
-// video with hundreds of matches can't blow the batch's token budget.
-const WINDOW_LINE_CAP = 200;
-
-function clock(seconds: number): string {
- const s = Math.max(0, Math.floor(seconds));
- return s === 0 ? "0:00" : formatDuration(s);
-}
-
// ─── Windowed page reading ───
//
// Pages were read one `await` at a time, which on a local corpus leaves the
@@ -335,18 +336,6 @@ export type ScanPlan = {
unknownVideos: number;
};
-// True when a filter set could actually exclude something. An all-permissive
-// filter (every state kept, both media types, both audiences, no dates) is the
-// same query as no filter at all, and must NOT trigger an index read — that is
-// the guard against making an unfiltered query slower by planning it.
-export function filterIsSelective(f: SearchFilters | null | undefined): boolean {
- if (!f) return false;
- if (!VIDEO_STATES.every((s) => f.states.has(s))) return true;
- if (!f.videos || !f.livestreams) return true;
- if (!f.allAges || !f.restricted) return true;
- return Boolean(f.dateFrom || f.dateTo);
-}
-
// Build the page plan for a scan. Falls back to "every page of every channel"
// whenever pruning is impossible or pointless: no selective filter, or a source
// with no summaries index (both in-memory test stubs, and any site that ships
@@ -408,72 +397,11 @@ export async function buildScanPlan(
};
}
-export type Matcher = (text: string) => boolean;
-
-// Build the combined, alias-aware matcher for a query, mirroring the browser's
-// buildSearchRoot OR-of-leaves semantics: a text matches if the plain query
-// substring matches OR any fired alias's suggestion regex matches. An explicit
-// `regex` query is taken verbatim with NO alias expansion (the caller is
-// crafting their own pattern). Returns the fired aliases so the tool can report
-// which curated expansions it applied.
-export function buildMatcher(opts: {
- query: string;
- regex?: boolean;
- useAliases?: boolean;
- aliases?: SearchAlias[];
-}): { match: Matcher; firedAliases: SearchAlias[] } {
- if (opts.regex) {
- const re = new RegExp(opts.query, "i");
- return { match: (t) => re.test(t), firedAliases: [] };
- }
- const needle = opts.query.toLowerCase();
- const plain: Matcher = (t) => t.toLowerCase().includes(needle);
-
- const useAliases = opts.useAliases !== false;
- const fired =
- useAliases && opts.aliases && opts.aliases.length > 0
- ? matchAliases(opts.query, "transcripts", opts.aliases)
- : [];
- if (fired.length === 0) return { match: plain, firedAliases: [] };
-
- const aliasMatchers: Matcher[] = fired.map((a) => {
- if (a.useRegex) {
- try {
- const re = new RegExp(a.suggestion, "i");
- return (t: string) => re.test(t);
- } catch {
- // malformed suggestion regex — fall back to substring on the literal
- }
- }
- const n = a.suggestion.toLowerCase();
- return (t: string) => t.toLowerCase().includes(n);
- });
-
- const match: Matcher = (t) => plain(t) || aliasMatchers.some((m) => m(t));
- return { match, firedAliases: fired };
-}
-
-function truncate(text: string, max = 240): string {
- const t = text.trim().replace(/\s+/g, " ");
- return t.length > max ? t.slice(0, max - 1) + "…" : t;
-}
-
-// Collapse cross-platform mirrors in a result list, IN PLACE, keeping one row
-// per recording. Returns what it did so the caller can report it.
-//
-// Two rules make this safe to have on by default:
-//
-// 1. The kept row is the cluster's canonical member WHEN that member is
-// itself among the matches — otherwise it is simply the first match. A
-// mirror is frequently the only surviving copy of a deleted upload, and
-// preferring an absent canonical would delete exactly the evidence a
-// "what did the removed videos say" question is asking for.
-// 2. The collapsed copies are NAMED on the row they folded into. Nothing
-// vanishes; the count stops double-counting. Pass collapse_duplicates:false
-// to see every upload as its own row.
-//
-// Timestamps are never mapped between copies here — that requires the per-pair
-// `aligned` gate, and this function does not move a single second of anything.
+// Fetch the duplicate index and apply the collapse rules (lib/search/collapse)
+// to a result list, IN PLACE. Only the read and the "this corpus ships no
+// duplicates.json" distinction live here: `available: false` means no claim
+// about mirrors can be made either way, which is different from "checked,
+// found none".
async function collapseDuplicates(
source: ShardSource,
all: SearchHit[],
@@ -490,51 +418,20 @@ async function collapseDuplicates(
}
if (index.size === 0) return { collapsed: 0, clusters: 0, available: true };
- const repIndexOf = new Map<string, number>(); // clusterId -> index in `kept`
- const kept: SearchHit[] = [];
- let collapsed = 0;
- for (const hit of all) {
- const membership = index.get(hit.slug);
- if (!membership) {
- kept.push(hit);
- continue;
- }
- const at = repIndexOf.get(membership.clusterId);
- if (at === undefined) {
- repIndexOf.set(membership.clusterId, kept.length);
- kept.push(hit);
- continue;
- }
- collapsed++;
- const rep = kept[at];
- const repIsCanonical = index.get(rep.slug)?.isCanonical === true;
- const fold = (into: SearchHit, gone: SearchHit): SearchHit => ({
- ...into,
- mirrors: [
- ...(into.mirrors ?? []),
- ...(gone.mirrors ?? []),
- { videoId: gone.videoId, channelName: gone.channelName, slug: gone.slug },
- ],
- });
- // Promote the canonical member to the representative if it turns up later;
- // otherwise fold this copy into the incumbent.
- kept[at] = repIsCanonical || !membership.isCanonical
- ? fold(rep, hit)
- : fold(hit, rep);
- }
+ const { collapsed, clusters, kept } = collapseHits(all, index);
if (collapsed > 0) {
all.length = 0;
all.push(...kept);
}
- return { collapsed, clusters: repIndexOf.size, available: true };
+ return { collapsed, clusters, available: true };
}
// Scan a source's transcript shards for `query` (alias-aware by default),
-// collecting ALL matched videos up to HARD_VIDEO_CAP so counting is stable, then
+// collecting ALL matched videos up to policy.hardVideoCap so counting is stable, then
// returning the [offset, offset+limit) slice with a `total`/`hasMore`. Plain
// substring by default, or a caller-supplied regex; either can be paged. Set
// `includeSnippets:false` for a cheap worklist (id/title/channel/date/matches,
-// no cue text). Stops early at MAX_PAGES (→ truncated) and HARD_VIDEO_CAP.
+// no cue text). Stops early at MAX_PAGES (→ truncated) and policy.hardVideoCap.
export async function searchTranscripts(
source: ShardSource,
opts: {
@@ -571,11 +468,17 @@ export async function searchTranscripts(
// collapsed copies are named on the row they fold into, never dropped
// silently.
collapseDuplicates?: boolean;
+ // The caps this scan runs under. Defaults to MCP_POLICY — the same four
+ // numbers this scanner used when they were private constants — so an
+ // existing caller reads exactly the same pages and parses exactly the same
+ // bytes. A caller with a different budget passes its own.
+ policy?: SearchPolicy;
},
): Promise<SearchResult> {
+ const policy = opts.policy ?? MCP_POLICY;
const limit = opts.limit ?? 20;
const offset = Math.max(0, opts.offset ?? 0);
- const maxPages = opts.maxPages ?? MAX_PAGES;
+ const maxPages = opts.maxPages ?? policy.maxPages;
const includeSnippets = opts.includeSnippets !== false;
const snippetsPerVideo = opts.snippetsPerVideo ?? 4;
@@ -699,7 +602,7 @@ export async function searchTranscripts(
push({
clock: clock(cue.start),
seconds: cue.start,
- text: truncate(cue.text),
+ text: truncate(cue.text, policy.snippetChars),
});
}
}
@@ -722,7 +625,7 @@ export async function searchTranscripts(
push({
clock: clock(0),
seconds: 0,
- text: truncate(rec.description),
+ text: truncate(rec.description, policy.snippetChars),
scope: "description",
});
}
@@ -730,7 +633,12 @@ export async function searchTranscripts(
const tags = (rec.tags ?? []).join(", ");
if (tags && match(tags)) {
otherHit = true;
- push({ clock: clock(0), seconds: 0, text: truncate(tags), scope: "tags" });
+ push({
+ clock: clock(0),
+ seconds: 0,
+ text: truncate(tags, policy.snippetChars),
+ scope: "tags",
+ });
}
}
if (wantChat && chatCuesFor) {
@@ -740,7 +648,7 @@ export async function searchTranscripts(
push({
clock: clock(cue.start),
seconds: cue.start,
- text: truncate(cue.text),
+ text: truncate(cue.text, policy.snippetChars),
scope: "chat",
});
}
@@ -761,7 +669,7 @@ export async function searchTranscripts(
matches: matches || 1,
snippets,
});
- if (all.length >= HARD_VIDEO_CAP) {
+ if (all.length >= policy.hardVideoCap) {
truncated = true;
channelStopped = {
channel: ch.name,
@@ -828,7 +736,7 @@ export async function searchTranscripts(
? [{ clock: "", seconds: 0, text: truncate(post.text, 480) }]
: [],
});
- if (all.length >= HARD_VIDEO_CAP) {
+ if (all.length >= policy.hardVideoCap) {
truncated = true;
break postsOuter;
}
@@ -961,11 +869,7 @@ export async function getThread(
if ((p.threadId || p.id) === threadId) thread.push(p);
}
}
- thread.sort((a, b) =>
- a.createdAt === b.createdAt
- ? a.id.localeCompare(b.id)
- : a.createdAt.localeCompare(b.createdAt),
- );
+ rankThread(thread);
return thread.length > 0 ? thread : [post];
}
@@ -982,33 +886,15 @@ export function getWindowedTranscript(
// the line's clock + start seconds. Bare clock when omitted.
stamp?: (clock: string, seconds: number) => string;
// Cap on merged excerpt lines emitted for this video (default
- // WINDOW_LINE_CAP). The earliest lines are kept; `maxCues` above stays the
- // per-window bound.
+ // MCP_POLICY.windowLineCap). The earliest lines are kept; `maxCues` above
+ // stays the per-window bound.
maxLines?: number;
} = {},
): { lines: string[]; matchCount: number } {
- const cues = record.cues ?? [];
- const timestamps = opts.timestamps !== false;
- let merged: WindowSnippet[] = [];
- let matchCount = 0;
- for (const cue of cues) {
- if (!matcher(cue.text)) continue;
- matchCount++;
- const win = cuesToSnippets(
- windowCues(cues, cue.start, {
- before: opts.before,
- after: opts.after,
- maxCues: opts.maxCues,
- }),
- );
- merged = mergeSnippets(merged, win, opts.maxLines ?? WINDOW_LINE_CAP);
- }
- const lines = merged.map((s) => {
- if (!timestamps) return s.text;
- const stamp = opts.stamp ? opts.stamp(s.clock, s.seconds) : s.clock;
- return `[${stamp}] ${s.text}`;
+ return windowedTranscript(record.cues ?? [], matcher, {
+ ...opts,
+ maxLines: opts.maxLines ?? MCP_POLICY.windowLineCap,
});
- return { lines, matchCount };
}
// Locate a single video across the source's channels via each channel's
@@ -1067,26 +953,10 @@ export async function findVideo(
// the simpler `searchTranscripts` scanner above (a trivial single-leaf query),
// so today's callers/tests are unaffected.
//
-// It mirrors the browser's tree algebra (searchEval.ts) and filter predicate
-// (SearchSessionContext.tsx `passesFilter`) — evaluated per record to a boolean
-// instead of over slug sets — and the per-scope text extraction of
-// searchPipeline.ts.
-
-// The positive share-filter selection (parseShareV1), as a predicate input.
-export type SearchFilters = {
- // ft — video / livestream types kept.
- videos: boolean;
- livestreams: boolean;
- // fa — all-ages / age-restricted kept.
- allAges: boolean;
- restricted: boolean;
- // fav — the VideoState values kept (see common/lib/availability). Absent
- // from the set means filtered out.
- states: ReadonlySet<VideoState>;
- // fdf / fdt — inclusive upload-date bounds, "YYYYMMDD".
- dateFrom?: string;
- dateTo?: string;
-};
+// The tree algebra, the leaf evaluation and the filter predicate are
+// `lib/search/evalTree.ts` — the same module the viewer's `lib/searchEval.ts`
+// points at for the meaning of AND / OR / negate. What stays here is the walk:
+// which pages to read, in what order, and when to stop.
export type SearchSpec = {
// The root of the composite query (a `qt=` tree, or a synthesized single leaf
@@ -1102,17 +972,6 @@ export type SearchSpec = {
useAliases?: boolean;
};
-// A snippet tagged with the scope it came from, carrying the seconds needed to
-// build a moment link. `seconds` is 0 for the non-timed scopes (metadata /
-// description / tags).
-export type ScopedSnippet = {
- scope: LayerScope;
- track?: string;
- clock: string;
- seconds: number;
- text: string;
-};
-
export type SpecHit = {
videoId: string;
slug: string;
@@ -1139,226 +998,6 @@ export type SpecResult = {
truncated: boolean;
};
-type LeafMatcher = { scope: LayerScope; test: Matcher };
-
-// Compile a matcher per active leaf. Transcripts leaves are alias-aware (unless
-// they're regex); every other scope matches plain-substring / regex only. Fired
-// aliases are unioned for the caller to report.
-function buildLeafMatchers(
- root: QueryNode,
- aliases: SearchAlias[],
- useAliases: boolean,
-): { matchers: Map<string, LeafMatcher>; fired: SearchAlias[] } {
- const matchers = new Map<string, LeafMatcher>();
- const fired: SearchAlias[] = [];
- forEachLeaf(root, (leaf) => {
- if (!isLeafActive(leaf)) return;
- const aliasAware =
- leaf.scope === "transcripts" && !leaf.useRegex && useAliases;
- const built = buildMatcher({
- query: leaf.query,
- regex: leaf.useRegex,
- useAliases: aliasAware,
- aliases: aliasAware ? aliases : [],
- });
- matchers.set(leaf.id, { scope: leaf.scope, test: built.match });
- for (const a of built.firedAliases) {
- if (!fired.some((x) => x.id === a.id)) fired.push(a);
- }
- });
- return { matchers, fired };
-}
-
-// Per-record context the tree evaluates against.
-type RecordCtx = {
- title: string;
- channel: string;
- description: string;
- tags: string;
- cues: Cue[];
- chatCues: Cue[];
- // The post body, when this record IS a post rather than a video. Empty for a
- // video record, so a posts-scope leaf never matches one.
- postText?: string;
- snippetsPerVideo: number;
- includeSnippets: boolean;
-};
-
-type LeafOutcome = { matched: boolean; count: number; hits: ScopedSnippet[] };
-
-function evalLeaf(leaf: QueryNode, m: LeafMatcher, ctx: RecordCtx): LeafOutcome {
- if (!isLeaf(leaf)) return { matched: false, count: 0, hits: [] };
- const hits: ScopedSnippet[] = [];
- const push = (s: ScopedSnippet): void => {
- if (ctx.includeSnippets && hits.length < ctx.snippetsPerVideo) hits.push(s);
- };
- let count = 0;
- switch (m.scope) {
- case "transcripts":
- for (const cue of ctx.cues) {
- if (!m.test(cue.text)) continue;
- count++;
- push({
- scope: "transcripts",
- clock: clock(cue.start),
- seconds: cue.start,
- text: truncate(cue.text),
- });
- }
- break;
- case "chat":
- for (const cue of ctx.chatCues) {
- if (!m.test(cue.text)) continue;
- count++;
- push({
- scope: "chat",
- track: "live_chat",
- clock: clock(cue.start),
- seconds: cue.start,
- text: truncate(cue.text),
- });
- }
- break;
- case "posts":
- // A post has no timeline: one hit, seconds 0 — the same convention the
- // metadata / description / tags scopes already use.
- if (ctx.postText && m.test(ctx.postText)) {
- count++;
- push({
- scope: "posts",
- clock: clock(0),
- seconds: 0,
- text: truncate(ctx.postText),
- });
- }
- break;
- case "metadata": {
- const titleHit = m.test(ctx.title);
- const channelHit = m.test(ctx.channel);
- if (titleHit || channelHit) {
- count++;
- if (titleHit) {
- push({ scope: "metadata", clock: clock(0), seconds: 0, text: truncate(ctx.title) });
- } else {
- push({ scope: "metadata", clock: clock(0), seconds: 0, text: `Channel: ${ctx.channel}` });
- }
- }
- break;
- }
- case "description":
- if (ctx.description && m.test(ctx.description)) {
- count++;
- push({ scope: "description", clock: clock(0), seconds: 0, text: truncate(ctx.description) });
- }
- break;
- case "tags":
- if (ctx.tags && m.test(ctx.tags)) {
- count++;
- push({ scope: "tags", clock: clock(0), seconds: 0, text: truncate(ctx.tags) });
- }
- break;
- }
- return { matched: count > 0, count, hits };
-}
-
-type NodeOutcome = { match: boolean; count: number; hits: ScopedSnippet[] };
-
-// Evaluate the tree against one record — mirrors searchEval's AND/OR/negate,
-// per record. A negated node contributes no hits (like the browser's `diff`).
-// An inactive (empty) subtree is identity (matches, no hits).
-function evalNode(
- node: QueryNode,
- matchers: Map<string, LeafMatcher>,
- ctx: RecordCtx,
-): NodeOutcome {
- if (!isNodeActive(node)) return { match: true, count: 0, hits: [] };
- if (isLeaf(node)) {
- const m = matchers.get(node.id);
- if (!m) return { match: true, count: 0, hits: [] };
- const r = evalLeaf(node, m, ctx);
- if (node.negate) return { match: !r.matched, count: 0, hits: [] };
- return {
- match: r.matched,
- count: node.contributeHits ? r.count : 0,
- hits: node.contributeHits ? r.hits : [],
- };
- }
- const active = node.children.filter(isNodeActive);
- if (active.length === 0) return { match: true, count: 0, hits: [] };
- if (node.op === "AND") {
- let allMatch = true;
- let count = 0;
- const hits: ScopedSnippet[] = [];
- for (const c of active) {
- const r = evalNode(c, matchers, ctx);
- if (!r.match) {
- allMatch = false;
- break;
- }
- count += r.count;
- hits.push(...r.hits);
- }
- const match = node.negate ? !allMatch : allMatch;
- return match && !node.negate
- ? { match, count, hits }
- : { match, count: 0, hits: [] };
- }
- // OR
- let any = false;
- let count = 0;
- const hits: ScopedSnippet[] = [];
- for (const c of active) {
- const r = evalNode(c, matchers, ctx);
- if (r.match) {
- any = true;
- count += r.count;
- hits.push(...r.hits);
- }
- }
- const match = node.negate ? !any : any;
- return match && !node.negate
- ? { match, count, hits }
- : { match, count: 0, hits: [] };
-}
-
-// The share-filter predicate over a transcript record + its availability.
-// Mirrors SearchSessionContext.tsx `passesFilter` exactly.
-//
-// Typed on the three fields it actually reads rather than on TranscriptDetail,
-// so the SAME predicate can be applied to a summaries index record while
-// planning a scan and to the full transcript record while deciding a hit.
-// There is deliberately only one of these: a second, "cheap" predicate for
-// planning is exactly how a pruner starts silently disagreeing with the
-// scanner about what matches.
-type FilterableRecord = {
- isLivestream?: boolean;
- ageRestricted?: boolean;
- uploadDate: string;
-};
-
-function passesFilters(
- rec: FilterableRecord,
- f: SearchFilters,
- avail: VideoAvailability | undefined,
-): boolean {
- // ft — type
- if (rec.isLivestream ? !f.livestreams : !f.videos) return false;
- // fa — audience
- if (rec.ageRestricted ? !f.restricted : !f.allAges) return false;
- // fav — presence on the source platform
- if (!f.states.has(avail?.state ?? "available")) return false;
- // fdf / fdt — upload-date range (lexicographic on YYYYMMDD)
- if (f.dateFrom && rec.uploadDate < f.dateFrom) return false;
- if (f.dateTo && rec.uploadDate > f.dateTo) return false;
- return true;
-}
-
-// True when the fav filter could exclude something (so availability must be
-// fetched). If every availability bucket is kept, there's nothing to look up.
-function needsAvailability(f: SearchFilters | null | undefined): boolean {
- return !!f && !VIDEO_STATES.every((s) => f.states.has(s));
-}
-
// A small lazy live-chat fetcher: per-channel subs manifest + page caches, so a
// chat-scope leaf only pulls the subs shards it actually touches.
function makeChatFetcher(source: ShardSource) {
@@ -1399,11 +1038,13 @@ export async function runSearchSpec(
includeSnippets?: boolean;
maxPages?: number;
snippetsPerVideo?: number;
+ policy?: SearchPolicy;
} = {},
): Promise<SpecResult> {
+ const policy = opts.policy ?? MCP_POLICY;
const limit = opts.limit ?? 20;
const offset = Math.max(0, opts.offset ?? 0);
- const maxPages = opts.maxPages ?? MAX_PAGES;
+ const maxPages = opts.maxPages ?? policy.maxPages;
const includeSnippets = opts.includeSnippets !== false;
const snippetsPerVideo = opts.snippetsPerVideo ?? 4;
const filters = spec.filters ?? null;
@@ -1460,6 +1101,7 @@ export async function runSearchSpec(
chatCues: chatCuesFor ? await chatCuesFor(ch, rec) : [],
snippetsPerVideo,
includeSnippets,
+ snippetChars: policy.snippetChars,
};
const r = evalNode(spec.tree, matchers, ctx);
if (!r.match) continue;
@@ -1477,7 +1119,7 @@ export async function runSearchSpec(
matches: r.count || 1,
snippets: r.hits,
});
- if (all.length >= HARD_VIDEO_CAP) {
+ if (all.length >= policy.hardVideoCap) {
truncated = true;
break outer;
}
@@ -1523,6 +1165,7 @@ export async function runSearchSpec(
postText: post.text,
snippetsPerVideo,
includeSnippets,
+ snippetChars: policy.snippetChars,
};
const r = evalNode(spec.tree, matchers, ctx);
if (!r.match) continue;
@@ -1540,7 +1183,7 @@ export async function runSearchSpec(
matches: r.count || 1,
snippets: r.hits,
});
- if (all.length >= HARD_VIDEO_CAP) {
+ if (all.length >= policy.hardVideoCap) {
truncated = true;
break postsOuter;
}