// ONE SEARCH PIPELINE — the excerpt layer. // // Every place that turns cues into something a person or an agent reads lives // here, over the primitives in `lib/transcriptWindow.ts` (`windowCues`, // `cuesToSnippets`, `mergeSnippets`). Before this file the MCP server owned // one excerpt shape and `components/searchPipeline.ts` owned another, in two // directories that never referenced each other. // // THEY ARE STILL TWO SHAPES, deliberately, and the module says so out loud: // // windowedTranscript() a ±seconds window around every matched cue, merged // and deduped — the batch read a sweep prompt drives. // Wide context, timestamps, bounded by a line cap. // findHitsInCues() one line per matched cue, widened to the previous // and next cue ONLY when the match straddles a cue // boundary — the result-card row a reader scans. // // A sweep needs the paragraph around a claim; a result card needs the line // that matched and nothing else. Collapsing them would change both outputs, // and the second one is pinned by the export e2e's rendered hit text. What the // move buys is that there is now exactly ONE implementation of each, one // `truncate`, and one `clock` — and the next excerpt shape has an obvious // place to land next to its siblings rather than in whichever app needed it. import { formatDuration } from "../format"; import type { Cue } from "../vtt"; import { windowCues, cuesToSnippets, mergeSnippets, type WindowSnippet, } from "../transcriptWindow"; // A zero-second cue reads as "0:00", not as the empty string formatDuration // returns for a falsy input. export function clock(seconds: number): string { const s = Math.max(0, Math.floor(seconds)); return s === 0 ? "0:00" : formatDuration(s); } // Collapse whitespace and clip to `max` characters, ellipsis included in the // budget. // // `max` is REQUIRED, and that is the point. It used to default to // `MCP_POLICY.snippetChars`, which meant a viewer adopter that simply forgot to // pass its policy would silently clip every excerpt at the MCP's 240 characters // with every test still green — the failure mode where nothing is broken, the // output is just quietly wrong. A shared module does not get to hold one // caller's budget as its default. Pass `policy.snippetChars`, or a literal when // the width is genuinely not the policy's to set (a post's 120-character // stand-in title). export function truncate(text: string, max: number): string { const t = text.trim().replace(/\s+/g, " "); return t.length > max ? t.slice(0, max - 1) + "…" : t; } export type Matcher = (text: string) => boolean; // Render one record's transcript around the cues that match `matcher`: for each // matched cue take a bounded window of surrounding cues (±`before`/`after` // seconds), merge overlapping windows deduped by timestamp (capped by // `maxLines`), and return timestamped excerpt lines plus the total match count. export function windowedTranscript( cues: readonly Cue[], matcher: Matcher, opts: { before?: number; after?: number; maxCues?: number; timestamps?: boolean; // Optional formatter for the bracketed stamp's CONTENTS (e.g. an inline // Markdown link "[m:ss](url)" or the compact base form "m:ss|156"), given // the line's clock + start seconds. Bare clock when omitted. stamp?: (clock: string, seconds: number) => string; // Cap on merged excerpt lines emitted for this video (`policy.windowLineCap`). // REQUIRED, for the same reason `truncate`'s `max` is. `maxCues` above // stays the per-window bound. maxLines: number; }, ): { lines: string[]; matchCount: number } { const list = cues as Cue[]; const timestamps = opts.timestamps !== false; let merged: WindowSnippet[] = []; let matchCount = 0; for (const cue of list) { if (!matcher(cue.text)) continue; matchCount++; const win = cuesToSnippets( windowCues(list, cue.start, { before: opts.before, after: opts.after, maxCues: opts.maxCues, }), ); merged = mergeSnippets(merged, win, opts.maxLines); } const lines = merged.map((s) => { if (!timestamps) return s.text; const stamp = opts.stamp ? opts.stamp(s.clock, s.seconds) : s.clock; return `[${stamp}] ${s.text}`; }); return { lines, matchCount }; } // ─── The result-row excerpt ─── // // One hit per matched cue. The match is looked for in a three-cue window // (previous + current + next) but only ACCEPTED when it starts inside the // current cue, so a phrase split across a caption break is found exactly once // — on the cue it starts in — and the emitted text widens to the window only // in that case. Everything else emits the matched cue verbatim. // `track`: a transcript hit from one of the record's ALTERNATE English tracks // (lib/captionTracks.ts, hitsAcrossTracks). Absent for a hit in the primary. export type Hit = { start: number; text: string; track?: string }; export type SubsHit = { track: string; start: number; text: string }; export function findHitsInCues( cues: readonly { start: number; text: string }[], query: string, useRegex: boolean, regex: RegExp | null, limit: number, ): Hit[] { const hits: Hit[] = []; for (let i = 0; i < cues.length && hits.length < limit; i++) { const cur = cues[i]; const prevText = i > 0 ? cues[i - 1].text : ""; const nextText = i < cues.length - 1 ? cues[i + 1].text : ""; const sep1 = prevText ? " " : ""; const sep2 = nextText ? " " : ""; const windowText = prevText + sep1 + cur.text + sep2 + nextText; const curStart = prevText.length + sep1.length; const curEnd = curStart + cur.text.length; const m = findFirstMatchInRange( windowText, curStart, curEnd, query, useRegex, regex, ); if (!m) continue; const crosses = m.idx < curStart || m.idx + m.length > curEnd; hits.push({ start: Math.round(cur.start), text: crosses ? windowText : cur.text, }); } return hits; } // Single-document text match (description / tags / post body): emit at most one // hit with a ±PAD window around the first match, so the result row shows // context without shipping a whole document into a card. export function findHitsInText( text: string, query: string, useRegex: boolean, regex: RegExp | null, ): Hit[] { if (!text) return []; const m = findFirstMatchInRange(text, 0, text.length, query, useRegex, regex); if (!m) return []; const PAD = 80; const from = Math.max(0, m.idx - PAD); const to = Math.min(text.length, m.idx + m.length + PAD); const snippet = (from > 0 ? "…" : "") + text.slice(from, to).replace(/\s+/g, " ").trim() + (to < text.length ? "…" : ""); return [{ start: 0, text: snippet }]; } export function findFirstMatchInRange( haystack: string, rangeStart: number, rangeEnd: number, query: string, useRegex: boolean, regex: RegExp | null, ): { idx: number; length: number } | null { if (useRegex) { if (!regex) return null; const flags = regex.flags.includes("g") ? regex.flags : regex.flags + "g"; const re = new RegExp(regex.source, flags); let m: RegExpExecArray | null; while ((m = re.exec(haystack)) !== null) { if (m.index >= rangeStart && m.index < rangeEnd) { return { idx: m.index, length: m[0].length }; } if (m.index >= rangeEnd) return null; if (m[0].length === 0) re.lastIndex++; } return null; } if (!query) return null; const lower = haystack.toLowerCase(); const ql = query.toLowerCase(); let from = 0; while (from <= haystack.length) { const idx = lower.indexOf(ql, from); if (idx === -1) return null; if (idx >= rangeStart && idx < rangeEnd) return { idx, length: ql.length }; if (idx >= rangeEnd) return null; from = idx + 1; } return null; }