import { runQueryTree, type TreeProgress, } from "yt-dlp-transcript-common/lib/searchEval"; import { newGroup, newLeaf, type QueryNode, } from "yt-dlp-transcript-common/lib/searchQuery"; import { matchAliases, tokenizeQuery, type SearchAlias, } from "yt-dlp-transcript-common/lib/searchAliases"; import { searchRuntime, type LayerHit, } from "yt-dlp-transcript-common/components/searchPipeline"; import { peekPost } from "yt-dlp-transcript-common/components/postsCache"; import type { DisplaySummary } from "yt-dlp-transcript-common/lib/transcripts"; import { formatDuration } from "yt-dlp-transcript-common/lib/format"; // Browser-side retrieval for the /ask chat. Rather than hand-scanning the 8MB // transcript shards (which forced an early cut-off that returned the same first // videos for every question), this drives the SAME search engine the on-site // TranscriptSearch uses: runQueryTree over an OR of the question's keywords. // That engine fetches transcript pages page-deduped + capped, caches raw pages // and per-leaf results in IndexedDB (shared with on-site search), and — in hub // mode, via MultiSiteDataProvider — searches across member origins for free. // The caller supplies `summaries` from useSearchData(); we never fetch corpus // files ourselves and add no new build artifacts. export type RetrievedVideo = { key: string; videoId: string; title: string; channel: string; siteTitle?: string; uploadDate: string; url?: string; snippets: { clock: string; seconds: number; text: string }[]; // True for an entry from the social-post corpus. A post has no timeline, so // its context lines carry no [clock] and its citations render no seek button. isPost?: boolean; }; // Small English stopword set so common question words ("what", "does", "about") // don't become search leaves. Kept intentionally short — the ranking step // tolerates a few weak terms, so we only need to strip the highest-frequency // glue words. const STOPWORDS = new Set([ "the", "and", "for", "are", "was", "were", "with", "that", "this", "they", "them", "from", "have", "has", "had", "what", "when", "where", "which", "who", "whom", "why", "how", "does", "did", "done", "your", "you", "our", "his", "her", "hers", "its", "their", "about", "into", "over", "than", "then", "there", "these", "those", "some", "any", "all", "can", "could", "would", "should", "will", "shall", "may", "might", "must", "not", "but", "get", "got", "say", "said", "says", "tell", "told", "talk", "talked", "video", "videos", "transcript", "transcripts", "channel", "please", "give", "show", ]); // Turn a natural-language question into a short list of distinct content terms. // Falls back to the whole trimmed phrase as a single term when nothing survives // (e.g. a question made entirely of stopwords or very short words). export function extractKeywords(question: string, max = 10): string[] { const words = question.toLowerCase().match(/[a-z0-9][a-z0-9'’-]*/g) ?? []; const seen = new Set(); const kept: string[] = []; for (const w of words) { const t = w.replace(/^['’-]+|['’-]+$/g, ""); if (t.length < 3 || STOPWORDS.has(t) || seen.has(t)) continue; seen.add(t); kept.push(t); if (kept.length >= max) break; } if (kept.length === 0) { const phrase = question.trim().toLowerCase(); return phrase ? [phrase] : []; } return kept; } function clock(seconds: number): string { const s = Math.max(0, Math.floor(seconds)); return s === 0 ? "0:00" : formatDuration(s); } // Choose up to `n` snippet cues for one video, favouring diversity of matched // terms (round-robin across the leaves that hit) and de-duplicating by start // time, then present them in chronological order. function pickSnippets(hits: LayerHit[], n: number): LayerHit[] { const byLeaf = new Map(); for (const h of hits) { const list = byLeaf.get(h.leafId); if (list) list.push(h); else byLeaf.set(h.leafId, [h]); } const queues = Array.from(byLeaf.values()).map((a) => a.slice().sort((x, y) => x.start - y.start), ); const seenStart = new Set(); const out: LayerHit[] = []; let progressed = true; while (out.length < n && progressed) { progressed = false; for (const q of queues) { const h = q.shift(); if (!h) continue; progressed = true; if (seenStart.has(h.start)) continue; seenStart.add(h.start); out.push(h); if (out.length >= n) break; } } out.sort((a, b) => a.start - b.start); return out; } export type RankOptions = { limit?: number; snippetsPerVideo?: number; siteTitleOf?: (summary: DisplaySummary) => string | undefined; }; // Rank the engine's matched videos and shape them for the chat context/citations. // Score = distinct query-term coverage (dominant) then raw hit density; ties // break toward the more recent upload. Pure + deterministic so it's unit-testable. export function rankResults( progress: TreeProgress, summaries: DisplaySummary[], opts: RankOptions = {}, ): RetrievedVideo[] { const limit = opts.limit ?? 12; const perVideo = opts.snippetsPerVideo ?? 5; const bySlug = new Map(summaries.map((s) => [s.slug, s])); // Collapse a keyword's per-scope leaves ("t#0"/"m#0") to one term ("0"); leaf // ids without a "#" (e.g. tests, other callers) map to themselves. const termKey = (leafId: string) => leafId.includes("#") ? leafId.slice(leafId.indexOf("#") + 1) : leafId; const scored: { v: RetrievedVideo; score: number }[] = []; for (const slug of progress.slugs) { const summary = bySlug.get(slug); if (!summary) { // Not a video — it may be a post (a disjoint slug namespace). Posts are // resolved from the posts page cache the retrieval pass already warmed. const post = peekPost(slug); if (!post) continue; const hits = progress.hits.get(slug) ?? []; const coverage = new Set(hits.map((h) => termKey(h.leafId))).size; scored.push({ score: coverage * 1000 + Math.min(hits.length, 500), v: { key: slug, videoId: post.id, title: post.text.slice(0, 120), channel: post.authorName || post.author, uploadDate: post.uploadDate, url: post.url, isPost: true, // One snippet: the post body. It is already short, and the pipeline // truncates hit text to a ±80-char window. snippets: [ { clock: "", seconds: 0, text: post.text.trim().replace(/\s+/g, " ").slice(0, 480), }, ], }, }); continue; } const hits = progress.hits.get(slug) ?? []; const coverage = new Set(hits.map((h) => termKey(h.leafId))).size; const score = coverage * 1000 + Math.min(hits.length, 500); const chosen = pickSnippets(hits, perVideo); scored.push({ score, v: { key: slug, videoId: summary.id, title: summary.title, channel: summary.channel, siteTitle: opts.siteTitleOf?.(summary), uploadDate: summary.uploadDate, url: summary.webpageUrl, snippets: chosen.map((h) => ({ clock: clock(h.start), seconds: h.start, text: h.text.trim().replace(/\s+/g, " ").slice(0, 240), })), }, }); } scored.sort( (a, b) => b.score - a.score || (b.v.uploadDate || "").localeCompare(a.v.uploadDate || ""), ); return scored.slice(0, limit).map((s) => s.v); } export type RetrieveOptions = RankOptions & { question: string; summaries: DisplaySummary[]; signal?: AbortSignal; // Known search aliases. When a keyword matches an alias trigger, the leaf is // built from the alias's replacement pattern (`suggestion` + `useRegex`) so a // query catches known AI-transcription misspellings (e.g. a name spelled // several ways). Empty/omitted = plain substring leaves, as before. aliases?: SearchAlias[]; // Every post slug available to search (from the posts manifests). Omitted = // no posts corpus, and every posts leaf resolves empty. postScopeSlugs?: ReadonlySet | null; // Total hit cap. Bounds how many transcript pages get fetched — the engine // stops once this many hits accumulate. NOT Infinity (that would scan the // whole ~800MB corpus). Capped results still work; they just aren't persisted // to the layer cache. hitLimit?: number; }; // Build the OR query root for a search. Aliases are matched against the WHOLE // query string (phrase-aware — matchAliases fires on single- OR multi-word // triggers like "graham platner" when their tokens appear in order), and each // fired alias contributes a regex leaf (`suggestion` + `useRegex`) so the search // catches known AI-transcription misspellings (e.g. "Grand Platina"). The plain // per-keyword leaves cover everything else — but a keyword already covered by a // fired alias's trigger is dropped so a broad bare leaf (e.g. "graham") doesn't // dilute ranking. Leaf ids encode the term after "#" so ranking collapses a // term's transcript + metadata leaves into one unit of coverage (see termKey()). // // NOTE: alias firing depends on the *query string* containing the trigger phrase. // The gather prompt steers the model to search the full aliased term (not a // fragment like "graham"), which is what lets a phrase alias fire here. export function buildSearchRoot( query: string, keywords: string[], aliases: SearchAlias[] = [], ): { root: ReturnType; firedT: SearchAlias[]; firedM: SearchAlias[] } { const children: QueryNode[] = []; const firedT = aliases.length ? matchAliases(query, "transcripts", aliases) : []; const firedM = aliases.length ? matchAliases(query, "metadata", aliases) : []; // Tokens covered by any fired alias's trigger — skip plain leaves for these. const covered = new Set(); for (const a of [...firedT, ...firedM]) { for (const t of a.triggers) for (const tok of tokenizeQuery(t)) covered.add(tok); } firedT.forEach((a, i) => children.push( newLeaf({ id: `p#a${i}`, query: a.suggestion, scope: "posts", useRegex: a.useRegex, contributeHits: true, }), ), ); firedT.forEach((a, i) => children.push( newLeaf({ id: `t#a${i}`, query: a.suggestion, scope: "transcripts", useRegex: a.useRegex, contributeHits: true, }), ), ); firedM.forEach((a, i) => children.push( newLeaf({ id: `m#a${i}`, query: a.suggestion, scope: "metadata", useRegex: a.useRegex, contributeHits: true, }), ), ); keywords.forEach((kw, i) => { if (covered.has(kw)) return; children.push(newLeaf({ id: `t#${i}`, query: kw, scope: "transcripts", contributeHits: true })); children.push(newLeaf({ id: `m#${i}`, query: kw, scope: "metadata", contributeHits: true })); // The social-post corpus is searched alongside transcripts, so the chat is // grounded in BOTH datasets. The `p#` id shares the term suffix with // `t#`/`m#`, so rankResults' distinct-term coverage collapses all // three scopes into one unit rather than triple-counting a keyword. children.push(newLeaf({ id: `p#${i}`, query: kw, scope: "posts", contributeHits: true })); }); return { root: newGroup({ op: "OR", children }), firedT, firedM }; } // The distinct aliases that fired for a query across both scopes, deduped by id // (an alias enabled in transcripts + metadata fires in both). This is what // `retrieve` reports as `firedAliases` so the answer prompt can focus on only // the aliases the search actually used. function unionFired(firedT: SearchAlias[], firedM: SearchAlias[]): SearchAlias[] { const byId = new Map(); for (const a of [...firedT, ...firedM]) if (!byId.has(a.id)) byId.set(a.id, a); return [...byId.values()]; } // Run the question through the shared search engine and return ranked videos. // Browser-only: runQueryTree depends on window timers + fetch. Resolves once the // engine reports done; rejects with an AbortError if the signal fires first. export function retrieve( opts: RetrieveOptions, ): Promise<{ videos: RetrievedVideo[]; truncated: boolean; firedAliases: SearchAlias[] }> { const { question, summaries, signal } = opts; const aliases = opts.aliases ?? []; const keywords = extractKeywords(question); return new Promise((resolve, reject) => { if (signal?.aborted) { reject(new DOMException("Aborted", "AbortError")); return; } if (summaries.length === 0 || keywords.length === 0) { resolve({ videos: [], truncated: false, firedAliases: [] }); return; } // OR of the query's alias-regex leaves (phrase-matched on the full query) // and its keyword leaves. Each term gets a transcript-cue leaf (timestamped // snippets) plus a fetch-free metadata leaf (title/channel) for cheap title // recall. contributeHits stays true so both surface citations. const { root, firedT, firedM } = buildSearchRoot(question, keywords, aliases); // The aliases this search actually applied — surfaced so the answer prompt // can tell the model exactly which curated terms may be mis-transcribed in // the excerpts it's about to read. const firedAliases = unionFired(firedT, firedM); let settled = false; const finish = (fn: () => void) => { if (settled) return; settled = true; fn(); }; // The scope universe spans BOTH corpora: video slugs from summaries plus // post slugs from the caller-supplied posts scope. searchEval partitions // them per-leaf so a transcripts leaf never fetches a post and vice versa. const postSlugs = opts.postScopeSlugs ?? null; const controller = runQueryTree({ root, runtime: searchRuntime, globalScope: [ ...summaries.map((s) => s.slug), ...(postSlugs ? Array.from(postSlugs) : []), ], summaries, chatScopeSlugs: null, postScopeSlugs: postSlugs, initialHitLimit: opts.hitLimit ?? 400, concurrency: 6, flushIntervalMs: 120, emit: (p) => { if (!p.done) return; finish(() => resolve({ videos: rankResults(p, summaries, opts), truncated: p.capped, firedAliases, }), ); }, }); if (signal) { signal.addEventListener( "abort", () => { controller.cancel(); finish(() => reject(new DOMException("Aborted", "AbortError"))); }, { once: true }, ); } }); } // Assemble numbered retrieved excerpts into the context block appended to the // user's question (the system prompt cites by [n]). By default each video is // numbered 1-based by its position (the chat/grounded path). `numberOf` overrides // this so a report write can number by GLOBAL registry index — making a report's // `[n]` stable across every batch/turn folded into it, not just within one batch. export function buildContext( videos: RetrievedVideo[], numberOf?: (video: RetrievedVideo, index: number) => number, ): string { if (videos.length === 0) return "(no matching transcript excerpts were found)"; return videos .map((v, i) => { const n = numberOf ? numberOf(v, i) : i + 1; const head = v.isPost ? `[${n}] post by ${v.channel}${v.siteTitle ? ` (${v.siteTitle})` : ""} — ${v.uploadDate}` : `[${n}] "${v.title}" — ${v.channel}${v.siteTitle ? ` (${v.siteTitle})` : ""}`; // A post has no timeline — a "[0:00]" prefix would be noise the model // could mistake for a real timestamp and cite. const lines = v.snippets .map((s) => (v.isPost ? ` ${s.text}` : ` [${s.clock}] ${s.text}`)) .join("\n"); return `${head}\n${lines}`; }) .join("\n\n"); }