Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 17d4399f8e969b0b29708c6b1e7c0a0538196435
parent 98ff443526f3b1993450a70cdf80239abf2029f2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  6 Jul 2026 23:47:09 -0400

Fix /ask retrieval: drive the real search engine instead of a crude shard scan

The BYO-AI chat returned the same ~12 videos for every question: retrieve()
hand-scanned transcript shards, substring-matched any 3-char word, collected
videos in shard order, and broke at the first 12 matches with no ranking — so
the earliest videos on page 0 matched nearly everything. Scanning harder isn't
viable (pages are ~8MB each, 400-800MB per large channel).

Reuse the site's own search engine (runQueryTree, behind on-site search):
- askRetrieval.ts: tokenize the question -> keywords, build an OR of
  transcript+metadata leaves, run with a bounded hit cap, then rank videos by
  distinct-keyword coverage -> hit density -> recency. Snippets/timestamps come
  straight from the engine's hits; shares the IndexedDB page/result cache and is
  page-deduped, so downloads stay bounded.
- ask/page.tsx + new AskHub.tsx: wrap AskChat in SingleSiteDataProvider (site)
  or MultiSiteDataProvider (hub) so retrieval works and federates across member
  origins for free.
- AskChat.tsx: read summaries from useSearchData(), gate on summariesReady,
  drop the old corpus.json fetch.
- Add askRetrieval unit tests (keyword extraction + ranking); update changelog.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

Diffstat:
Mexport/CHANGELOG.md | 1+
Mexport/app/ask/AskChat.tsx | 39++++++++++++++++-----------------------
Aexport/app/ask/AskHub.tsx | 32++++++++++++++++++++++++++++++++
Mexport/app/ask/page.tsx | 13+++++++++++--
Aexport/app/lib/askRetrieval.test.ts | 154+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/app/lib/askRetrieval.ts | 360++++++++++++++++++++++++++++++++++++++++++++++++-------------------------------
6 files changed, 431 insertions(+), 168 deletions(-)

diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **The "Ask AI" chat now finds the right videos.** The `/ask` retrieval previously skimmed the first few transcript shards and cut off early, so it kept answering from the same handful of videos no matter what you asked. It now uses the **site's own transcript search engine** (the one behind the search box): your question's keywords rank matches across the whole archive, snippets carry real timestamps, and it reuses the cached index — so different questions surface different, relevant videos, and on the federated hub it searches every member site. - **Search now suggests a better query when you type a known term.** When a search term matches a curated alias — e.g. typing `loli`, `lolly`, or `loly` — a quiet chip appears under the box offering a more robust regex like `\blol(i|ly)`. Click **Apply** to swap it in (and switch that layer to regex mode) or **Dismiss** to ignore it; a new term re-offers. It's never forced, matching is whole-word (so `lolight` won't trigger it), and it's suppressed while you're already writing a regex. The dictionary is authored in the editor's new Search aliases page and shipped per site as `/search-aliases.json` (the global list merged with per-site overrides). See `common/components/QueryLeafView.tsx`, `common/components/{aliasesCache,SearchDataContext}.tsx`, `common/lib/searchAliases.ts`, and `export/e2e/alias-suggestion.spec.ts`. ## [0.7.0] - 2026-07-06 diff --git a/export/app/ask/AskChat.tsx b/export/app/ask/AskChat.tsx @@ -7,11 +7,10 @@ import { type Provider, type ChatMessage, } from "../lib/askProvider"; +import { useSearchData } from "yt-dlp-transcript-common/components/SearchDataContext"; import { - loadCorpus, retrieve, buildContext, - type CorpusInfo, type RetrievedVideo, } from "../lib/askRetrieval"; @@ -43,8 +42,9 @@ export default function AskChat() { const [remember, setRemember] = useState(true); const [showKey, setShowKey] = useState(false); - const [corpus, setCorpus] = useState<CorpusInfo | null>(null); - const [corpusError, setCorpusError] = useState<string | null>(null); + const { summariesState, channels } = useSearchData(); + const { summaries, summariesReady } = summariesState; + const corpusError = summariesState.error?.message ?? null; const [messages, setMessages] = useState<UiMessage[]>([]); const [input, setInput] = useState(""); const [busy, setBusy] = useState(false); @@ -67,15 +67,6 @@ export default function AskChat() { // eslint-disable-next-line react-hooks/exhaustive-deps }, []); - // Load the channel list once. - useEffect(() => { - const ac = new AbortController(); - loadCorpus(ac.signal) - .then(setCorpus) - .catch((e: Error) => setCorpusError(e.message)); - return () => ac.abort(); - }, []); - useEffect(() => { scrollRef.current?.scrollTo({ top: scrollRef.current.scrollHeight }); }, [messages]); @@ -119,7 +110,7 @@ export default function AskChat() { const question = input.trim(); if (!question || busy) return; if (!apiKey.trim()) return; - if (!corpus) return; + if (!summariesReady) return; persistKey(provider, apiKey, model, remember); @@ -146,7 +137,9 @@ export default function AskChat() { }); try { - const { videos, truncated } = await retrieve(corpus.channels, question, { + const { videos, truncated } = await retrieve({ + question, + summaries, signal: ac.signal, }); patchLast((m) => ({ ...m, sources: videos, truncated })); @@ -185,12 +178,12 @@ export default function AskChat() { setBusy(false); abortRef.current = null; } - }, [input, busy, apiKey, corpus, provider, model, remember, messages]); + }, [input, busy, apiKey, summariesReady, summaries, provider, model, remember, messages]); const stop = () => abortRef.current?.abort(); const info = PROVIDERS[provider]; - const channelCount = corpus?.channels.length ?? 0; + const channelCount = channels.length; return ( <div className="flex flex-col gap-5"> @@ -274,8 +267,8 @@ export default function AskChat() { {corpusError && ( <p className="text-sm text-warning"> - Couldn&apos;t load the corpus index ({corpusError}). This chat needs the - site&apos;s <code className="font-mono">corpus.json</code>. + Couldn&apos;t load the transcript index ({corpusError}). This chat needs + the site&apos;s <code className="font-mono">summaries</code> manifest. </p> )} @@ -350,17 +343,17 @@ export default function AskChat() { }} rows={2} placeholder={ - corpus + summariesReady ? "Ask about the transcripts… (⌘/Ctrl+Enter to send)" - : "Loading corpus…" + : "Loading transcripts…" } - disabled={!corpus} + disabled={!summariesReady} className="w-full resize-y rounded-md border border-border bg-background px-3 py-2 text-sm text-foreground" /> <div className="flex items-center gap-2"> <button type="submit" - disabled={busy || !corpus || !input.trim() || !apiKey.trim()} + disabled={busy || !summariesReady || !input.trim() || !apiKey.trim()} className="rounded-md bg-primary px-4 py-2 text-sm font-medium text-primary-foreground transition-colors hover:bg-brand-strong disabled:opacity-50" > {busy ? "Thinking…" : "Ask"} diff --git a/export/app/ask/AskHub.tsx b/export/app/ask/AskHub.tsx @@ -0,0 +1,32 @@ +"use client"; + +// Hub variant of the /ask chat: wires the federated site registry into the +// multi-origin search data source so AskChat's retrieval (runQueryTree) searches +// across every shelved archive. Mirrors HubHome's MultiSiteDataProvider wiring. + +import { useMemo } from "react"; +import { + MultiSiteDataProvider, + type FederatedSite, +} from "yt-dlp-transcript-common/components/SearchDataContext"; +import { useRegistry } from "yt-dlp-transcript-common/components/siteRegistry"; +import AskChat from "./AskChat"; + +export default function AskHub() { + const { sites } = useRegistry(); + const federated = useMemo<FederatedSite[]>( + () => + sites.map((s) => ({ + origin: s.origin, + siteTitle: s.siteTitle, + accent: s.accent, + })), + [sites], + ); + + return ( + <MultiSiteDataProvider sites={federated}> + <AskChat /> + </MultiSiteDataProvider> + ); +} diff --git a/export/app/ask/page.tsx b/export/app/ask/page.tsx @@ -1,8 +1,10 @@ import type { Metadata } from "next"; import Link from "next/link"; +import { SingleSiteDataProvider } from "yt-dlp-transcript-common/components/SearchDataContext"; import { currentSite } from "../lib/site"; import { instanceMode } from "../lib/mode"; import AskChat from "./AskChat"; +import AskHub from "./AskHub"; export const metadata: Metadata = { title: "Ask AI" }; @@ -11,7 +13,8 @@ export const metadata: Metadata = { title: "Ask AI" }; // visitor's chosen provider. The export stays fully static. export default function AskPage() { const site = currentSite(); - const scope = instanceMode() === "hub" ? "the federation" : "the transcripts"; + const isHub = instanceMode() === "hub"; + const scope = isHub ? "the federation" : "the transcripts"; return ( <div className="mx-auto flex max-w-3xl flex-col gap-6"> @@ -34,7 +37,13 @@ export default function AskPage() { </p> </header> - <AskChat /> + {isHub ? ( + <AskHub /> + ) : ( + <SingleSiteDataProvider> + <AskChat /> + </SingleSiteDataProvider> + )} </div> ); } diff --git a/export/app/lib/askRetrieval.test.ts b/export/app/lib/askRetrieval.test.ts @@ -0,0 +1,154 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import type { DisplaySummary } from "yt-dlp-transcript-common/lib/transcripts"; +import type { LayerHit } from "yt-dlp-transcript-common/components/searchPipeline"; +import type { TreeProgress } from "yt-dlp-transcript-common/lib/searchEval"; +import { extractKeywords, rankResults, buildContext } from "./askRetrieval"; + +function summary(over: Partial<DisplaySummary>): DisplaySummary { + return { + slug: "s", + id: "s", + channelSlug: "chan", + title: "Title", + uploadDate: "20200101", + date: "2020-01-01", + duration: "1:00", + channel: "Channel", + isLivestream: false, + ageRestricted: false, + isDeleted: false, + isUnlisted: false, + platform: "youtube", + webpageUrl: "https://example.com/s", + ...over, + }; +} + +function hit(over: Partial<LayerHit>): LayerHit { + return { leafId: "l1", scope: "transcripts", start: 0, text: "x", ...over }; +} + +function progress( + slugs: string[], + hits: Record<string, LayerHit[]>, + capped = false, +): TreeProgress { + return { + slugs: new Set(slugs), + hits: new Map(Object.entries(hits)), + leafStates: new Map(), + groupStates: new Map(), + done: true, + capped, + }; +} + +test("extractKeywords drops stopwords, dedupes, keeps content terms", () => { + const kw = extractKeywords("What did they say about the travel ban and travel?"); + assert.deepEqual(kw, ["travel", "ban"]); +}); + +test("extractKeywords ignores words shorter than 3 chars", () => { + // "ai"/"ml" (2 chars) are dropped; "models" survives — no phrase fallback. + assert.deepEqual(extractKeywords("Is AI or ML in the models?"), ["models"]); +}); + +test("extractKeywords falls back to the phrase when nothing survives", () => { + // all stopwords -> fall back to the trimmed lowercased phrase + assert.deepEqual(extractKeywords("what about that"), ["what about that"]); +}); + +test("extractKeywords caps the number of terms", () => { + const q = "alpha bravo charlie delta echo foxtrot golf hotel india juliet kilo"; + assert.equal(extractKeywords(q, 4).length, 4); +}); + +test("rankResults orders by distinct-term coverage then hit density", () => { + const summaries = [ + summary({ slug: "a", id: "a", title: "A" }), + summary({ slug: "b", id: "b", title: "B" }), + summary({ slug: "c", id: "c", title: "C" }), + ]; + // a: 2 distinct leaves (highest coverage) + // b: 1 leaf but 3 hits + // c: 1 leaf, 1 hit + const p = progress( + ["a", "b", "c"], + { + a: [hit({ leafId: "l1", start: 10 }), hit({ leafId: "l2", start: 20 })], + b: [ + hit({ leafId: "l1", start: 5 }), + hit({ leafId: "l1", start: 6 }), + hit({ leafId: "l1", start: 7 }), + ], + c: [hit({ leafId: "l1", start: 1 })], + }, + ); + const ranked = rankResults(p, summaries); + assert.deepEqual( + ranked.map((r) => r.videoId), + ["a", "b", "c"], + ); +}); + +test("rankResults ties break toward the more recent upload", () => { + const summaries = [ + summary({ slug: "old", id: "old", uploadDate: "20190101" }), + summary({ slug: "new", id: "new", uploadDate: "20210101" }), + ]; + const p = progress( + ["old", "new"], + { + old: [hit({ leafId: "l1", start: 1 })], + new: [hit({ leafId: "l1", start: 1 })], + }, + ); + assert.deepEqual( + rankResults(p, summaries).map((r) => r.videoId), + ["new", "old"], + ); +}); + +test("rankResults skips slugs missing from summaries and honors the limit", () => { + const summaries = [summary({ slug: "a", id: "a" })]; + const p = progress(["a", "ghost"], { + a: [hit({ leafId: "l1", start: 1 })], + ghost: [hit({ leafId: "l1", start: 1 })], + }); + const ranked = rankResults(p, summaries, { limit: 5 }); + assert.equal(ranked.length, 1); + assert.equal(ranked[0].videoId, "a"); +}); + +test("rankResults picks diverse snippets and formats timestamps", () => { + const summaries = [summary({ slug: "a", id: "a" })]; + const p = progress(["a"], { + a: [ + hit({ leafId: "l1", start: 5, text: " first hit " }), + hit({ leafId: "l2", start: 125, text: "second hit" }), + ], + }); + const [v] = rankResults(p, summaries, { snippetsPerVideo: 2 }); + assert.deepEqual( + v.snippets.map((s) => s.clock), + ["0:05", "2:05"], + ); + assert.equal(v.snippets[0].text, "first hit"); +}); + +test("buildContext numbers videos and indents snippets", () => { + const ctx = buildContext([ + { + key: "a", + videoId: "a", + title: "Travel Bans", + channel: "Rekieta", + uploadDate: "20200101", + url: "https://example.com/a", + snippets: [{ clock: "0:05", seconds: 5, text: "about travel bans" }], + }, + ]); + assert.match(ctx, /\[1\] "Travel Bans" — Rekieta/); + assert.match(ctx, /\[0:05\] about travel bans/); +}); diff --git a/export/app/lib/askRetrieval.ts b/export/app/lib/askRetrieval.ts @@ -1,24 +1,21 @@ import { - transcriptPageFileName, - type ChannelTranscriptsManifest, -} from "yt-dlp-transcript-common/lib/manifest"; -import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; + runQueryTree, + type TreeProgress, +} from "yt-dlp-transcript-common/lib/searchEval"; +import { newGroup, newLeaf } from "yt-dlp-transcript-common/lib/searchQuery"; +import type { LayerHit } from "yt-dlp-transcript-common/components/searchPipeline"; +import type { DisplaySummary } from "yt-dlp-transcript-common/lib/transcripts"; import { formatDuration } from "yt-dlp-transcript-common/lib/format"; -// Browser-side retrieval for the /ask chat. Reads the site's own published -// static shards via the documented corpus.json contract — same for a single -// site (same-origin) or a hub (fanning out over member origins, which serve -// CORS-* shards). No server, no new files. Shards are fetched force-cache so -// the service worker / HTTP cache absorbs repeat questions. - -export type ChannelRef = { - slug: string; - name: string; - base: string; // origin ("" = same-origin); member origin in hub mode - siteTitle?: string; -}; - -export type CorpusInfo = { channels: ChannelRef[]; hub: boolean; title: string }; +// Browser-side retrieval for the /ask chat. Rather than hand-scanning the 8MB +// transcript shards (which forced an early cut-off that returned the same first +// videos for every question), this drives the SAME search engine the on-site +// TranscriptSearch uses: runQueryTree over an OR of the question's keywords. +// That engine fetches transcript pages page-deduped + capped, caches raw pages +// and per-leaf results in IndexedDB (shared with on-site search), and — in hub +// mode, via MultiSiteDataProvider — searches across member origins for free. +// The caller supplies `summaries` from useSearchData(); we never fetch corpus +// files ourselves and add no new build artifacts. export type RetrievedVideo = { key: string; @@ -31,152 +28,229 @@ export type RetrievedVideo = { snippets: { clock: string; seconds: number; text: string }[]; }; -async function getJson<T>(url: string, signal?: AbortSignal): Promise<T> { - const res = await fetch(url, { cache: "force-cache", signal }); - if (!res.ok) throw new Error(`GET ${url} -> ${res.status}`); - return (await res.json()) as T; +// Small English stopword set so common question words ("what", "does", "about") +// don't become search leaves. Kept intentionally short — the ranking step +// tolerates a few weak terms, so we only need to strip the highest-frequency +// glue words. +const STOPWORDS = new Set([ + "the", "and", "for", "are", "was", "were", "with", "that", "this", "they", + "them", "from", "have", "has", "had", "what", "when", "where", "which", "who", + "whom", "why", "how", "does", "did", "done", "your", "you", "our", "his", + "her", "hers", "its", "their", "about", "into", "over", "than", "then", + "there", "these", "those", "some", "any", "all", "can", "could", "would", + "should", "will", "shall", "may", "might", "must", "not", "but", "get", + "got", "say", "said", "says", "tell", "told", "talk", "talked", "video", + "videos", "transcript", "transcripts", "channel", "please", "give", "show", +]); + +// Turn a natural-language question into a short list of distinct content terms. +// Falls back to the whole trimmed phrase as a single term when nothing survives +// (e.g. a question made entirely of stopwords or very short words). +export function extractKeywords(question: string, max = 10): string[] { + const words = question.toLowerCase().match(/[a-z0-9][a-z0-9'’-]*/g) ?? []; + const seen = new Set<string>(); + const kept: string[] = []; + for (const w of words) { + const t = w.replace(/^['’-]+|['’-]+$/g, ""); + if (t.length < 3 || STOPWORDS.has(t) || seen.has(t)) continue; + seen.add(t); + kept.push(t); + if (kept.length >= max) break; + } + if (kept.length === 0) { + const phrase = question.trim().toLowerCase(); + return phrase ? [phrase] : []; + } + return kept; } -type CorpusChannelJson = { slug: string; name?: string }; -type SiteCorpusJson = { - channels?: CorpusChannelJson[]; - site?: { title?: string }; -}; -type HubCorpusJson = { - kind?: string; - hub?: { title?: string }; - sites?: { siteId: string; title: string; url: string }[]; -}; +function clock(seconds: number): string { + const s = Math.max(0, Math.floor(seconds)); + return s === 0 ? "0:00" : formatDuration(s); +} -// Load the channel list from /corpus.json. On a hub, fetch every member's own -// corpus.json and tag each channel with its origin + site title. -export async function loadCorpus(signal?: AbortSignal): Promise<CorpusInfo> { - const root = await getJson<SiteCorpusJson & HubCorpusJson>( - "/corpus.json", - signal, +// Choose up to `n` snippet cues for one video, favouring diversity of matched +// terms (round-robin across the leaves that hit) and de-duplicating by start +// time, then present them in chronological order. +function pickSnippets(hits: LayerHit[], n: number): LayerHit[] { + const byLeaf = new Map<string, LayerHit[]>(); + for (const h of hits) { + const list = byLeaf.get(h.leafId); + if (list) list.push(h); + else byLeaf.set(h.leafId, [h]); + } + const queues = Array.from(byLeaf.values()).map((a) => + a.slice().sort((x, y) => x.start - y.start), ); - if (root.kind === "hub" && Array.isArray(root.sites)) { - const channels: ChannelRef[] = []; - for (const site of root.sites) { - const base = site.url.replace(/\/+$/, ""); - try { - const c = await getJson<SiteCorpusJson>(`${base}/corpus.json`, signal); - for (const ch of c.channels ?? []) { - channels.push({ - slug: ch.slug, - name: ch.name ?? ch.slug, - base, - siteTitle: site.title, - }); - } - } catch { - // skip an unreachable member - } + const seenStart = new Set<number>(); + const out: LayerHit[] = []; + let progressed = true; + while (out.length < n && progressed) { + progressed = false; + for (const q of queues) { + const h = q.shift(); + if (!h) continue; + progressed = true; + if (seenStart.has(h.start)) continue; + seenStart.add(h.start); + out.push(h); + if (out.length >= n) break; } - return { channels, hub: true, title: root.hub?.title ?? "the federation" }; } - const channels: ChannelRef[] = (root.channels ?? []).map((ch) => ({ - slug: ch.slug, - name: ch.name ?? ch.slug, - base: "", - })); - return { channels, hub: false, title: root.site?.title ?? "this archive" }; + out.sort((a, b) => a.start - b.start); + return out; } -function clock(seconds: number): string { - const s = Math.max(0, Math.floor(seconds)); - return s === 0 ? "0:00" : formatDuration(s); +export type RankOptions = { + limit?: number; + snippetsPerVideo?: number; + siteTitleOf?: (summary: DisplaySummary) => string | undefined; +}; + +// Rank the engine's matched videos and shape them for the chat context/citations. +// Score = distinct query-term coverage (dominant) then raw hit density; ties +// break toward the more recent upload. Pure + deterministic so it's unit-testable. +export function rankResults( + progress: TreeProgress, + summaries: DisplaySummary[], + opts: RankOptions = {}, +): RetrievedVideo[] { + const limit = opts.limit ?? 12; + const perVideo = opts.snippetsPerVideo ?? 5; + const bySlug = new Map(summaries.map((s) => [s.slug, s])); + // Collapse a keyword's per-scope leaves ("t#0"/"m#0") to one term ("0"); leaf + // ids without a "#" (e.g. tests, other callers) map to themselves. + const termKey = (leafId: string) => + leafId.includes("#") ? leafId.slice(leafId.indexOf("#") + 1) : leafId; + + const scored: { v: RetrievedVideo; score: number }[] = []; + for (const slug of progress.slugs) { + const summary = bySlug.get(slug); + if (!summary) continue; + const hits = progress.hits.get(slug) ?? []; + const coverage = new Set(hits.map((h) => termKey(h.leafId))).size; + const score = coverage * 1000 + Math.min(hits.length, 500); + const chosen = pickSnippets(hits, perVideo); + scored.push({ + score, + v: { + key: slug, + videoId: summary.id, + title: summary.title, + channel: summary.channel, + siteTitle: opts.siteTitleOf?.(summary), + uploadDate: summary.uploadDate, + url: summary.webpageUrl, + snippets: chosen.map((h) => ({ + clock: clock(h.start), + seconds: h.start, + text: h.text.trim().replace(/\s+/g, " ").slice(0, 240), + })), + }, + }); + } + + scored.sort( + (a, b) => + b.score - a.score || + (b.v.uploadDate || "").localeCompare(a.v.uploadDate || ""), + ); + return scored.slice(0, limit).map((s) => s.v); } -// Scan the channels' transcript shards for the query and return the top matching -// videos with a few cue snippets each. Bounded by `limit` and `maxPages` so a -// question can't fetch an unbounded slice of a large (or hub-wide) corpus. -export async function retrieve( - channels: ChannelRef[], - query: string, - opts: { - limit?: number; - snippetsPerVideo?: number; - maxPages?: number; - signal?: AbortSignal; - } = {}, +export type RetrieveOptions = RankOptions & { + question: string; + summaries: DisplaySummary[]; + signal?: AbortSignal; + // Total hit cap. Bounds how many transcript pages get fetched — the engine + // stops once this many hits accumulate. NOT Infinity (that would scan the + // whole ~800MB corpus). Capped results still work; they just aren't persisted + // to the layer cache. + hitLimit?: number; +}; + +// Run the question through the shared search engine and return ranked videos. +// Browser-only: runQueryTree depends on window timers + fetch. Resolves once the +// engine reports done; rejects with an AbortError if the signal fires first. +export function retrieve( + opts: RetrieveOptions, ): Promise<{ videos: RetrievedVideo[]; truncated: boolean }> { - const limit = opts.limit ?? 12; - const perVideo = opts.snippetsPerVideo ?? 4; - const maxPages = opts.maxPages ?? 60; - const q = query.toLowerCase(); - const terms = q.split(/\s+/).filter((t) => t.length >= 3); - const match = (text: string): boolean => { - const t = text.toLowerCase(); - return terms.length === 0 ? t.includes(q) : terms.some((term) => t.includes(term)); - }; - - const videos: RetrievedVideo[] = []; - let pages = 0; - let truncated = false; - - outer: for (const ch of channels) { - let manifest: ChannelTranscriptsManifest; - try { - manifest = await getJson( - `${ch.base}/transcripts/${ch.slug}/manifest.json`, - opts.signal, - ); - } catch { - continue; + const { question, summaries, signal } = opts; + const keywords = extractKeywords(question); + + return new Promise((resolve, reject) => { + if (signal?.aborted) { + reject(new DOMException("Aborted", "AbortError")); + return; + } + if (summaries.length === 0 || keywords.length === 0) { + resolve({ videos: [], truncated: false }); + return; } - for (let p = 0; p < manifest.pageCount; p++) { - if (pages >= maxPages) { - truncated = true; - break outer; - } - let records: TranscriptDetail[]; - try { - records = await getJson( - `${ch.base}/transcripts/${ch.slug}/${transcriptPageFileName(p)}`, - opts.signal, + + // OR of the keywords. Each term gets a transcript-cue leaf (contributes + // timestamped snippets) plus a fetch-free metadata leaf (title/channel) for + // cheap title recall. contributeHits stays true so both surface citations. + // Leaf ids encode the keyword index after a "#" so ranking can collapse a + // keyword's transcript + metadata leaves into ONE unit of term coverage + // (see termKey() in rankResults) rather than double-counting the two scopes. + const root = newGroup({ + op: "OR", + children: keywords.flatMap((kw, i) => [ + newLeaf({ id: `t#${i}`, query: kw, scope: "transcripts", contributeHits: true }), + newLeaf({ id: `m#${i}`, query: kw, scope: "metadata", contributeHits: true }), + ]), + }); + + let settled = false; + const finish = (fn: () => void) => { + if (settled) return; + settled = true; + fn(); + }; + + const controller = runQueryTree({ + root, + globalScope: summaries.map((s) => s.slug), + summaries, + chatScopeSlugs: null, + initialHitLimit: opts.hitLimit ?? 400, + concurrency: 6, + flushIntervalMs: 120, + emit: (p) => { + if (!p.done) return; + finish(() => + resolve({ + videos: rankResults(p, summaries, opts), + truncated: p.capped, + }), ); - } catch { - continue; - } - pages++; - for (const rec of records) { - const snippets: RetrievedVideo["snippets"] = []; - for (const cue of rec.cues ?? []) { - if (!match(cue.text)) continue; - snippets.push({ - clock: clock(cue.start), - seconds: cue.start, - text: cue.text.trim().replace(/\s+/g, " ").slice(0, 240), - }); - if (snippets.length >= perVideo) break; - } - if (snippets.length === 0) continue; - videos.push({ - key: `${ch.base}|${rec.id}`, - videoId: rec.id, - title: rec.title, - channel: ch.name, - siteTitle: ch.siteTitle, - uploadDate: rec.uploadDate, - url: rec.webpageUrl, - snippets, - }); - if (videos.length >= limit) break outer; - } + }, + }); + + if (signal) { + signal.addEventListener( + "abort", + () => { + controller.cancel(); + finish(() => reject(new DOMException("Aborted", "AbortError"))); + }, + { once: true }, + ); } - } - return { videos, truncated }; + }); } -// Assemble numbered retrieved excerpts into the context block that gets appended -// to the user's question, plus the matching system instruction to cite by index. +// Assemble numbered retrieved excerpts into the context block appended to the +// user's question (unchanged contract — the system prompt cites by [n]). export function buildContext(videos: RetrievedVideo[]): string { if (videos.length === 0) return "(no matching transcript excerpts were found)"; return videos .map((v, i) => { const head = `[${i + 1}] "${v.title}" — ${v.channel}${v.siteTitle ? ` (${v.siteTitle})` : ""}`; - const lines = v.snippets.map((s) => ` [${s.clock}] ${s.text}`).join("\n"); + const lines = v.snippets + .map((s) => ` [${s.clock}] ${s.text}`) + .join("\n"); return `${head}\n${lines}`; }) .join("\n\n");