Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 64f0a6b2ea8d9b79580df8ed14d8b69b84e4870d
parent 89b8136e0ec9de9bfb911c82a111d43c21aecb02
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu,  9 Jul 2026 00:21:37 -0400

Ask chat: drill into a hit — read the surrounding transcript on demand

Pinned search results were fixed to their ~240-char matched snippets, too
thin to answer "why did they say that / what surrounded it." Add on-demand
depth two ways, both sharing one windowing mechanism (fetch transcript →
window cues around a moment → merge as snippets into the grounding a video
actually feeds the answer):

- User-driven "Load context" in the pinned panel — deterministic, no AI call,
  works in strict AND expand mode on every provider.
- AI fetch_context tool — expand mode + native transports only; the model
  pulls context itself when a snippet is too thin. Scripted-fallback providers
  degrade gracefully (no tool; user-expand covers them).

The answer phase reads only the grounding block, so fetched context is merged
back into the video's snippets to survive into the answer + citations. Windows
are bounded (±45s, ≤60 cues, ≤30 excerpts/video) so multi-hour transcripts are
never dumped; enriched excerpts persist with the conversation. Pagination was
considered and skipped (ranking front-loads relevance; expand-mode re-search
covers "more").

New common/lib/transcriptWindow.ts (windowCues/cuesToSnippets/mergeSnippets,
reused both sides). fetch_context registered conditionally across the three
native providers with generalized parsers. +7 unit (transcriptWindow, parsers),
+3 ask-chat e2e (AI tool, user-expand w/ reload, scripted degrade).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

Diffstat:
Mcommon/lib/aiHandoff.ts | 2+-
Acommon/lib/transcriptWindow.test.ts | 84+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/transcriptWindow.ts | 77+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/CHANGELOG.md | 1+
Mexport/app/ask/AskChat.tsx | 2++
Mexport/app/ask/MessageBubble.tsx | 16+++++++++++-----
Mexport/app/ask/PinnedResultsPanel.tsx | 77++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------
Mexport/app/ask/PipelineStatus.tsx | 43+++++++++++++++++++++++++------------------
Mexport/app/ask/useAskChat.ts | 80+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/app/lib/askConversation.ts | 20+++++++++++++++-----
Mexport/app/lib/nativeTools/anthropic.ts | 108+++++++++++++++++++++++++++++++++++++++++++++++++++----------------------------
Mexport/app/lib/nativeTools/gemini.ts | 122+++++++++++++++++++++++++++++++++++++++++++++++++++++++------------------------
Mexport/app/lib/nativeTools/nativeTools.test.ts | 12++++++------
Mexport/app/lib/nativeTools/openai.ts | 98+++++++++++++++++++++++++++++++++++++++++++++++++++++++------------------------
Aexport/app/lib/nativeTools/parsers.test.ts | 90+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/app/lib/nativeTools/shared.ts | 43+++++++++++++++++++++++++++++++++++++++++++
Mexport/app/lib/searchAgent.ts | 80++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mexport/e2e/ask-chat.spec.ts | 207++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
18 files changed, 1011 insertions(+), 151 deletions(-)

diff --git a/common/lib/aiHandoff.ts b/common/lib/aiHandoff.ts @@ -52,7 +52,7 @@ export type HandoffSummaryRef = { siteTitle?: string; }; -function hms(s: number): string { +export function hms(s: number): string { const n = Math.max(0, Math.floor(s)); const h = Math.floor(n / 3600); const m = Math.floor((n % 3600) / 60); diff --git a/common/lib/transcriptWindow.test.ts b/common/lib/transcriptWindow.test.ts @@ -0,0 +1,84 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { windowCues, cuesToSnippets, mergeSnippets } from "./transcriptWindow"; +import type { Cue } from "./vtt"; + +function cue(start: number, text = `t${start}`): Cue { + return { start, end: start + 2, text }; +} + +test("windowCues keeps cues within [center-before, center+after]", () => { + const cues = [cue(0), cue(30), cue(60), cue(90), cue(120)]; + const win = windowCues(cues, 60, { before: 45, after: 45 }); + assert.deepEqual( + win.map((c) => c.start), + [30, 60, 90], + ); +}); + +test("windowCues handles a center at 0s", () => { + const cues = [cue(0), cue(10), cue(50), cue(100)]; + const win = windowCues(cues, 0, { before: 45, after: 45 }); + assert.deepEqual( + win.map((c) => c.start), + [0, 10], + ); +}); + +test("windowCues caps to the maxCues closest, restoring chronological order", () => { + const cues = Array.from({ length: 20 }, (_, i) => cue(i * 5)); // 0,5,…,95 + const win = windowCues(cues, 50, { before: 100, after: 100, maxCues: 3 }); + // Closest to 50 are 50,45,55 → returned sorted. + assert.deepEqual( + win.map((c) => c.start), + [45, 50, 55], + ); +}); + +test("cuesToSnippets formats clock, collapses whitespace, caps to 240 chars", () => { + const long = "a ".repeat(200); // 400 chars pre-slice + const snips = cuesToSnippets([cue(65, "hello world"), cue(0, " "), cue(3600, long)]); + // Empty cue dropped. + assert.equal(snips.length, 2); + assert.deepEqual(snips[0], { clock: "1:05", seconds: 65, text: "hello world" }); + assert.equal(snips[1].clock, "1:00:00"); + assert.ok(snips[1].text.length <= 240); +}); + +test("mergeSnippets dedupes by seconds, keeps existing, sorts, caps", () => { + const existing = [ + { seconds: 100, text: "hit-a" }, + { seconds: 20, text: "hit-b" }, + ]; + const incoming = [ + { seconds: 20, text: "dupe-should-not-replace" }, + { seconds: 10, text: "ctx-1" }, + { seconds: 30, text: "ctx-2" }, + ]; + const merged = mergeSnippets(existing, incoming, 10); + assert.deepEqual( + merged.map((s) => s.seconds), + [10, 20, 30, 100], + ); + // Existing snippet at 20 wins over the incoming duplicate. + assert.equal(merged.find((s) => s.seconds === 20)?.text, "hit-b"); +}); + +test("mergeSnippets never drops existing hits and stops adding at cap", () => { + const existing = [ + { seconds: 1, text: "a" }, + { seconds: 2, text: "b" }, + ]; + const incoming = [ + { seconds: 3, text: "c" }, + { seconds: 4, text: "d" }, + { seconds: 5, text: "e" }, + ]; + const merged = mergeSnippets(existing, incoming, 3); + // Both existing kept; only one incoming added to reach cap 3. + assert.equal(merged.length, 3); + assert.deepEqual( + merged.map((s) => s.seconds), + [1, 2, 3], + ); +}); diff --git a/common/lib/transcriptWindow.ts b/common/lib/transcriptWindow.ts @@ -0,0 +1,77 @@ +import type { Cue } from "./vtt"; +import { hms } from "./aiHandoff"; + +// Turning a hit into its surrounding transcript. Both the /ask chat's user-driven +// "expand this hit" and the AI's fetch_context tool share these helpers: fetch a +// full transcript (client-side, IDB-cached), take a bounded window of cues around +// a timestamp, and merge them into a video's snippets (the citable excerpts). All +// pure so they unit-test without I/O. + +// Same shape as HandoffSnippet / RetrievedVideo's snippet, so a windowed slice +// drops straight into the grounding. +export type WindowSnippet = { clock: string; seconds: number; text: string }; + +// Match askRetrieval's per-snippet slice so windowed lines stay the same size as +// search hits. +const SNIPPET_MAX_CHARS = 240; + +// Cues whose start falls within [center-before, center+after] seconds. When more +// than `maxCues` qualify, keep the `maxCues` closest to the center (a token guard +// for very dense stretches) but return them back in chronological order. +export function windowCues( + cues: Cue[], + centerSeconds: number, + opts: { before?: number; after?: number; maxCues?: number } = {}, +): Cue[] { + const before = opts.before ?? 45; + const after = opts.after ?? 45; + const maxCues = opts.maxCues ?? 60; + const lo = centerSeconds - before; + const hi = centerSeconds + after; + const inWindow = cues.filter((c) => c.start >= lo && c.start <= hi); + if (inWindow.length <= maxCues) return inWindow; + return inWindow + .slice() + .sort( + (a, b) => + Math.abs(a.start - centerSeconds) - Math.abs(b.start - centerSeconds), + ) + .slice(0, maxCues) + .sort((a, b) => a.start - b.start); +} + +// Render cues as snippet objects (clock + seconds + collapsed, capped text). +// Empty cues are dropped. +export function cuesToSnippets(cues: Cue[]): WindowSnippet[] { + const out: WindowSnippet[] = []; + for (const c of cues) { + const text = c.text.trim().replace(/\s+/g, " ").slice(0, SNIPPET_MAX_CHARS); + if (!text) continue; + out.push({ + clock: hms(c.start), + seconds: Math.max(0, Math.floor(c.start)), + text, + }); + } + return out; +} + +// Merge freshly-windowed snippets into a video's existing ones: keep every +// existing snippet (the original hits are the most relevant), append incoming +// ones (deduped by second) until `cap`, and return sorted by timestamp. Never +// drops existing hits, so enrichment only ever adds context. +export function mergeSnippets<T extends { seconds: number }>( + existing: T[], + incoming: T[], + cap: number, +): T[] { + const seen = new Set(existing.map((s) => s.seconds)); + const merged = existing.slice(); + for (const s of incoming) { + if (merged.length >= cap) break; + if (seen.has(s.seconds)) continue; + seen.add(s.seconds); + merged.push(s); + } + return merged.sort((a, b) => a.seconds - b.seconds); +} diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Drill into a result — read the transcript around any hit.** Pinned search results used to be fixed to their matched snippets (~240 characters), which is often too little to answer "*why* did they say that / what surrounded it." Now the surrounding transcript can be pulled in on demand, two ways. In the pinned panel, expand a video and click a timestamp (or **context**) to add the neighbouring transcript to what the assistant reads — deterministic, no extra AI call, and it works in strict mode and on every provider. And in expand mode on tool-capable providers, the assistant can do this itself via a new **fetch_context** tool when a snippet is too thin, showing a *reading* step in the live pipeline. Windows are bounded (±45s, capped cues, ≤30 excerpts per video) so full multi-hour transcripts are never dumped, and enriched excerpts persist with the conversation. See `common/lib/transcriptWindow.ts`, `export/app/lib/nativeTools/*`, `export/app/lib/searchAgent.ts`, `export/app/ask/{useAskChat.ts,PinnedResultsPanel.tsx,PipelineStatus.tsx}`, and `export/e2e/ask-chat.spec.ts`. - **Hand a search's results straight to the "Ask AI" chat.** The search results header gains an **Ask AI about these results** button (next to *Copy for AI*) that opens the chat grounded in *exactly* the videos you found, instead of the assistant deciding its own search. By default it answers **only** from those results (fast and predictable); a per-chat toggle — *Answer only from these results* — lets the assistant also search the archive, using your results as a starting point. The pinned set is shown with the search that produced it, survives reloads, and can be detached (**Clear**) or replaced with a new hand-off. See `common/lib/aiHandoff.ts`, `common/components/TranscriptSearch.tsx`, `export/app/ask/{useAskChat.ts,PinnedResultsPanel.tsx,AskChat.tsx}`, `export/app/lib/searchAgent.ts`, and `export/e2e/{ask-chat,query-tree}.spec.ts`. ## [0.7.5] - 2026-07-07 diff --git a/export/app/ask/AskChat.tsx b/export/app/ask/AskChat.tsx @@ -117,8 +117,10 @@ export default function AskChat() { pinned={s.pinned} strictGrounding={s.strictGrounding} busy={busy} + expanding={s.expanding} onSetStrict={s.setStrictGrounding} onClear={s.clearPinned} + onExpandVideo={s.expandPinnedVideo} /> )} diff --git a/export/app/ask/MessageBubble.tsx b/export/app/ask/MessageBubble.tsx @@ -76,11 +76,17 @@ export function MessageBubble({ </div> )} - {!isUser && message.phase === "done" && (message.searchSteps?.length ?? 0) > 0 && ( - <p className="font-mono text-xs text-muted-foreground/70"> - Searched: {message.searchSteps!.map((s) => s.query).join(" · ")} - </p> - )} + {!isUser && + message.phase === "done" && + (message.searchSteps?.filter((s) => s.kind !== "fetch").length ?? 0) > 0 && ( + <p className="font-mono text-xs text-muted-foreground/70"> + Searched:{" "} + {message + .searchSteps!.filter((s) => s.kind !== "fetch") + .map((s) => s.query) + .join(" · ")} + </p> + )} {!isUser && message.phase === "done" && message.content && ( <div className="flex items-center"> diff --git a/export/app/ask/PinnedResultsPanel.tsx b/export/app/ask/PinnedResultsPanel.tsx @@ -1,24 +1,35 @@ "use client"; import { useState } from "react"; -import { ChevronDownIcon, PinIcon, XIcon } from "lucide-react"; +import { + ChevronDownIcon, + Loader2Icon, + PinIcon, + PlusIcon, + XIcon, +} from "lucide-react"; import type { SearchHandoff } from "yt-dlp-transcript-common/lib/aiHandoff"; // Shows the search results handed off from the search page as the chat's pinned // grounding: what they are, whether the assistant may look beyond them (the -// strict/expand toggle), and a way to detach them. +// strict/expand toggle), a way to read more transcript around any hit, and a way +// to detach them. export function PinnedResultsPanel({ pinned, strictGrounding, busy, + expanding, onSetStrict, onClear, + onExpandVideo, }: { pinned: SearchHandoff; strictGrounding: boolean; busy: boolean; + expanding: Record<string, boolean>; onSetStrict: (on: boolean) => void; onClear: () => void; + onExpandVideo: (key: string, aroundSeconds?: number) => void; }) { const [showList, setShowList] = useState(false); const n = pinned.videos.length; @@ -78,15 +89,63 @@ export function PinnedResultsPanel({ {showList ? "Hide" : "Show"} the {n} video{n === 1 ? "" : "s"} </button> {showList && ( - <ul className="mt-2 flex flex-col gap-1 border-t border-border pt-2"> - {pinned.videos.map((v) => ( - <li key={v.key} className="truncate text-xs text-muted-foreground"> - <span className="text-foreground">{v.title}</span> - {v.channel ? ` — ${v.channel}` : ""} - </li> - ))} + <ul className="mt-2 flex flex-col gap-3 border-t border-border pt-2"> + {pinned.videos.map((v) => { + const loading = !!expanding[v.key]; + return ( + <li key={v.key} className="flex flex-col gap-1 text-xs"> + <div className="flex items-start gap-2"> + <div className="min-w-0 flex-1 truncate text-muted-foreground"> + <span className="text-foreground">{v.title}</span> + {v.channel ? ` — ${v.channel}` : ""} + <span className="text-muted-foreground/60"> + {" "} + · {v.snippets.length} excerpt + {v.snippets.length === 1 ? "" : "s"} + </span> + </div> + <button + type="button" + onClick={() => onExpandVideo(v.key)} + disabled={busy || loading} + title="Read the transcript around the top hit and add it to what the assistant reads" + className="inline-flex shrink-0 items-center gap-1 rounded-md border border-border px-1.5 py-0.5 text-muted-foreground transition-colors hover:text-foreground disabled:opacity-50" + > + {loading ? ( + <Loader2Icon className="size-3 animate-spin motion-reduce:animate-none" /> + ) : ( + <PlusIcon className="size-3" /> + )} + context + </button> + </div> + {v.snippets.length > 0 && ( + <ul className="flex flex-col gap-0.5 border-l border-border pl-2 font-mono text-muted-foreground/80"> + {v.snippets.map((sn) => ( + <li key={sn.seconds} className="flex gap-2"> + <button + type="button" + onClick={() => onExpandVideo(v.key, sn.seconds)} + disabled={busy || loading} + title="Read the transcript around this moment" + className="shrink-0 text-brand transition-colors hover:underline disabled:opacity-50" + > + {sn.clock} + </button> + <span className="truncate">{sn.text}</span> + </li> + ))} + </ul> + )} + </li> + ); + })} </ul> )} + <p className="mt-2 text-xs text-muted-foreground/70"> + Click a timestamp (or “context”) to add the surrounding transcript to + what the assistant reads. + </p> </div> </div> ); diff --git a/export/app/ask/PipelineStatus.tsx b/export/app/ask/PipelineStatus.tsx @@ -31,25 +31,32 @@ export function PipelineStatus({ </div> )} - {steps.map((s, i) => ( - <div - key={i} - className="flex items-center gap-2 animate-in fade-in slide-in-from-top-1 motion-reduce:animate-none" - > - {s.count === undefined ? ( - <Loader2Icon className="size-3.5 shrink-0 animate-spin text-brand motion-reduce:animate-none" /> - ) : ( - <CheckIcon className="size-3.5 shrink-0 text-brand" /> - )} - <span className="shrink-0">searching</span> - <span className="truncate text-foreground">“{s.query}”</span> - {s.count !== undefined && ( - <span className="shrink-0 text-muted-foreground/70"> - · {s.count} result{s.count === 1 ? "" : "s"} + {steps.map((s, i) => { + const isFetch = s.kind === "fetch"; + return ( + <div + key={i} + className="flex items-center gap-2 animate-in fade-in slide-in-from-top-1 motion-reduce:animate-none" + > + {s.count === undefined ? ( + <Loader2Icon className="size-3.5 shrink-0 animate-spin text-brand motion-reduce:animate-none" /> + ) : ( + <CheckIcon className="size-3.5 shrink-0 text-brand" /> + )} + <span className="shrink-0">{isFetch ? "reading" : "searching"}</span> + <span className="truncate text-foreground"> + {isFetch ? s.query : `“${s.query}”`} </span> - )} - </div> - ))} + {s.count !== undefined && ( + <span className="shrink-0 text-muted-foreground/70"> + {isFetch + ? `· ${s.count} line${s.count === 1 ? "" : "s"}` + : `· ${s.count} result${s.count === 1 ? "" : "s"}`} + </span> + )} + </div> + ); + })} {phase === "answering" && ( <div className="flex items-center gap-2 animate-in fade-in slide-in-from-top-1 motion-reduce:animate-none"> diff --git a/export/app/ask/useAskChat.ts b/export/app/ask/useAskChat.ts @@ -6,6 +6,12 @@ import { AI_HANDOFF_KEY, type SearchHandoff, } from "yt-dlp-transcript-common/lib/aiHandoff"; +import { fetchTranscript } from "yt-dlp-transcript-common/components/transcriptCache"; +import { + cuesToSnippets, + mergeSnippets, + windowCues, +} from "yt-dlp-transcript-common/lib/transcriptWindow"; import { PROVIDERS, type Provider } from "../lib/askProvider"; import { runAskTurn, type AgentEvent, type AgentMode } from "../lib/searchAgent"; import type { RetrievedVideo } from "../lib/askRetrieval"; @@ -61,6 +67,8 @@ export function useAskChat() { // Pinned mode only: answer strictly from the pinned results (skip gather) vs. // use them as a starting point the AI may expand with its own searches. const [strictGrounding, setStrictGroundingState] = useState(true); + // Per-video key → true while its "Load context" fetch is in flight. + const [expanding, setExpanding] = useState<Record<string, boolean>>({}); const [input, setInput] = useState(""); const [busy, setBusy] = useState(false); const abortRef = useRef<AbortController | null>(null); @@ -269,6 +277,28 @@ export function useAskChat() { return { ...m, searchSteps: steps }; }); break; + case "fetch_start": + patchAt(assistantIndex, (m) => ({ + ...m, + phase: "gathering", + searchSteps: [ + ...(m.searchSteps ?? []), + { query: e.label, kind: "fetch" }, + ], + })); + break; + case "fetch_done": + patchAt(assistantIndex, (m) => { + const steps = (m.searchSteps ?? []).slice(); + for (let i = steps.length - 1; i >= 0; i--) { + if (steps[i].kind === "fetch" && steps[i].count === undefined) { + steps[i] = { ...steps[i], count: e.count }; + break; + } + } + return { ...m, searchSteps: steps }; + }); + break; case "answer_start": // Store grounding NOW (before streaming) so a failed stream can be // retried without re-searching, and the excerpts persist. @@ -448,6 +478,54 @@ export function useAskChat() { const setStrictGrounding = (on: boolean) => setStrictGroundingState(on); + // User-driven "expand this hit": read the surrounding transcript for a pinned + // video (client-side, IDB-cached) and merge it into that video's snippets, so + // the assistant reads the fuller context on the next question. Works in strict + // AND expand mode, on every provider — no LLM call. Persisted with the convo. + const expandPinnedVideo = useCallback( + async (key: string, aroundSeconds?: number) => { + const pin = pinnedRef.current; + const v = pin?.videos.find((x) => x.key === key); + if (!v) return; + const center = + typeof aroundSeconds === "number" + ? aroundSeconds + : v.snippets[0]?.seconds ?? 0; + setExpanding((e) => ({ ...e, [key]: true })); + try { + const detail = await fetchTranscript(key); + const snips = cuesToSnippets( + windowCues(detail.cues ?? [], center, { + before: 45, + after: 45, + maxCues: 60, + }), + ); + setPinned((prev) => + prev + ? { + ...prev, + videos: prev.videos.map((x) => + x.key === key + ? { ...x, snippets: mergeSnippets(x.snippets, snips, 30) } + : x, + ), + } + : prev, + ); + } catch { + /* transcript unavailable — leave the pin untouched */ + } finally { + setExpanding((e) => { + const next = { ...e }; + delete next[key]; + return next; + }); + } + }, + [], + ); + // The effective context that will be sent next turn, serialized for the panel: // the override (if any) prepended to the replayed conversation. const contextText = useMemo(() => { @@ -509,6 +587,8 @@ export function useAskChat() { strictGrounding, setStrictGrounding, clearPinned, + expandPinnedVideo, + expanding, }; } diff --git a/export/app/lib/askConversation.ts b/export/app/lib/askConversation.ts @@ -14,9 +14,11 @@ import type { ChatMessage } from "./askProvider"; import { buildContext, type RetrievedVideo } from "./askRetrieval"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; -// One search the agent ran this turn (shown live in the pipeline UI). `count` -// is undefined while the search is in flight, then set to the match count. -export type SearchStep = { query: string; count?: number }; +// One retrieval step the agent ran this turn (shown live in the pipeline UI). +// `kind` is "search" (default) or "fetch" (reading more of a pinned video's +// transcript); for a fetch step `query` holds the video title. `count` is +// undefined while in flight, then the match/line count. +export type SearchStep = { query: string; count?: number; kind?: "search" | "fetch" }; // Top-level stage of a pending assistant turn, for the status indicator. export type AssistantPhase = @@ -162,6 +164,7 @@ export function gatherSystemPrompt( aliases: SearchAlias[], mode: "native" | "scripted", budget: number, + canFetch = false, ): string { const base = "You are helping answer a question about a video-transcript archive. " + @@ -172,10 +175,17 @@ export function gatherSystemPrompt( "Only search when you need transcript evidence: if the latest message just " + "asks to reformat, summarise, translate, or expand on the previous answer, " + `do not search. You may run at most ${budget} searches.`; + const fetchLine = canFetch + ? " When a snippet is too short to answer confidently, call the " + + "fetch_context tool with the video's ref (and optionally a timestamp in " + + "seconds) to read the surrounding transcript before answering." + : ""; const tail = mode === "native" - ? "Call the search_transcripts tool to search. Call the finish tool as " + - "soon as you have enough excerpts, or immediately if no search is needed." + ? "Call the search_transcripts tool to search." + + fetchLine + + " Call the finish tool as soon as you have enough excerpts, or " + + "immediately if no search is needed." : "Reply with EXACTLY one line and nothing else: either " + "`SEARCH: <query>` to run a search, or `DONE` when you have enough " + "excerpts (or need no search)."; diff --git a/export/app/lib/nativeTools/anthropic.ts b/export/app/lib/nativeTools/anthropic.ts @@ -5,8 +5,13 @@ import { PROVIDERS } from "../askProvider"; import { abortError, EMPTY_PARAMS, + FETCH_BUDGET, + FETCH_PARAMS, + FETCH_TOOL_DESCRIPTION, + FETCH_TOOL_NAME, FINISH_TOOL_DESCRIPTION, FINISH_TOOL_NAME, + type ParsedToolCall, postJson, SEARCH_PARAMS, SEARCH_TOOL_DESCRIPTION, @@ -14,38 +19,53 @@ import { type NativeGatherContext, } from "./shared"; -const TOOLS = [ - { - name: SEARCH_TOOL_NAME, - description: SEARCH_TOOL_DESCRIPTION, - input_schema: SEARCH_PARAMS, - }, - { +// The fetch_context tool is offered only when the turn can read transcripts +// (expand mode with a pinned set); search + finish are always present. +function buildTools(includeFetch: boolean) { + const tools: { name: string; description: string; input_schema: unknown }[] = [ + { + name: SEARCH_TOOL_NAME, + description: SEARCH_TOOL_DESCRIPTION, + input_schema: SEARCH_PARAMS, + }, + ]; + if (includeFetch) { + tools.push({ + name: FETCH_TOOL_NAME, + description: FETCH_TOOL_DESCRIPTION, + input_schema: FETCH_PARAMS, + }); + } + tools.push({ name: FINISH_TOOL_NAME, description: FINISH_TOOL_DESCRIPTION, input_schema: EMPTY_PARAMS, - }, -]; + }); + return tools; +} // Pure: extract tool_use blocks from an Anthropic message response. -export function parseAnthropicToolUses( - json: unknown, -): { id: string; name: string; query: string }[] { +export function parseAnthropicToolUses(json: unknown): ParsedToolCall[] { const content = (json as { content?: unknown }).content; if (!Array.isArray(content)) return []; - const out: { id: string; name: string; query: string }[] = []; + const out: ParsedToolCall[] = []; for (const block of content) { const b = block as { type?: string; id?: string; name?: string; - input?: { query?: unknown }; + input?: { query?: unknown; video?: unknown; aroundSeconds?: unknown }; }; if (b.type === "tool_use" && typeof b.name === "string") { out.push({ id: b.id ?? "", name: b.name, query: typeof b.input?.query === "string" ? b.input.query : "", + video: typeof b.input?.video === "string" ? b.input.video : "", + aroundSeconds: + typeof b.input?.aroundSeconds === "number" + ? b.input.aroundSeconds + : undefined, }); } } @@ -53,13 +73,17 @@ export function parseAnthropicToolUses( } export async function anthropicGather(ctx: NativeGatherContext): Promise<void> { + const canFetch = !!ctx.runFetchContext; + const tools = buildTools(canFetch); const messages: { role: string; content: unknown }[] = [ ...ctx.history.map((m) => ({ role: m.role, content: m.content })), { role: "user", content: ctx.question }, ]; let searchesRun = 0; + let fetchesRun = 0; + const maxRounds = ctx.budget + 1 + (canFetch ? FETCH_BUDGET : 0); - for (let round = 0; round <= ctx.budget + 1; round++) { + for (let round = 0; round <= maxRounds; round++) { if (ctx.signal?.aborted) throw abortError(); const json = await postJson( "https://api.anthropic.com/v1/messages", @@ -72,7 +96,7 @@ export async function anthropicGather(ctx: NativeGatherContext): Promise<void> { model: ctx.model || PROVIDERS.anthropic.defaultModel, max_tokens: 512, system: ctx.system, - tools: TOOLS, + tools, messages, }, "Anthropic", @@ -80,11 +104,14 @@ export async function anthropicGather(ctx: NativeGatherContext): Promise<void> { ); const calls = parseAnthropicToolUses(json); - const searches = calls.filter( + const wantsSearch = calls.some( (c) => c.name === SEARCH_TOOL_NAME && c.query.trim() !== "", ); - // No search requested → the model finished (or answered directly). Done. - if (searches.length === 0) return; + const wantsFetch = + canFetch && + calls.some((c) => c.name === FETCH_TOOL_NAME && c.video.trim() !== ""); + // No actionable tool call → the model finished (or answered directly). Done. + if (!wantsSearch && !wantsFetch) return; // Replay the assistant's tool_use turn verbatim, then answer each tool_use. messages.push({ @@ -93,30 +120,35 @@ export async function anthropicGather(ctx: NativeGatherContext): Promise<void> { }); const toolResults: unknown[] = []; for (const call of calls) { + let content: string; if (call.name === SEARCH_TOOL_NAME && call.query.trim() !== "") { - const content = - searchesRun < ctx.budget - ? await ((): Promise<string> => { - searchesRun += 1; - return ctx.runSearch(call.query); - })() - : "Search budget reached — answer with the excerpts gathered so far."; - toolResults.push({ - type: "tool_result", - tool_use_id: call.id, - content, - }); + if (searchesRun < ctx.budget) { + searchesRun += 1; + content = await ctx.runSearch(call.query); + } else { + content = "Search budget reached — answer with the excerpts gathered so far."; + } + } else if ( + canFetch && + call.name === FETCH_TOOL_NAME && + call.video.trim() !== "" + ) { + if (fetchesRun < FETCH_BUDGET) { + fetchesRun += 1; + content = await ctx.runFetchContext!(call.video, call.aroundSeconds); + } else { + content = "Reached the limit on transcript reads this turn."; + } } else { // finish (or any other tool): acknowledge so the block is satisfied. - toolResults.push({ - type: "tool_result", - tool_use_id: call.id, - content: "Acknowledged.", - }); + content = "Acknowledged."; } + toolResults.push({ + type: "tool_result", + tool_use_id: call.id, + content, + }); } messages.push({ role: "user", content: toolResults }); - - if (searchesRun >= ctx.budget) return; } } diff --git a/export/app/lib/nativeTools/gemini.ts b/export/app/lib/nativeTools/gemini.ts @@ -5,8 +5,13 @@ import { PROVIDERS } from "../askProvider"; import { abortError, EMPTY_PARAMS, + FETCH_BUDGET, + FETCH_PARAMS, + FETCH_TOOL_DESCRIPTION, + FETCH_TOOL_NAME, FINISH_TOOL_DESCRIPTION, FINISH_TOOL_NAME, + type ParsedToolCall, postJson, SEARCH_PARAMS, SEARCH_TOOL_DESCRIPTION, @@ -14,41 +19,61 @@ import { type NativeGatherContext, } from "./shared"; -const TOOLS = [ - { - functionDeclarations: [ - { - name: SEARCH_TOOL_NAME, - description: SEARCH_TOOL_DESCRIPTION, - parameters: SEARCH_PARAMS, - }, - { - name: FINISH_TOOL_NAME, - description: FINISH_TOOL_DESCRIPTION, - parameters: EMPTY_PARAMS, - }, - ], - }, -]; +function buildTools(includeFetch: boolean) { + const functionDeclarations: { + name: string; + description: string; + parameters: unknown; + }[] = [ + { + name: SEARCH_TOOL_NAME, + description: SEARCH_TOOL_DESCRIPTION, + parameters: SEARCH_PARAMS, + }, + ]; + if (includeFetch) { + functionDeclarations.push({ + name: FETCH_TOOL_NAME, + description: FETCH_TOOL_DESCRIPTION, + parameters: FETCH_PARAMS, + }); + } + functionDeclarations.push({ + name: FINISH_TOOL_NAME, + description: FINISH_TOOL_DESCRIPTION, + parameters: EMPTY_PARAMS, + }); + return [{ functionDeclarations }]; +} // Pure: extract functionCall parts from a Gemini generateContent response. -export function parseGeminiFunctionCalls( - json: unknown, -): { name: string; query: string }[] { +export function parseGeminiFunctionCalls(json: unknown): ParsedToolCall[] { const parts = ( json as { candidates?: { content?: { parts?: unknown[] } }[]; } ).candidates?.[0]?.content?.parts ?? []; - const out: { name: string; query: string }[] = []; + const out: ParsedToolCall[] = []; for (const part of parts) { - const fc = (part as { functionCall?: { name?: string; args?: { query?: unknown } } }) - .functionCall; + const fc = ( + part as { + functionCall?: { + name?: string; + args?: { query?: unknown; video?: unknown; aroundSeconds?: unknown }; + }; + } + ).functionCall; if (fc?.name) { out.push({ + id: "", name: fc.name, query: typeof fc.args?.query === "string" ? fc.args.query : "", + video: typeof fc.args?.video === "string" ? fc.args.video : "", + aroundSeconds: + typeof fc.args?.aroundSeconds === "number" + ? fc.args.aroundSeconds + : undefined, }); } } @@ -56,6 +81,8 @@ export function parseGeminiFunctionCalls( } export async function geminiGather(ctx: NativeGatherContext): Promise<void> { + const canFetch = !!ctx.runFetchContext; + const tools = buildTools(canFetch); const model = ctx.model || PROVIDERS.gemini.defaultModel; const url = `https://generativelanguage.googleapis.com/v1beta/models/` + @@ -68,15 +95,17 @@ export async function geminiGather(ctx: NativeGatherContext): Promise<void> { { role: "user", parts: [{ text: ctx.question }] }, ]; let searchesRun = 0; + let fetchesRun = 0; + const maxRounds = ctx.budget + 1 + (canFetch ? FETCH_BUDGET : 0); - for (let round = 0; round <= ctx.budget + 1; round++) { + for (let round = 0; round <= maxRounds; round++) { if (ctx.signal?.aborted) throw abortError(); const json = await postJson( url, {}, { system_instruction: { parts: [{ text: ctx.system }] }, - tools: TOOLS, + tools, contents, }, "Gemini", @@ -84,16 +113,29 @@ export async function geminiGather(ctx: NativeGatherContext): Promise<void> { ); const calls = parseGeminiFunctionCalls(json); - const searches = calls.filter( + const wantsSearch = calls.some( (c) => c.name === SEARCH_TOOL_NAME && c.query.trim() !== "", ); - if (searches.length === 0) return; // finish or a plain answer → done + const wantsFetch = + canFetch && + calls.some((c) => c.name === FETCH_TOOL_NAME && c.video.trim() !== ""); + if (!wantsSearch && !wantsFetch) return; // finish or a plain answer → done // Replay the model's functionCall turn, then send functionResponse parts. const modelParts = calls.map((c) => ({ functionCall: { name: c.name, - args: c.name === SEARCH_TOOL_NAME ? { query: c.query } : {}, + args: + c.name === SEARCH_TOOL_NAME + ? { query: c.query } + : c.name === FETCH_TOOL_NAME + ? { + video: c.video, + ...(c.aroundSeconds != null + ? { aroundSeconds: c.aroundSeconds } + : {}), + } + : {}, }, })); contents.push({ role: "model", parts: modelParts }); @@ -102,13 +144,23 @@ export async function geminiGather(ctx: NativeGatherContext): Promise<void> { for (const call of calls) { let result: string; if (call.name === SEARCH_TOOL_NAME && call.query.trim() !== "") { - result = - searchesRun < ctx.budget - ? await (() => { - searchesRun += 1; - return ctx.runSearch(call.query); - })() - : "Search budget reached — answer with the excerpts gathered so far."; + if (searchesRun < ctx.budget) { + searchesRun += 1; + result = await ctx.runSearch(call.query); + } else { + result = "Search budget reached — answer with the excerpts gathered so far."; + } + } else if ( + canFetch && + call.name === FETCH_TOOL_NAME && + call.video.trim() !== "" + ) { + if (fetchesRun < FETCH_BUDGET) { + fetchesRun += 1; + result = await ctx.runFetchContext!(call.video, call.aroundSeconds); + } else { + result = "Reached the limit on transcript reads this turn."; + } } else { result = "Acknowledged."; } @@ -118,7 +170,5 @@ export async function geminiGather(ctx: NativeGatherContext): Promise<void> { } // Function responses are sent back as a user-role turn. contents.push({ role: "user", parts: responseParts }); - - if (searchesRun >= ctx.budget) return; } } diff --git a/export/app/lib/nativeTools/nativeTools.test.ts b/export/app/lib/nativeTools/nativeTools.test.ts @@ -13,8 +13,8 @@ test("parseAnthropicToolUses extracts tool_use blocks", () => { ], }; assert.deepEqual(parseAnthropicToolUses(json), [ - { id: "tu_1", name: "search_transcripts", query: "platner" }, - { id: "tu_2", name: "finish", query: "" }, + { id: "tu_1", name: "search_transcripts", query: "platner", video: "", aroundSeconds: undefined }, + { id: "tu_2", name: "finish", query: "", video: "", aroundSeconds: undefined }, ]); assert.deepEqual(parseAnthropicToolUses({ content: "no tools" }), []); }); @@ -36,8 +36,8 @@ test("parseOpenAIToolCalls parses JSON arguments", () => { ], }; assert.deepEqual(parseOpenAIToolCalls(json), [ - { id: "call_1", name: "search_transcripts", query: "senate campaign" }, - { id: "call_2", name: "finish", query: "" }, + { id: "call_1", name: "search_transcripts", query: "senate campaign", video: "", aroundSeconds: undefined }, + { id: "call_2", name: "finish", query: "", video: "", aroundSeconds: undefined }, ]); // Malformed arguments degrade to an empty query, not a throw. const bad = { @@ -55,7 +55,7 @@ test("parseOpenAIToolCalls parses JSON arguments", () => { ], }; assert.deepEqual(parseOpenAIToolCalls(bad), [ - { id: "x", name: "search_transcripts", query: "" }, + { id: "x", name: "search_transcripts", query: "", video: "", aroundSeconds: undefined }, ]); assert.deepEqual(parseOpenAIToolCalls({ choices: [{ message: {} }] }), []); }); @@ -74,7 +74,7 @@ test("parseGeminiFunctionCalls extracts functionCall parts", () => { ], }; assert.deepEqual(parseGeminiFunctionCalls(json), [ - { name: "search_transcripts", query: "timeline" }, + { id: "", name: "search_transcripts", query: "timeline", video: "", aroundSeconds: undefined }, ]); assert.deepEqual(parseGeminiFunctionCalls({ candidates: [] }), []); }); diff --git a/export/app/lib/nativeTools/openai.ts b/export/app/lib/nativeTools/openai.ts @@ -5,8 +5,13 @@ import { PROVIDERS } from "../askProvider"; import { abortError, EMPTY_PARAMS, + FETCH_BUDGET, + FETCH_PARAMS, + FETCH_TOOL_DESCRIPTION, + FETCH_TOOL_NAME, FINISH_TOOL_DESCRIPTION, FINISH_TOOL_NAME, + type ParsedToolCall, postJson, SEARCH_PARAMS, SEARCH_TOOL_DESCRIPTION, @@ -14,24 +19,40 @@ import { type NativeGatherContext, } from "./shared"; -const TOOLS = [ - { - type: "function", - function: { - name: SEARCH_TOOL_NAME, - description: SEARCH_TOOL_DESCRIPTION, - parameters: SEARCH_PARAMS, +function buildTools(includeFetch: boolean) { + const tools: { + type: "function"; + function: { name: string; description: string; parameters: unknown }; + }[] = [ + { + type: "function", + function: { + name: SEARCH_TOOL_NAME, + description: SEARCH_TOOL_DESCRIPTION, + parameters: SEARCH_PARAMS, + }, }, - }, - { + ]; + if (includeFetch) { + tools.push({ + type: "function", + function: { + name: FETCH_TOOL_NAME, + description: FETCH_TOOL_DESCRIPTION, + parameters: FETCH_PARAMS, + }, + }); + } + tools.push({ type: "function", function: { name: FINISH_TOOL_NAME, description: FINISH_TOOL_DESCRIPTION, parameters: EMPTY_PARAMS, }, - }, -]; + }); + return tools; +} type RawToolCall = { id?: string; @@ -39,34 +60,40 @@ type RawToolCall = { }; // Pure: extract tool calls from an OpenAI chat-completion response. -export function parseOpenAIToolCalls( - json: unknown, -): { id: string; name: string; query: string }[] { +export function parseOpenAIToolCalls(json: unknown): ParsedToolCall[] { const msg = (json as { choices?: { message?: { tool_calls?: unknown } }[] }) .choices?.[0]?.message; const calls = (msg as { tool_calls?: unknown })?.tool_calls; if (!Array.isArray(calls)) return []; return (calls as RawToolCall[]).map((c) => { let query = ""; + let video = ""; + let aroundSeconds: number | undefined; try { const args = JSON.parse(c.function?.arguments ?? "{}"); if (typeof args.query === "string") query = args.query; + if (typeof args.video === "string") video = args.video; + if (typeof args.aroundSeconds === "number") aroundSeconds = args.aroundSeconds; } catch { - /* malformed args → empty query */ + /* malformed args → empty */ } - return { id: c.id ?? "", name: c.function?.name ?? "", query }; + return { id: c.id ?? "", name: c.function?.name ?? "", query, video, aroundSeconds }; }); } export async function openaiGather(ctx: NativeGatherContext): Promise<void> { + const canFetch = !!ctx.runFetchContext; + const tools = buildTools(canFetch); const messages: unknown[] = [ { role: "system", content: ctx.system }, ...ctx.history.map((m) => ({ role: m.role, content: m.content })), { role: "user", content: ctx.question }, ]; let searchesRun = 0; + let fetchesRun = 0; + const maxRounds = ctx.budget + 1 + (canFetch ? FETCH_BUDGET : 0); - for (let round = 0; round <= ctx.budget + 1; round++) { + for (let round = 0; round <= maxRounds; round++) { if (ctx.signal?.aborted) throw abortError(); const json = await postJson( "https://api.openai.com/v1/chat/completions", @@ -74,7 +101,7 @@ export async function openaiGather(ctx: NativeGatherContext): Promise<void> { { model: ctx.model || PROVIDERS.openai.defaultModel, max_tokens: 512, - tools: TOOLS, + tools, tool_choice: "auto", messages, }, @@ -85,11 +112,14 @@ export async function openaiGather(ctx: NativeGatherContext): Promise<void> { const rawMsg = (json as { choices?: { message?: unknown }[] }).choices?.[0] ?.message; const calls = parseOpenAIToolCalls(json); - const searches = calls.filter( + const wantsSearch = calls.some( (c) => c.name === SEARCH_TOOL_NAME && c.query.trim() !== "", ); - // No tool call (or only a plain answer) → done gathering. - if (calls.length === 0 || searches.length === 0) return; + const wantsFetch = + canFetch && + calls.some((c) => c.name === FETCH_TOOL_NAME && c.video.trim() !== ""); + // No actionable tool call → done gathering. + if (!wantsSearch && !wantsFetch) return; // Replay the assistant message with its tool_calls, then answer EVERY call // (OpenAI requires a tool response for each tool_call id). @@ -97,19 +127,27 @@ export async function openaiGather(ctx: NativeGatherContext): Promise<void> { for (const call of calls) { let content: string; if (call.name === SEARCH_TOOL_NAME && call.query.trim() !== "") { - content = - searchesRun < ctx.budget - ? await (() => { - searchesRun += 1; - return ctx.runSearch(call.query); - })() - : "Search budget reached — answer with the excerpts gathered so far."; + if (searchesRun < ctx.budget) { + searchesRun += 1; + content = await ctx.runSearch(call.query); + } else { + content = "Search budget reached — answer with the excerpts gathered so far."; + } + } else if ( + canFetch && + call.name === FETCH_TOOL_NAME && + call.video.trim() !== "" + ) { + if (fetchesRun < FETCH_BUDGET) { + fetchesRun += 1; + content = await ctx.runFetchContext!(call.video, call.aroundSeconds); + } else { + content = "Reached the limit on transcript reads this turn."; + } } else { content = "Acknowledged."; } messages.push({ role: "tool", tool_call_id: call.id, content }); } - - if (searchesRun >= ctx.budget) return; } } diff --git a/export/app/lib/nativeTools/parsers.test.ts b/export/app/lib/nativeTools/parsers.test.ts @@ -0,0 +1,90 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { parseAnthropicToolUses } from "./anthropic"; +import { parseOpenAIToolCalls } from "./openai"; +import { parseGeminiFunctionCalls } from "./gemini"; +import { FETCH_TOOL_NAME, SEARCH_TOOL_NAME } from "./shared"; + +test("parseAnthropicToolUses reads search and fetch_context calls", () => { + const calls = parseAnthropicToolUses({ + content: [ + { type: "text", text: "thinking" }, + { type: "tool_use", id: "a", name: SEARCH_TOOL_NAME, input: { query: "platner" } }, + { + type: "tool_use", + id: "b", + name: FETCH_TOOL_NAME, + input: { video: "chan/xyz", aroundSeconds: 125 }, + }, + ], + }); + assert.equal(calls.length, 2); + assert.deepEqual( + { ...calls[0] }, + { id: "a", name: SEARCH_TOOL_NAME, query: "platner", video: "", aroundSeconds: undefined }, + ); + assert.deepEqual( + { ...calls[1] }, + { id: "b", name: FETCH_TOOL_NAME, query: "", video: "chan/xyz", aroundSeconds: 125 }, + ); +}); + +test("parseOpenAIToolCalls reads search and fetch_context calls", () => { + const calls = parseOpenAIToolCalls({ + choices: [ + { + message: { + tool_calls: [ + { + id: "1", + function: { name: SEARCH_TOOL_NAME, arguments: JSON.stringify({ query: "platner" }) }, + }, + { + id: "2", + function: { + name: FETCH_TOOL_NAME, + arguments: JSON.stringify({ video: "chan/xyz", aroundSeconds: 60 }), + }, + }, + ], + }, + }, + ], + }); + assert.equal(calls.length, 2); + assert.equal(calls[0].name, SEARCH_TOOL_NAME); + assert.equal(calls[0].query, "platner"); + assert.equal(calls[1].name, FETCH_TOOL_NAME); + assert.equal(calls[1].video, "chan/xyz"); + assert.equal(calls[1].aroundSeconds, 60); +}); + +test("parseOpenAIToolCalls tolerates malformed arguments", () => { + const calls = parseOpenAIToolCalls({ + choices: [ + { message: { tool_calls: [{ id: "1", function: { name: SEARCH_TOOL_NAME, arguments: "{bad" } }] } }, + ], + }); + assert.equal(calls[0].query, ""); + assert.equal(calls[0].video, ""); +}); + +test("parseGeminiFunctionCalls reads search and fetch_context calls", () => { + const calls = parseGeminiFunctionCalls({ + candidates: [ + { + content: { + parts: [ + { functionCall: { name: SEARCH_TOOL_NAME, args: { query: "platner" } } }, + { functionCall: { name: FETCH_TOOL_NAME, args: { video: "chan/xyz", aroundSeconds: 30 } } }, + ], + }, + }, + ], + }); + assert.equal(calls.length, 2); + assert.equal(calls[0].query, "platner"); + assert.equal(calls[1].name, FETCH_TOOL_NAME); + assert.equal(calls[1].video, "chan/xyz"); + assert.equal(calls[1].aroundSeconds, 30); +}); diff --git a/export/app/lib/nativeTools/shared.ts b/export/app/lib/nativeTools/shared.ts @@ -19,17 +19,32 @@ export type NativeGatherContext = { budget: number; // Runs one search and returns the results text to feed back to the model. runSearch: (query: string) => Promise<string>; + // Reads more of a pinned video's transcript around a moment and returns the + // window text. Present only in expand mode with a pinned set (native only); + // when undefined the fetch_context tool is not offered. + runFetchContext?: (video: string, aroundSeconds?: number) => Promise<string>; signal?: AbortSignal; }; export const SEARCH_TOOL_NAME = "search_transcripts"; +export const FETCH_TOOL_NAME = "fetch_context"; export const FINISH_TOOL_NAME = "finish"; +// Bounds transcript reads per turn (independent of the search budget) so the +// user's key isn't spent on unbounded fetching. +export const FETCH_BUDGET = 4; + export const SEARCH_TOOL_DESCRIPTION = "Search the video-transcript archive for excerpts relevant to a query. " + "Returns matching videos with timestamped snippet lines. Use focused keyword " + "or name queries; call it again to refine based on what you find."; +export const FETCH_TOOL_DESCRIPTION = + "Read more of a specific pinned video's transcript around a moment, to get the " + + "context surrounding a snippet. Pass the video's ref (shown with each pinned " + + "result) and optionally a timestamp in seconds to centre on. The returned " + + "lines are added to your citable excerpts for that video."; + export const FINISH_TOOL_DESCRIPTION = "Call this when you have gathered enough excerpts (or none are needed) and " + "are ready to answer."; @@ -47,8 +62,36 @@ export const SEARCH_PARAMS = { required: ["query"], } as const; +// JSON-Schema for the fetch_context tool's input. +export const FETCH_PARAMS = { + type: "object", + properties: { + video: { + type: "string", + description: "The video ref to read, exactly as shown with the pinned results.", + }, + aroundSeconds: { + type: "number", + description: + "Centre the excerpt on this timestamp (seconds). Optional; defaults to " + + "the video's first matched moment.", + }, + }, + required: ["video"], +} as const; + export const EMPTY_PARAMS = { type: "object", properties: {} } as const; +// Shape returned by every provider's tool-call parser: `query` for search, +// `video`/`aroundSeconds` for fetch_context (all optional, filled per tool). +export type ParsedToolCall = { + id: string; + name: string; + query: string; + video: string; + aroundSeconds?: number; +}; + // Thrown when a native tool-calling request fails in a way that suggests the // model/endpoint doesn't support tools (HTTP 400/404). The agent catches this // and retries the turn with the scripted transport. diff --git a/export/app/lib/searchAgent.ts b/export/app/lib/searchAgent.ts @@ -32,12 +32,22 @@ import { } from "./nativeTools/shared"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import type { DisplaySummary } from "yt-dlp-transcript-common/lib/transcripts"; +import { fetchTranscript } from "yt-dlp-transcript-common/components/transcriptCache"; +import { + cuesToSnippets, + mergeSnippets, + windowCues, +} from "yt-dlp-transcript-common/lib/transcriptWindow"; +import { hms } from "yt-dlp-transcript-common/lib/aiHandoff"; export type AgentMode = "auto" | "native" | "scripted"; export type AgentEvent = | { type: "search_start"; query: string } | { type: "search_done"; query: string; count: number } + // The model read more of a pinned video's transcript (fetch_context tool). + | { type: "fetch_start"; ref: string; label: string } + | { type: "fetch_done"; ref: string; count: number } // Fired once gather is done, BEFORE the answer streams — carries the grounding // so the UI can persist it and a failed answer stream can be retried without // re-searching. @@ -50,6 +60,10 @@ export const DEFAULT_BUDGET = 4; // Per-search result cap and overall context cap (bounds tokens sent to the model). const PER_SEARCH_LIMIT = 8; const MAX_CONTEXT_VIDEOS = 15; +// fetch_context windowing: ± seconds around the moment, a hard cue cap, and the +// max snippets a video may accumulate (so enrichment can't blow the budget). +const FETCH_WINDOW = { before: 45, after: 45, maxCues: 60 }; +const SNIPPETS_PER_VIDEO_CAP = 30; // ─── transport selection ─── @@ -151,6 +165,26 @@ export function formatResultsForModel(videos: RetrievedVideo[]): string { .join("\n\n"); } +// Tell the model about the pinned results it's grounded in (expand mode), each +// with the `ref` it passes to fetch_context and the moments that matched — so it +// can reason about them and read more around any moment. +export function buildSeedDigest(videos: RetrievedVideo[]): string { + const lines = videos.map((v) => { + const times = v.snippets + .slice(0, 6) + .map((s) => s.clock) + .join(", "); + const site = v.siteTitle ? ` (${v.siteTitle})` : ""; + return `- ref "${v.key}": "${v.title}" — ${v.channel}${site}${times ? `; matched at ${times}` : ""}`; + }); + return ( + "PINNED RESULTS — the user handed you these search results as your starting " + + "grounding. You may search the archive for more, and you may read more of " + + "any of these around a moment:\n" + + lines.join("\n") + ); +} + // ─── the turn ─── export type RunAskTurnOptions = { @@ -262,6 +296,46 @@ export async function runAskTurn( return formatResultsForModel(r.videos); }; + // Expand mode with a pinned set: let the model read more of a grounding video's + // transcript around a moment. Windows the cues client-side (the channel page is + // already warm from the search), merges them into that video's snippets — so the + // deeper context reaches the answer + citations, not just the gather loop. + const canFetchThisTurn = seedVideos.length > 0; + const runFetchContext = async ( + rawRef: string, + aroundSeconds?: number, + ): Promise<string> => { + const ref = rawRef.trim(); + const v = videos.get(ref); + if (!v) return `No pinned video with ref "${ref}".`; + const center = + typeof aroundSeconds === "number" + ? aroundSeconds + : v.snippets[0]?.seconds ?? 0; + onEvent({ type: "fetch_start", ref: v.key, label: v.title }); + let cues; + try { + const detail = await fetchTranscript(v.key); + cues = detail.cues ?? []; + } catch { + onEvent({ type: "fetch_done", ref: v.key, count: 0 }); + return `Couldn't load the transcript for "${v.title}".`; + } + const snips = cuesToSnippets(windowCues(cues, center, FETCH_WINDOW)); + const before = v.snippets.length; + v.snippets = mergeSnippets(v.snippets, snips, SNIPPETS_PER_VIDEO_CAP); + const added = v.snippets.length - before; + onEvent({ type: "fetch_done", ref: v.key, count: snips.length }); + if (snips.length === 0) { + return `No transcript lines found near ${hms(center)} in "${v.title}".`; + } + const body = snips.map((s) => `[${s.clock}] ${s.text}`).join("\n"); + return ( + `Transcript excerpt from "${v.title}" around ${hms(center)} ` + + `(${added} new line${added === 1 ? "" : "s"} added to your citable excerpts):\n${body}` + ); + }; + const gatherCtx: NativeGatherContext = { apiKey, model, @@ -274,9 +348,13 @@ export async function runAskTurn( }; const doGather = async (kind: "native" | "scripted"): Promise<void> => { + const canFetch = kind === "native" && canFetchThisTurn; + let system = gatherSystemPrompt(aliases, kind, budget, canFetch); + if (canFetchThisTurn) system += `\n\n${buildSeedDigest(seedVideos)}`; const ctx: NativeGatherContext = { ...gatherCtx, - system: gatherSystemPrompt(aliases, kind, budget), + system, + runFetchContext: canFetch ? runFetchContext : undefined, }; if (kind === "scripted") return scriptedGather(ctx, provider); switch (provider) { diff --git a/export/e2e/ask-chat.spec.ts b/export/e2e/ask-chat.spec.ts @@ -264,11 +264,214 @@ test.describe("ask chat", () => { }, HANDOFF); } - async function keyIn(page: Page) { - await page.getByRole("button", { name: "Scripted" }).click(); + async function keyIn(page: Page, mode: "Scripted" | "Native tools" = "Scripted") { + await page.getByRole("button", { name: mode }).click(); await page.locator('input[placeholder^="sk-ant"]').fill("sk-ant-test"); } + // A hand-off whose video key resolves against the transcript fixtures + // (installRoutes), so fetch_context / "Load context" can pull real cues. The + // fixture cues sit at 5s, 50s, 100s; a window around the 5s hit pulls in the + // 50s "beta line" — the signal that a fetch enriched the grounding. + const REAL_KEY = "test-channel/vid-transcript-only"; + const FETCH_HANDOFF = { + label: "alpha", + videos: [ + { + key: REAL_KEY, + videoId: "vid-transcript-only", + title: "Transcript only", + channel: "Test Channel", + uploadDate: "20200101", + url: "https://x/v1", + snippets: [ + { clock: "0:05", seconds: 5, text: "transcript-only video — alpha line" }, + ], + }, + ], + totalVideos: 1, + truncated: false, + }; + + // Seed once — guarded so a reload doesn't re-seed (which would clobber the + // enriched pin restored from localStorage). + async function seedFetchHandoff(page: Page) { + await page.addInitScript((h) => { + if (!sessionStorage.getItem("seeded_once")) { + sessionStorage.setItem("ytdlp-tb:ai:handoff", JSON.stringify(h)); + sessionStorage.setItem("seeded_once", "1"); + } + }, FETCH_HANDOFF); + } + + test("expand mode: the model reads more transcript via fetch_context", async ({ + page, + }) => { + await installRoutes(page); + let sawFetchTool = false; + let answerSawWindow = false; + let round = 0; + await page.route("https://api.anthropic.com/**", async (route) => { + if (route.request().method() === "OPTIONS") { + await route.fulfill({ status: 204, headers: CORS }); + return; + } + const body = route.request().postDataJSON() as { + system?: string; + tools?: { name?: string }[]; + messages?: { role: string; content: unknown }[]; + }; + const system = body.system ?? ""; + if (system.includes("Markdown")) { + answerSawWindow = JSON.stringify(body.messages ?? []).includes("beta line"); + await route.fulfill({ + status: 200, + headers: { ...CORS, "content-type": "text/event-stream" }, + body: sse("Fetched answer [1]"), + }); + return; + } + // Native gather turn. + if (Array.isArray(body.tools)) { + if (body.tools.some((t) => t.name === "fetch_context")) sawFetchTool = true; + const msgs = body.messages ?? []; + const lastUser = [...msgs].reverse().find((m) => m.role === "user"); + const isToolResult = Array.isArray(lastUser?.content); + round += 1; + await route.fulfill({ + status: 200, + headers: { ...CORS, "content-type": "application/json" }, + body: isToolResult + ? toolUse("finish", {}) + : toolUse("fetch_context", { video: REAL_KEY, aroundSeconds: 5 }), + }); + return; + } + await route.fulfill({ + status: 200, + headers: { ...CORS, "content-type": "text/event-stream" }, + body: sse("DONE"), + }); + }); + await seedFetchHandoff(page); + await page.goto("/ask/"); + await keyIn(page, "Native tools"); + + await expect(page.getByText(/Grounded in 1 result/)).toBeVisible(); + // Expand mode → the fetch_context tool is offered and used. + await page.getByLabel("Answer only from these results").uncheck(); + await ask(page, "what surrounds the alpha moment"); + + await expect(page.getByText(/Fetched answer/)).toBeVisible(); + expect(sawFetchTool).toBe(true); + // The windowed line (50s "beta line") reached the answer's grounding. + expect(answerSawWindow).toBe(true); + expect(round).toBeGreaterThanOrEqual(2); // fetch round + finish round + }); + + test("user 'Load context' enriches the pin and persists (strict, no search)", async ({ + page, + }) => { + await installRoutes(page); + let gatherCalls = 0; + let answerSawWindow = false; + await page.route("https://api.anthropic.com/**", async (route) => { + if (route.request().method() === "OPTIONS") { + await route.fulfill({ status: 204, headers: CORS }); + return; + } + const body = route.request().postDataJSON() as { + system?: string; + messages?: { role: string; content: unknown }[]; + }; + const system = body.system ?? ""; + if (system.includes("Markdown")) { + answerSawWindow = JSON.stringify(body.messages ?? []).includes("beta line"); + await route.fulfill({ + status: 200, + headers: { ...CORS, "content-type": "text/event-stream" }, + body: sse("Answer over expanded excerpts [1]"), + }); + return; + } + gatherCalls += 1; // strict grounding → this must never fire + await route.fulfill({ + status: 200, + headers: { ...CORS, "content-type": "text/event-stream" }, + body: sse("SEARCH: alpha"), + }); + }); + await seedFetchHandoff(page); + await page.goto("/ask/"); + await keyIn(page); + + await expect(page.getByText(/Grounded in 1 result/)).toBeVisible(); + await page.getByRole("button", { name: /Show the 1 video/ }).click(); + // Starts with the single matched excerpt. + await expect(page.getByText(/1 excerpt/)).toBeVisible(); + await page.getByRole("button", { name: "context" }).click(); + // Windowing pulled in the neighbouring cue → the count grows. + await expect(page.getByText(/2 excerpts/)).toBeVisible(); + + // Persists across a reload (stored in the conversation). The key is only + // saved on send, so re-enter it after reloading. + await page.waitForTimeout(600); + await page.reload(); + await keyIn(page); + await page.getByRole("button", { name: /Show the 1 video/ }).click(); + await expect(page.getByText(/2 excerpts/)).toBeVisible(); + + // Strict answer carries the enriched excerpts, with no gather search. + await ask(page, "summarize these"); + await expect(page.getByText(/Answer over expanded excerpts/)).toBeVisible(); + expect(gatherCalls).toBe(0); + expect(answerSawWindow).toBe(true); + }); + + test("scripted transport degrades gracefully: no fetch tool, still answers", async ({ + page, + }) => { + await installRoutes(page); + let fetchToolEverOffered = false; + await page.route("https://api.anthropic.com/**", async (route) => { + if (route.request().method() === "OPTIONS") { + await route.fulfill({ status: 204, headers: CORS }); + return; + } + const body = route.request().postDataJSON() as { + system?: string; + tools?: { name?: string }[]; + messages?: { role: string; content: unknown }[]; + }; + if (Array.isArray(body.tools) && body.tools.some((t) => t.name === "fetch_context")) { + fetchToolEverOffered = true; + } + const system = body.system ?? ""; + const msgs = body.messages ?? []; + const lastUser = [...msgs].reverse().find((m) => m.role === "user"); + const lastText = typeof lastUser?.content === "string" ? lastUser.content : ""; + let text: string; + if (system.includes("Markdown")) text = "Scripted answer [1]"; + else if (lastText.includes('Results for "')) text = "DONE"; + else text = "SEARCH: alpha"; + await route.fulfill({ + status: 200, + headers: { ...CORS, "content-type": "text/event-stream" }, + body: sse(text), + }); + }); + await seedFetchHandoff(page); + await page.goto("/ask/"); + await keyIn(page, "Scripted"); + + await expect(page.getByText(/Grounded in 1 result/)).toBeVisible(); + await page.getByLabel("Answer only from these results").uncheck(); + await ask(page, "what surrounds the alpha moment"); + + await expect(page.getByText(/Scripted answer/)).toBeVisible(); + expect(fetchToolEverOffered).toBe(false); + }); + test("pinned strict grounding answers from the handed-off results, no search", async ({ page, }) => {